diff --git a/.github/workflows/architecture.yml b/.github/workflows/architecture.yml index ca07799c6..2b4ffc4bc 100644 --- a/.github/workflows/architecture.yml +++ b/.github/workflows/architecture.yml @@ -163,7 +163,7 @@ jobs: run: exit 1 cell-runtime-rustfs: - name: Server, Cell and LTX RustFS recovery + name: Server Cell RustFS recovery runs-on: ubuntu-latest timeout-minutes: 45 env: @@ -172,11 +172,6 @@ jobs: AWS_DEFAULT_REGION: us-east-1 AWS_REGION: us-east-1 AWS_EC2_METADATA_DISABLED: true - CRAB_LTX_TEST_BUCKET: crab-cell-runtime - CRAB_LTX_TEST_ENDPOINT: http://127.0.0.1:9000 - CRAB_CELL_TEST_BUCKET: crab-cell-runtime - CRAB_CELL_TEST_ENDPOINT: http://127.0.0.1:9000 - CRAB_CELL_TEST_PREFIX: cell-${{ github.run_id }}-${{ github.run_attempt }} CRAB_HTTP_CELL_TEST_BUCKET: crab-cell-runtime CRAB_HTTP_CELL_TEST_ENDPOINT: http://127.0.0.1:9000 CRAB_HTTP_CELL_TEST_PREFIX: server-${{ github.run_id }}-${{ github.run_attempt }} @@ -223,10 +218,10 @@ jobs: -v "$data_root:/data" \ "$RUSTFS_IMAGE" server /data --address :9000 for _ in $(seq 1 90); do - if aws --endpoint-url "$CRAB_CELL_TEST_ENDPOINT" \ + if aws --endpoint-url "$CRAB_HTTP_CELL_TEST_ENDPOINT" \ s3api list-buckets >/dev/null 2>&1; then - aws --endpoint-url "$CRAB_CELL_TEST_ENDPOINT" \ - s3api create-bucket --bucket "$CRAB_CELL_TEST_BUCKET" + aws --endpoint-url "$CRAB_HTTP_CELL_TEST_ENDPOINT" \ + s3api create-bucket --bucket "$CRAB_HTTP_CELL_TEST_BUCKET" exit 0 fi sleep 1 @@ -234,7 +229,7 @@ jobs: docker logs cell-runtime-rustfs >&2 || true exit 1 - - name: Verify canonical LTX and Cell source-loss recovery + - name: Verify server Cell source-loss recovery run: | set -euo pipefail run_test() { @@ -248,22 +243,6 @@ jobs: return 1 fi } - run_test cargo test -p crab-ltx --locked --features replica --test cell \ - cell::roots::lifecycle::exact_root_inventory_verifies_every_remote_dependency \ - -- --exact --nocapture - CRAB_CELL_TEST_PREFIX="${CRAB_CELL_TEST_PREFIX}/scheduler-discovery" \ - run_test cargo test -p crab-cell-runtime --locked --test fleet \ - fleet::takeover::rustfs_scheduler_membership_and_recovery_share_one_fresh_directory_scan \ - -- --ignored --exact --nocapture - run_test cargo test -p crab-cell-runtime --locked --test runtime \ - runtime::lifecycle::ownership::recovery::rustfs_source_loss_takeover_restores_exact_root_and_continues_publication \ - -- --ignored --exact --nocapture - CRAB_CELL_TEST_PREFIX="${CRAB_CELL_TEST_PREFIX}/read-replicas" \ - run_test cargo test -p crab-cell-runtime --locked --test runtime read_replica \ - -- --include-ignored --test-threads=6 --nocapture - run_test cargo test -p crab-cell-runtime --locked --lib \ - recovery::retention::tests::rustfs_maintenance_collection_preserves_live_and_pinned_graphs \ - -- --ignored --exact --nocapture run_test cargo test -p crab-http-server --locked --lib \ server::peer_e2e_tests::rustfs_public_collaboration_reaches_remote_owner_and_publishes_ltx \ -- --ignored --exact --nocapture diff --git a/.github/workflows/cell-coordination-model.yml b/.github/workflows/cell-coordination-model.yml deleted file mode 100644 index b586f5bd9..000000000 --- a/.github/workflows/cell-coordination-model.yml +++ /dev/null @@ -1,99 +0,0 @@ -name: Cell coordination model - -on: - pull_request: - paths: - - "crates/crab-cell-runtime/model/**" - - "crates/crab-cell-runtime/src/coordination.rs" - - "crates/crab-cell-runtime/src/coordination/sim.rs" - - ".github/workflows/cell-coordination-model.yml" - schedule: - - cron: "17 3 * * 1" - workflow_dispatch: - -permissions: - contents: read - -jobs: - fast: - name: TLC fast and negative matrix - runs-on: ubuntu-latest - timeout-minutes: 10 - steps: - - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 - with: - persist-credentials: false - - uses: actions/setup-java@b6effb05e454b25005698d916606bdc6ffcbf961 # v5 - with: - distribution: temurin - java-version: "17" - - name: Run TLC fast and negative checks - run: | - chmod +x crates/crab-cell-runtime/model/check.sh - crates/crab-cell-runtime/model/check.sh fast - crates/crab-cell-runtime/model/check.sh negative - - broad: - if: github.event_name != 'pull_request' - name: TLC broad bounded exploration - runs-on: ubuntu-latest - timeout-minutes: 30 - steps: - - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 - with: - persist-credentials: false - - uses: actions/setup-java@b6effb05e454b25005698d916606bdc6ffcbf961 # v5 - with: - distribution: temurin - java-version: "17" - - name: Run TLC broad check - run: | - set -euo pipefail - chmod +x crates/crab-cell-runtime/model/check.sh - mkdir -p "$RUNNER_TEMP/cell-coordination-broad" - { - printf 'source_revision=%s\n' "$(git rev-parse HEAD)" - printf 'config=CellCoordination.cfg\nmode=broad\n' - cat crates/crab-cell-runtime/model/toolchain.env - crates/crab-cell-runtime/model/check.sh broad - } 2>&1 | tee "$RUNNER_TEMP/cell-coordination-broad/tlc-broad.log" - - name: Upload broad model evidence - if: ${{ always() }} - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 - with: - name: cell-coordination-tlc-broad-${{ github.run_id }} - path: ${{ runner.temp }}/cell-coordination-broad - if-no-files-found: error - - simulation: - if: github.event_name != 'pull_request' - name: Deterministic simulator seed corpus - runs-on: ubuntu-latest - timeout-minutes: 15 - steps: - - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 - with: - persist-credentials: false - - uses: dtolnay/rust-toolchain@2c7215f132e9ebf062739d9130488b56d53c060c # master - with: - toolchain: stable - - name: Run the fixed simulator corpus - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-coordination-simulation-target - run: | - set -euo pipefail - mkdir -p "$RUNNER_TEMP/cell-coordination-simulation" - { - printf 'source_revision=%s\n' "$(git rev-parse HEAD)" - printf 'seed_range=0..511\nsteps_per_seed=256\n' - cargo test -p crab-cell-runtime --lib coordination::sim::broad_seed_corpus_is_replayable --locked -- --exact --nocapture - } 2>&1 | tee "$RUNNER_TEMP/cell-coordination-simulation/simulator.log" - grep -F 'test coordination::sim::broad_seed_corpus_is_replayable ... ok' \ - "$RUNNER_TEMP/cell-coordination-simulation/simulator.log" - - name: Upload simulator evidence - if: ${{ always() }} - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 - with: - name: cell-coordination-simulator-${{ github.run_id }} - path: ${{ runner.temp }}/cell-coordination-simulation - if-no-files-found: error diff --git a/.github/workflows/cell-property-qualification.yml b/.github/workflows/cell-property-qualification.yml deleted file mode 100644 index f2f16176e..000000000 --- a/.github/workflows/cell-property-qualification.yml +++ /dev/null @@ -1,41 +0,0 @@ -name: Cell property qualification - -# The property suites run in every pull request with a small case count. This -# job raises it an order of magnitude on a schedule, so the same suites can find -# rarer schedules without slowing review. proptest reads PROPTEST_CASES from the -# environment and prefers it over each suite's own case count. - -on: - schedule: - - cron: "43 5 * * 1" - workflow_dispatch: - -concurrency: - group: cell-property-qualification - cancel-in-progress: false - -permissions: - contents: read - -jobs: - deep: - name: Cell and LTX property suites at depth - runs-on: ubuntu-latest - timeout-minutes: 60 - env: - PROPTEST_CASES: "1000" - steps: - - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 - with: - persist-credentials: false - - uses: dtolnay/rust-toolchain@2c7215f132e9ebf062739d9130488b56d53c060c # master - with: - toolchain: stable - - name: Run the Cell runtime suites - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-property-target - run: cargo test -p crab-cell-runtime --features test-support --locked - - name: Run the LTX suites - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-property-target - run: cargo test -p crab-ltx --features replica --locked diff --git a/.github/workflows/cell-reference-compose.yml b/.github/workflows/cell-reference-compose.yml deleted file mode 100644 index 55475e06d..000000000 --- a/.github/workflows/cell-reference-compose.yml +++ /dev/null @@ -1,120 +0,0 @@ -name: Cell reference Compose smoke - -on: - pull_request: - paths: - - ".github/workflows/cell-reference-compose.yml" - - "Cargo.toml" - - "Cargo.lock" - - ".cargo/**" - - "crates/crab-cell-app/Cargo.toml" - - "crates/crab-cell-app/src/**" - - "crates/crab-cell-host/Cargo.toml" - - "crates/crab-cell-host/src/**" - - "crates/crab-cell-runtime/Cargo.toml" - - "crates/crab-cell-runtime/build.rs" - - "crates/crab-cell-runtime/docs/contracts/peer.proto" - - "crates/crab-cell-runtime/src/**" - - "crates/crab-ltx/Cargo.toml" - - "crates/crab-ltx/src/**" - - "crates/crab-storage/Cargo.toml" - - "crates/crab-storage/src/**" - - "crates/crab-cell-app/tests/reference_application/**" - - "crates/crab-cell-app/tests/reference_application.rs" - - "crates/crab-cell-app/qualification/**" - workflow_dispatch: - -permissions: - contents: read - -concurrency: - group: cell-reference-compose-${{ github.event.pull_request.number || github.ref }} - cancel-in-progress: true - -jobs: - smoke: - runs-on: ubuntu-24.04 - timeout-minutes: 45 - env: - CRAB_REFERENCE_PROJECT: cell-reference - CRAB_CELL_PERF_ITERATIONS: "30" - steps: - - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 - with: - persist-credentials: false - - name: Pin qualification source and host profile - run: | - set -euo pipefail - CRAB_REFERENCE_STATE="$RUNNER_TEMP/cell-reference" - printf 'CRAB_REFERENCE_STATE=%s\n' "$CRAB_REFERENCE_STATE" >> "$GITHUB_ENV" - mkdir -p "$CRAB_REFERENCE_STATE/source" "$CRAB_REFERENCE_STATE/target-linux" "$CRAB_REFERENCE_STATE/evidence" - # Capability-free container root needs write permission on this runner-owned mount. - chmod 1777 "$CRAB_REFERENCE_STATE/evidence" - git archive HEAD | tar -x -C "$CRAB_REFERENCE_STATE/source" - git rev-parse HEAD > "$CRAB_REFERENCE_STATE/evidence/source-revision.txt" - uname -a > "$CRAB_REFERENCE_STATE/evidence/host.txt" - docker info --format '{{json .}}' > "$CRAB_REFERENCE_STATE/evidence/docker-info.json" - - name: Build release test binary - run: | - set -euo pipefail - python3 -B -m unittest discover \ - -s crates/crab-cell-app/qualification -p 'test_*.py' - docker compose -p "$CRAB_REFERENCE_PROJECT" \ - -f "$CRAB_REFERENCE_STATE/source/crates/crab-cell-app/qualification/compose.yaml" \ - run --name cell-reference-build build \ - 2>&1 | tee "$CRAB_REFERENCE_STATE/evidence/build.log" - - name: Exercise three constrained nodes and real object storage - run: | - set -euo pipefail - compose() { - docker compose -p "$CRAB_REFERENCE_PROJECT" \ - -f "$CRAB_REFERENCE_STATE/source/crates/crab-cell-app/qualification/compose.yaml" "$@" - } - compose up -d node-0 node-1 node-2 - compose run --name cell-reference-driver driver - for node in node-0 node-1 node-2; do - container=$(compose ps -aq "$node") - test -n "$container" - test "$(docker wait "$container")" = 0 - done - # These correctness gates host their runtimes in one container; - # independent-process scaling remains the following qualification. - compose run --no-deps --name cell-reference-rollout \ - -e CRAB_CELL_PERF_PROCESS_ROOT=reference-compose-additive-rollout \ - driver rollout - compose run --no-deps --name cell-reference-entities \ - -e CRAB_CELL_TEST_PREFIX=reference-compose-entities \ - driver entities - compose stop - - name: Scale readers across 3, 5, 10 and 20 constrained nodes - run: | - python3 "$CRAB_REFERENCE_STATE/source/crates/crab-cell-app/qualification/scale.py" \ - --state "$CRAB_REFERENCE_STATE" --project "$CRAB_REFERENCE_PROJECT-scale" - - name: Measure writable entity traffic across 3, 5, 10 and 20 constrained nodes - run: | - python3 "$CRAB_REFERENCE_STATE/source/crates/crab-cell-app/qualification/scale.py" \ - --state "$CRAB_REFERENCE_STATE" --project "$CRAB_REFERENCE_PROJECT-entities" \ - --workload entities - - name: Retain reports and container resource proof - if: always() - run: | - set -euo pipefail - compose() { - docker compose -p "$CRAB_REFERENCE_PROJECT" \ - -f "$CRAB_REFERENCE_STATE/source/crates/crab-cell-app/qualification/compose.yaml" "$@" - } - compose logs --no-color > "$CRAB_REFERENCE_STATE/evidence/compose.log" - containers=$(compose ps -aq) - if [ -n "$containers" ]; then - mapfile -t container_ids <<< "$containers" - docker inspect "${container_ids[@]}" > "$CRAB_REFERENCE_STATE/evidence/containers.json" - fi - docker image ls -a --digests --no-trunc > "$CRAB_REFERENCE_STATE/evidence/images.txt" - compose stop - - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 - if: always() - with: - name: cell-reference-compose-${{ github.run_id }}-${{ github.run_attempt }} - path: ${{ runner.temp }}/cell-reference/evidence/ - if-no-files-found: error - retention-days: 7 diff --git a/.github/workflows/cell-runtime-protected-qualification.yml b/.github/workflows/cell-runtime-protected-qualification.yml index c3571d30f..38774fc2c 100644 --- a/.github/workflows/cell-runtime-protected-qualification.yml +++ b/.github/workflows/cell-runtime-protected-qualification.yml @@ -74,10 +74,6 @@ jobs: exit 1 } test -z "$(git status --porcelain)" - cargo_target_dir="$RUNNER_TEMP/crab-cell-runtime-protected-target" - mkdir -p "$cargo_target_dir" - test -d "$cargo_target_dir" - test -w "$(dirname "$cargo_target_dir")" test -x "$QUALIFICATION_HARNESS" || { echo "::error::Protected harness is missing: $QUALIFICATION_HARNESS" >&2 echo "Install the provider/Kubernetes harness on the protected runner; no local or synthetic fallback is accepted." >&2 @@ -148,15 +144,15 @@ jobs: "fault-gcs-v1|qualification-matrix-fault-gcs.json" "fault-azure-v1|qualification-matrix-fault-azure.json" ) - export CARGO_TARGET_DIR="$RUNNER_TEMP/crab-cell-runtime-protected-target" + cellule_source="$(python3 crab/scripts/cellule-qualification.py source-root)" for requirement in "${required_profiles[@]}"; do IFS='|' read -r profile_name _matrix_name <<< "$requirement" profile="$protected_dir/$profile_name.json" - tracked="crates/crab-cell-runtime/qualification/profiles/$profile_name.json" + tracked="$cellule_source/crates/cellule-runtime/qualification/profiles/$profile_name.json" test -f "$profile" && test ! -L "$profile" cmp -- "$profile" "$tracked" done - cargo run --quiet --locked -p crab-cell-runtime --bin qualification_receipt -- \ + python3 crab/scripts/cellule-qualification.py run \ verify-protected-bundle "$protected_dir" "$SOURCE_SHA" "$IMAGE_DIGEST" \ "$QUALIFICATION_SIGNER" printf '%s\n' "$SOURCE_SHA" > "$evidence_root/source-revision" diff --git a/.github/workflows/cell-runtime-qualification-contract.yml b/.github/workflows/cell-runtime-qualification-contract.yml deleted file mode 100644 index e91db15a6..000000000 --- a/.github/workflows/cell-runtime-qualification-contract.yml +++ /dev/null @@ -1,302 +0,0 @@ -name: Cell runtime qualification contract - -on: - pull_request: - paths: - - "crates/crab-cell-runtime/**" - - "crates/crab-cell-app/**" - - "crates/crab-cell-host/**" - - "crates/crab-http-server/**" - - "crates/crab-ltx/**" - - "packages/ui/**" - - "Cargo.toml" - - "Cargo.lock" - - "crab/scripts/check-architecture-gates.py" - - "crab/scripts/check-cell-ltx-layout.py" - - "crab/scripts/check-policy-entry-points.py" - - "crates/crab-http-server/tests/qualify_compose_cluster.sh" - - ".github/workflows/http-server-container.yml" - - ".github/workflows/http-server-release.yml" - - ".github/workflows/cell-coordination-model.yml" - - ".github/workflows/cell-runtime-qualification-contract.yml" - - ".github/workflows/cell-runtime-protected-qualification.yml" - push: - branches: [main] - paths: - - "crates/crab-cell-runtime/**" - - "crates/crab-cell-app/**" - - "crates/crab-cell-host/**" - - "crates/crab-http-server/**" - - "crates/crab-ltx/**" - - "packages/ui/**" - - "Cargo.toml" - - "Cargo.lock" - - "crab/scripts/check-architecture-gates.py" - - "crab/scripts/check-cell-ltx-layout.py" - - "crab/scripts/check-policy-entry-points.py" - - "crates/crab-http-server/tests/qualify_compose_cluster.sh" - - ".github/workflows/http-server-container.yml" - - ".github/workflows/http-server-release.yml" - - ".github/workflows/cell-coordination-model.yml" - - ".github/workflows/cell-runtime-qualification-contract.yml" - - ".github/workflows/cell-runtime-protected-qualification.yml" - -jobs: - contract: - name: Signed receipt and canonical contract checks - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6 - - name: Install contract validation tools - run: | - sudo apt-get update - sudo apt-get install --no-install-recommends -y protobuf-compiler sqlite3 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 22 - cache: npm - cache-dependency-path: packages/ui/package-lock.json - - name: Build embedded repository frontend - run: | - npm ci --prefix packages/ui - npm run build --prefix packages/ui - - uses: dtolnay/rust-toolchain@2c7215f132e9ebf062739d9130488b56d53c060c # master - with: - toolchain: stable - - name: Validate architecture boundary - run: python3 crab/scripts/check-architecture-gates.py - - name: Validate executable design contracts - run: node crates/crab-cell-runtime/docs/validate.mjs - - name: Verify tracked qualification profiles - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-runtime-qualification-target - run: | - set -euo pipefail - root="${RUNNER_TEMP}/qualification-profiles" - mkdir -p "$root" - for profile in \ - pr-contract local-provider scale fault fault-s3 fault-gcs fault-azure \ - provider provider-s3 provider-gcs provider-azure compatibility; do - generated="$root/${profile}.json" - cargo run --quiet --locked -p crab-cell-runtime --bin qualification_receipt -- \ - profile "$generated" "$profile" - jq -S . "$generated" > "$root/${profile}.generated" - jq -S . "crates/crab-cell-runtime/qualification/profiles/${profile}-v1.json" \ - > "$root/${profile}.tracked" - cmp "$root/${profile}.generated" "$root/${profile}.tracked" - done - - name: Run production hot-path preflight guards - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-runtime-qualification-target - run: cargo test -p crab-cell-runtime --test qualification --locked - - name: Verify application and host boundaries - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-runtime-qualification-target - run: cargo test -p crab-cell-app -p crab-cell-host --locked - - name: Verify public CellNode typed primitive workload - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-runtime-qualification-target - run: >- - cargo test -p crab-http-server - --test public_cell_host_application - --test public_cell_qualification - --test public_cell_takeover - --test public_cell_process_fault - --test public_cell_response_loss - --test public_cell_retry - --test public_cell_activity_retry - --test public_cell_cancellation - --test public_cell_data_cancellation - --test public_cell_activity_cancellation - --test public_cell_effect_delivery_cancellation - --test public_cell_effect_delivery_expiry - --test public_cell_lease_expiry - --locked - - name: Keep Compose receipt schema gate synchronized - shell: bash - run: | - set -euo pipefail - harness_version="$(sed -n \ - 's/^[[:space:]]*version:[[:space:]]*\([0-9][0-9]*\),$/\1/p' \ - crates/crab-http-server/tests/qualify_compose_cluster.sh | head -1)" - test -n "$harness_version" - test "$harness_version" = 6 - grep -F 'server-d' crates/crab-http-server/tests/qualify_compose_cluster.sh - grep -F 'fallback:' crates/crab-http-server/tests/qualify_compose_cluster.sh - grep -F -- "validate-cluster \"\${RUNNER_TEMP}/crab-http-server-cluster-receipt.json\"" \ - .github/workflows/http-server-container.yml - - name: Keep release qualification bound to the promoted image - shell: bash - run: | - set -euo pipefail - grep -F 'image_ref:' .github/workflows/http-server-container.yml - grep -F 'image_digest:' .github/workflows/http-server-container.yml - grep -F "docker pull \"\$INPUT_IMAGE_REF\"" .github/workflows/http-server-container.yml - grep -F 'needs.candidate.outputs.image_ref' \ - .github/workflows/http-server-release.yml - grep -F 'needs.candidate.outputs.image_digest' \ - .github/workflows/http-server-release.yml - grep -F 'needs: [prepare, candidate]' .github/workflows/http-server-release.yml - grep -F 'docker buildx imagetools create' .github/workflows/http-server-release.yml - grep -F "test \"\$(cat \"\$evidence_dir/image-digest\")\" = \"\$IMAGE_DIGEST\"" \ - .github/workflows/http-server-release.yml - if grep -F 'Build and publish the image' .github/workflows/http-server-release.yml; then - echo "release workflow must not rebuild after qualification" >&2 - exit 1 - fi - - name: Keep candidate reuse bound to the original tag run - shell: bash - run: | - set -euo pipefail - workflow=.github/workflows/http-server-release.yml - grep -F 'candidate_run_id:' "$workflow" - grep -F 'steps.build.outputs.digest || steps.reuse.outputs.digest' "$workflow" - grep -F 'name: http-server-candidate-' "$workflow" - grep -F 'github.run_id' "$workflow" - grep -F 'github.run_attempt' "$workflow" - grep -F '.event == "push"' "$workflow" - # Compare literal source fragments without evaluating their variables. - while IFS= read -r fragment; do - grep -F -- "$fragment" "$workflow" - done <<'SOURCE_FRAGMENTS' - gh run download "$CANDIDATE_RUN_ID" - .headSha == $source - .databaseId == $run - test "$(cat "$candidate_dir/source-revision")" = "$SOURCE_SHA" - "$image_ref" != "$expected_ref" - Candidate tag moved from $IMAGE_DIGEST to $actual. - "$GITHUB_SHA" != "$source_sha" - "$GITHUB_REF" != "refs/tags/$tag" - SOURCE_FRAGMENTS - grep -A2 '^ image:' "$workflow" | grep -F "if: github.event_name == 'workflow_dispatch'" - test "$(grep -cF ' TAG: ' "$workflow")" -eq 3 - grep -F 'inputs.tag || github.ref_name' "$workflow" - - name: Keep protected evidence handoff explicit and exact-source bound - shell: bash - run: | - set -euo pipefail - # Assertions below match literal workflow text, so the dollar signs - # must reach grep verbatim rather than expand here. - test -f .github/workflows/cell-runtime-protected-qualification.yml - grep -F 'name: Cell runtime protected qualification' \ - .github/workflows/cell-runtime-protected-qualification.yml - grep -F 'runs-on: [self-hosted, linux, crab-cell-runtime-protected]' \ - .github/workflows/cell-runtime-protected-qualification.yml - grep -F '/opt/crab/bin/crab-cell-runtime-protected-qualifier' \ - .github/workflows/cell-runtime-protected-qualification.yml - grep -F "test \"\$GITHUB_SHA\" = \"\$SOURCE_REF\"" \ - .github/workflows/cell-runtime-protected-qualification.yml - # GitHub expands dollar-brace expressions before the shell runs, so - # the artifact-name contract is asserted from its literal pieces. - grep -F 'name: cell-runtime-protected-' \ - .github/workflows/cell-runtime-protected-qualification.yml - grep -F 'github.run_id' \ - .github/workflows/cell-runtime-protected-qualification.yml - grep -F 'github.run_attempt' \ - .github/workflows/cell-runtime-protected-qualification.yml - grep -F "verify-protected-bundle \"\$protected_dir\"" \ - .github/workflows/cell-runtime-protected-qualification.yml - grep -F 'cell_runtime_evidence_run_id:' .github/workflows/http-server-release.yml - grep -F 'cell_runtime_evidence_artifact:' .github/workflows/http-server-release.yml - grep -F 'actions: read' .github/workflows/http-server-release.yml - grep -F 'Cell runtime protected qualification' .github/workflows/http-server-release.yml - grep -F "if [ \"\$event\" != \"workflow_dispatch\" ]" .github/workflows/http-server-release.yml - grep -F "if [ \"\$evidence_sha\" != \"\$SOURCE_SHA\" ]" .github/workflows/http-server-release.yml - grep -F "cell-runtime-protected-\${run_id}-\${attempt}" .github/workflows/http-server-release.yml - grep -F "verify-protected-bundle \"\$protected_dir\"" .github/workflows/http-server-release.yml - grep -F 'exactly one protected/' .github/workflows/http-server-release.yml - release_workflow=.github/workflows/http-server-release.yml - protected_line="$(grep -nF ' - name: Require protected Cell runtime qualification matrix' "$release_workflow" | cut -d: -f1)" - promote_line="$(grep -nF ' - name: Promote the qualified candidate image' "$release_workflow" | cut -d: -f1)" - if [ -z "$protected_line" ] || [ -z "$promote_line" ] || [ "$protected_line" -ge "$promote_line" ]; then - echo "protected Cell evidence must verify before immutable image promotion" >&2 - exit 1 - fi - - name: Keep scheduled coordination replay nonempty and failure-sensitive - shell: bash - run: | - set -euo pipefail - workflow=.github/workflows/cell-coordination-model.yml - grep -F 'crates/crab-cell-runtime/src/coordination/sim.rs' "$workflow" - grep -F 'cargo test -p crab-cell-runtime --lib coordination::sim::broad_seed_corpus_is_replayable --locked -- --exact --nocapture' "$workflow" - grep -F 'test coordination::sim::broad_seed_corpus_is_replayable ... ok' "$workflow" - test "$(grep -cF 'set -euo pipefail' "$workflow")" -ge 2 - - name: Verify receipt, placement, pressure, and resource contracts - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-runtime-qualification-target - run: | - set -euo pipefail - for filter in qualification placement pressure resource; do - cargo test -p crab-cell-runtime --features test-support --lib "${filter}" --locked - cargo test -p crab-cell-runtime --features test-support \ - --test fleet --test qualification "${filter}" --locked - done - cargo test -p crab-cell-runtime --bin qualification_receipt --locked - - name: Verify the Cell and LTX crate layout - run: python3 crab/scripts/check-cell-ltx-layout.py - - name: Verify the Cell runtime policy entry points - run: python3 crab/scripts/check-policy-entry-points.py - - name: Verify documented Rust snippets parse - run: python3 crab/scripts/check-doc-rust-fences.py - - name: Verify the process-test-support suites - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-runtime-qualification-target - run: | - set -euo pipefail - cargo test -p crab-cell-runtime --features test-support --locked - cargo test -p crab-cell-app --locked - cargo test -p crab-cell-host --locked - - name: Verify canonical LTX contracts - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-runtime-qualification-target - run: cargo test -p crab-ltx --features replica --locked - - name: Deny rustdoc warnings - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-runtime-qualification-target - RUSTDOCFLAGS: -D warnings - run: >- - cargo doc - -p crab-cell-runtime -p crab-ltx -p crab-cell-app -p crab-cell-host - --no-deps --locked - - name: Smoke-test the complete qualification matrix in a fresh process - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-cell-runtime-qualification-target - run: | - set -euo pipefail - root="${RUNNER_TEMP}/qualification-matrix-contract" - mkdir -p "$root/receipts" "$root/artifacts" - source_revision=0123456789abcdef0123456789abcdef01234567 - image_digest=sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa - fixture="$root/fixture.json" - profile="$root/pr-contract-profile.json" - jq -cn '{schema: 1, kind: "qualification-matrix-contract-fixture"}' > "$fixture" - cargo run --quiet --locked -p crab-cell-runtime --bin qualification_receipt -- \ - profile "$profile" pr-contract - canonical_workload="$root/pr-contract-workload.json" - cargo run --quiet --locked -p crab-cell-runtime --bin qualification_receipt -- \ - workload "$canonical_workload" "$profile" 7 - cargo run --quiet --locked -p crab-cell-runtime --bin qualification_receipt -- \ - verify-workload "$canonical_workload" "$profile" - for workload in protocol storage publication warm-path churn fleet failover primitives accounting compatibility; do - artifact_dir="$root/artifacts/$workload" - mkdir -p "$artifact_dir" - artifact="$artifact_dir/primary.json" - receipt="$root/receipts/${workload}.json" - if [ "$workload" = primitives ]; then - cp "$canonical_workload" "$artifact" - else - cp "$fixture" "$artifact" - fi - cargo run --quiet --locked -p crab-cell-runtime --bin qualification_receipt -- \ - emit "$receipt" "$source_revision" "$image_digest" "$artifact" \ - github-actions "$workload" none "$profile" - done - cargo run --quiet --locked -p crab-cell-runtime --bin qualification_receipt -- \ - manifest "$root/qualification-matrix.json" "$root" - ( - cd "$root" - cargo run --quiet --locked --manifest-path "${GITHUB_WORKSPACE}/Cargo.toml" \ - -p crab-cell-runtime --bin qualification_receipt -- \ - verify-matrix qualification-matrix.json "$source_revision" "$image_digest" \ - pr-contract-profile.json - ) diff --git a/.github/workflows/crab-ltx-fuzz.yml b/.github/workflows/crab-ltx-fuzz.yml deleted file mode 100644 index dbf34fcf5..000000000 --- a/.github/workflows/crab-ltx-fuzz.yml +++ /dev/null @@ -1,82 +0,0 @@ -name: crab-ltx fuzz - -on: - schedule: - # A nightly search on top of the per-pull-request smoke run. - - cron: "17 5 * * *" - workflow_dispatch: - pull_request: - branches: [main] - paths: - - "crates/crab-ltx/**" - - ".github/workflows/crab-ltx-fuzz.yml" - -permissions: - contents: read - -concurrency: - group: crab-ltx-fuzz-${{ github.event.pull_request.number || github.ref }} - cancel-in-progress: true - -jobs: - fuzz: - name: Decoder fuzz smoke - runs-on: ubuntu-latest - # Installing cargo-fuzz and building five targets dominates the runtime. - timeout-minutes: 45 - env: - CARGO_TERM_COLOR: always - RUST_BACKTRACE: 1 - # Every pull request proves the targets run; the schedule searches longer. - FUZZ_SECONDS: ${{ github.event_name == 'schedule' && '900' || '90' }} - steps: - - name: Checkout - uses: actions/checkout@v4 - - - name: Set up nightly Rust - uses: dtolnay/rust-toolchain@nightly - - - name: Cache fuzz build artifacts - uses: Swatinem/rust-cache@v2 - with: - workspaces: "crates/crab-ltx/fuzz -> target" - cache-on-failure: true - - - name: Install cargo-fuzz - run: cargo install cargo-fuzz --locked - - - name: Seed the corpus from the external vectors - run: | - mkdir -p "${RUNNER_TEMP}/corpus" "${RUNNER_TEMP}/artifacts" - cp crates/crab-ltx/tests/vectors/*.ltx "${RUNNER_TEMP}/corpus/" - - - name: Replay the external vectors - # The stable-toolchain replay is the deterministic gate; the fuzz run - # below searches beyond it. - run: cargo test -p crab-ltx --features replica --locked --test ltx vectors - - - name: Fuzz every decoder - working-directory: crates/crab-ltx - run: | - set -euo pipefail - { - printf 'source_revision=%s\n' "$(git rev-parse HEAD)" - printf 'fuzz_seconds_per_target=%s\n' "$FUZZ_SECONDS" - printf 'targets=ltx root directory bundle node_frame\n' - } > "${RUNNER_TEMP}/artifacts/run-context.txt" - for target in ltx root directory bundle node_frame; do - cargo +nightly fuzz run "${target}" -- \ - -artifact_prefix="${RUNNER_TEMP}/artifacts/${target}-" \ - -max_total_time="${FUZZ_SECONDS}" \ - "${RUNNER_TEMP}/corpus" \ - 2>&1 | tee "${RUNNER_TEMP}/artifacts/${target}.log" - done - - - name: Upload fuzz replay evidence - if: ${{ always() && (github.event_name != 'pull_request' || failure()) }} - uses: actions/upload-artifact@v4 - with: - name: crab-ltx-fuzz-evidence-${{ github.run_id }} - path: ${{ runner.temp }}/artifacts - if-no-files-found: ignore - retention-days: 30 diff --git a/.github/workflows/http-server-container.yml b/.github/workflows/http-server-container.yml index 64b0ac0b1..3b53278b4 100644 --- a/.github/workflows/http-server-container.yml +++ b/.github/workflows/http-server-container.yml @@ -811,9 +811,7 @@ jobs: with: toolchain: stable - name: Build the cluster receipt validator - env: - CARGO_TARGET_DIR: ${{ runner.temp }}/crab-http-server-receipt-target - run: cargo build --locked -p crab-cell-runtime --bin qualification_receipt + run: python3 crab/scripts/cellule-qualification.py build - name: Exercise Compose startup, restore, crash recovery, and restart env: CRAB_HTTP_SERVER_IMAGE: crab-http-server:test @@ -943,7 +941,6 @@ jobs: CRAB_HTTP_SERVER_IMAGE=crab-http-server:test \ CRAB_HTTP_CLUSTER_BUILD=false \ CRAB_HTTP_CLUSTER_VALIDATE=true \ - CRAB_HTTP_CLUSTER_CARGO_TARGET_DIR="${RUNNER_TEMP}/crab-http-server-receipt-target" \ crates/crab-http-server/tests/qualify_compose_cluster.sh \ > "${RUNNER_TEMP}/crab-http-server-cluster-receipt.json" source_revision="$(git rev-parse HEAD)" @@ -951,8 +948,7 @@ jobs: if [ -n '${{ inputs.image_ref }}' ]; then receipt_mode=release fi - CARGO_TARGET_DIR="${RUNNER_TEMP}/crab-http-server-receipt-target" \ - cargo run --quiet --locked -p crab-cell-runtime --bin qualification_receipt -- \ + python3 crab/scripts/cellule-qualification.py run \ validate-cluster "${RUNNER_TEMP}/crab-http-server-cluster-receipt.json" \ "$source_revision" "${{ steps.image.outputs.digest }}" "$receipt_mode" - name: Prepare exact-source qualification evidence diff --git a/.github/workflows/http-server-release.yml b/.github/workflows/http-server-release.yml index 2fff6d646..ab31925f1 100644 --- a/.github/workflows/http-server-release.yml +++ b/.github/workflows/http-server-release.yml @@ -467,16 +467,12 @@ jobs: release_dir="${RUNNER_TEMP}/crab-http-server-release" mkdir -p "$release_dir" - export CARGO_TARGET_DIR="${RUNNER_TEMP}/crab-cell-runtime-target" - cargo run --locked --quiet -p crab-cell-runtime --bin qualification_receipt \ - -- validate-cluster "$evidence_dir/cluster-receipt.json" \ + python3 crab/scripts/cellule-qualification.py run validate-cluster "$evidence_dir/cluster-receipt.json" \ "$SOURCE_SHA" "$IMAGE_DIGEST" release - cargo run --locked --quiet -p crab-cell-runtime --bin qualification_receipt \ - -- emit "$release_dir/cell-runtime-qualification.json" \ + python3 crab/scripts/cellule-qualification.py run emit "$release_dir/cell-runtime-qualification.json" \ "$SOURCE_SHA" "$IMAGE_DIGEST" "$evidence_dir/cluster-receipt.json" \ github-actions http-server-release none - cargo run --locked --quiet -p crab-cell-runtime --bin qualification_receipt \ - -- verify "$release_dir/cell-runtime-qualification.json" \ + python3 crab/scripts/cellule-qualification.py run verify "$release_dir/cell-runtime-qualification.json" \ "$SOURCE_SHA" "$IMAGE_DIGEST" "$evidence_dir/cluster-receipt.json" - name: Require protected Cell runtime qualification matrix @@ -494,7 +490,7 @@ jobs: echo "::error::CRAB_CELL_RUNTIME_QUALIFICATION_SIGNER must be a pinned 32-byte lowercase Ed25519 public key." exit 1 fi - export CARGO_TARGET_DIR="${RUNNER_TEMP}/crab-cell-runtime-target" + cellule_source="$(python3 crab/scripts/cellule-qualification.py source-root)" required_profiles=( "local-provider-v1|qualification-matrix-local-provider.json" "scale-v1|qualification-matrix.json" @@ -510,7 +506,7 @@ jobs: IFS='|' read -r profile_name matrix_name <<< "$requirement" matrix="$protected_dir/$matrix_name" profile="$protected_dir/$profile_name.json" - tracked_profile="crates/crab-cell-runtime/qualification/profiles/$profile_name.json" + tracked_profile="$cellule_source/crates/cellule-runtime/qualification/profiles/$profile_name.json" test -s "$matrix" || { echo "::error::Protected Cell runtime matrix is required at $matrix; fixture receipts cannot qualify a release." exit 1 @@ -532,8 +528,7 @@ jobs: exit 1 } done - cargo run --locked --quiet -p crab-cell-runtime --bin qualification_receipt \ - -- verify-protected-bundle "$protected_dir" \ + python3 crab/scripts/cellule-qualification.py run verify-protected-bundle "$protected_dir" \ "$SOURCE_SHA" "$IMAGE_DIGEST" "$QUALIFICATION_SIGNER" - name: Package verified Cell runtime qualification evidence diff --git a/.github/workflows/rust.yml b/.github/workflows/rust.yml index b90a6f5d8..dcce516e8 100644 --- a/.github/workflows/rust.yml +++ b/.github/workflows/rust.yml @@ -159,24 +159,11 @@ jobs: npm test --prefix packages/ui npm run build --prefix packages/ui - - name: Check HTTP server and local LTX engine + - name: Check HTTP server run: | - cargo clippy -p crab-http-server -p crab-ltx --all-targets --locked -- -D warnings + cargo clippy -p crab-http-server --all-targets --locked -- -D warnings cargo test -p crab-http-server --example qualify_http_load --locked - - name: Check the Cell runtime, application, and host crates - run: | - # The HTTP server and LTX lint above does not cover these crates, and the - # runtime's `test-support` feature changes which items compile, so both - # configurations are linted. - cargo clippy -p crab-cell-runtime -p crab-cell-app -p crab-cell-host --all-targets --locked -- -D warnings - cargo clippy -p crab-cell-runtime --all-targets --features test-support --locked -- -D warnings - - - name: Check LTX replica feature - run: | - cargo clippy -p crab-ltx --all-targets --features replica --locked -- -D warnings - cargo test -p crab-ltx --features replica --locked - - name: Check formatting run: cargo fmt --all -- --check diff --git a/AGENTS.md b/AGENTS.md index b32e27b7a..fd2022bf0 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -32,7 +32,7 @@ CrabBuild/ └── .codex/ Repository-local Codex skills ``` -Cargo workspace: 23 members — 22 crates under `crates/`, plus `crab`. +Cargo workspace: 25 members — 24 crates under `crates/`, plus `crab`. `crates/crab-sdk` is an unpublished SDK under construction; its delivery gates live in `crab/docs/architecture/crab-sdk.md`. There is no desktop application or Python package; desktop material under `packages/web/` is documentation and diff --git a/Cargo.lock b/Cargo.lock index 75f227b3b..0ada278fb 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2290,88 +2290,6 @@ dependencies = [ "tracing", ] -[[package]] -name = "crab-cell-app" -version = "0.1.0" -dependencies = [ - "async-trait", - "blake3", - "crab-cell-host", - "crab-cell-runtime", - "crab-ltx", - "crab-storage", - "ed25519-dalek", - "futures-util", - "object_store", - "tempfile", - "tokio", - "tokio-util", - "tracing-subscriber", -] - -[[package]] -name = "crab-cell-host" -version = "0.1.0" -dependencies = [ - "crab-cell-app", - "crab-cell-runtime", - "futures-util", - "tempfile", - "tokio", - "tokio-util", - "tracing", - "uuid", -] - -[[package]] -name = "crab-cell-peer-http" -version = "0.1.0" -dependencies = [ - "axum 0.8.9", - "crab-cell-runtime", - "ed25519-dalek", - "futures-util", - "http 1.5.0", - "reqwest 0.12.28", - "rustls 0.23.40", - "rustls-pemfile", - "sha2 0.10.9", - "thiserror 2.0.18", - "tokio", - "tokio-rustls 0.26.4", - "tracing", - "url", - "x509-cert", -] - -[[package]] -name = "crab-cell-runtime" -version = "0.1.0" -dependencies = [ - "async-trait", - "blake3", - "bytes", - "crab-ltx", - "crab-storage", - "ed25519-dalek", - "fs4", - "futures-util", - "object_store", - "proptest", - "prost", - "prost-build", - "protoc-bin-vendored", - "rand 0.9.4", - "rusqlite", - "serde", - "serde_json", - "tempfile", - "thiserror 2.0.18", - "tokio", - "tokio-util", - "tracing", -] - [[package]] name = "crab-coordination" version = "0.1.0" @@ -2529,28 +2447,6 @@ dependencies = [ "tracing", ] -[[package]] -name = "crab-ltx" -version = "0.1.0" -dependencies = [ - "async-trait", - "blake3", - "bytes", - "crab-storage", - "crc-fast", - "futures-util", - "lz4_flex 0.11.6", - "object_store", - "proptest", - "rusqlite", - "serde", - "serde_json", - "tempfile", - "thiserror 2.0.18", - "tokio", - "tokio-util", -] - [[package]] name = "crab-metadata" version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index e33689d7a..7490a0038 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -7,16 +7,11 @@ members = [ "crates/crab-cache", "crates/crab-cache-store", "crates/crab-cache-server", - "crates/crab-cell-runtime", - "crates/crab-cell-app", - "crates/crab-cell-host", - "crates/crab-cell-peer-http", "crates/crab-coordination", "crates/crab-diff", "crates/crab-git", "crates/crab-http-server", "crates/crab-lfs", - "crates/crab-ltx", "crates/crab-metadata", "crates/crab-read", "crates/crab-remote", @@ -50,16 +45,11 @@ crab-auth-store = { version = "=0.1.0", path = "crates/crab-auth-store", default crab-cache = { version = "=0.1.0", path = "crates/crab-cache", default-features = false } crab-cache-server = { version = "=1.2.4", path = "crates/crab-cache-server", default-features = false } crab-cache-store = { version = "=0.1.0", path = "crates/crab-cache-store", default-features = false } -crab-cell-runtime = { version = "=0.1.0", path = "crates/crab-cell-runtime", default-features = false } -crab-cell-app = { version = "=0.1.0", path = "crates/crab-cell-app", default-features = false } -crab-cell-host = { version = "=0.1.0", path = "crates/crab-cell-host", default-features = false } -crab-cell-peer-http = { version = "=0.1.0", path = "crates/crab-cell-peer-http", default-features = false } crab-coordination = { version = "=0.1.0", path = "crates/crab-coordination", default-features = false } crab-diff = { version = "=0.1.0", path = "crates/crab-diff", default-features = false } crab-git = { version = "=0.1.0", path = "crates/crab-git", default-features = false } crab-http-server = { version = "=0.1.0", path = "crates/crab-http-server", default-features = false } crab-lfs = { version = "=0.1.0", path = "crates/crab-lfs", default-features = false } -crab-ltx = { version = "=0.1.0", path = "crates/crab-ltx", default-features = false } crab-metadata = { version = "=0.1.0", path = "crates/crab-metadata", default-features = false } crab-read = { version = "=0.1.0", path = "crates/crab-read", default-features = false } crab-remote = { version = "=0.1.0", path = "crates/crab-remote", default-features = false } diff --git a/advisor-plans/036-cell-read-replicas-and-fenced-promotion.md b/advisor-plans/036-cell-read-replicas-and-fenced-promotion.md index a8d94894c..62f4a7aa5 100644 --- a/advisor-plans/036-cell-read-replicas-and-fenced-promotion.md +++ b/advisor-plans/036-cell-read-replicas-and-fenced-promotion.md @@ -311,7 +311,7 @@ reference host uses 32 MiB so old and replacement snapshots remain charged during refresh. The same generated proof also passed on three constrained Compose nodes with GA RustFS: each node had 1 CPU, 1 GiB, and zero swap; all exited zero and withdrew renewed sessions. The -[source-bound receipt](../crates/crab-cell-app/performance/2026-09-27-replica-compose.md) +[source-bound receipt](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-app/performance/2026-09-27-replica-compose.md) records the 7/6 successful reader split and cgroup evidence. Protected capacity and fault gates remain open. See the reference application's `PERFORMANCE.md` commands. @@ -325,7 +325,7 @@ fixture owner hint; general application-host recruitment was separate work. Shutdown cancels provider waits before joining the activation lane, and a retained manager cannot reopen after drain. A stalled-store regression covers this ordering. Product authentication stays in the server. -The [three-container GA RustFS receipt](../crates/crab-cell-app/performance/2026-09-27-host-readers-compose.md) +The [three-container GA RustFS receipt](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-app/performance/2026-09-27-host-readers-compose.md) binds these automatic lifecycle checks to source and binary hashes, one CPU / 1 GiB / zero swap per node, and successful renewed-session withdrawal. It does not qualify performance improvement or a supported capacity. @@ -335,7 +335,7 @@ reference hosts install one bounded, cancellable loop across their compiled application namespaces. Activation and status share the signed runtime peer client/dispatcher; receiver admission still checks current owner, policy, membership and resources. The -[recruitment receipt](../crates/crab-cell-app/performance/2026-09-27-reader-recruitment.md) +[recruitment receipt](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-app/performance/2026-09-27-reader-recruitment.md) proves automatic placement, refresh, eviction and drain on three constrained Compose nodes. A separate native three-to-five-process run killed a selected reader, automatically replaced it, and verified twelve generated reads at the @@ -345,7 +345,7 @@ limits; constrained application capacity at 5/10/20 nodes, sustained freshness and throughput, owner loss during arrivals, and protected gates remain open. The reference application now also has a constrained -[3/5/10/20-node reader run](../crates/crab-cell-app/performance/2026-09-27-reader-scaling.md). +[3/5/10/20-node reader run](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-app/performance/2026-09-27-reader-scaling.md). It verifies thirty generated queries per selected reader, two acknowledged writes per stage, automatic refresh, an actual reader-container kill at five nodes, replacement without changing writer ownership, target-zero eviction, @@ -358,7 +358,7 @@ took 34.830 s in one sample and needs latency investigation. Sustained capacity/freshness, many-Cell admission, arrivals during faults and protected qualification remain open. -The [activation-expiry follow-up](../crates/crab-cell-app/performance/2026-09-27-reader-expiry.md) +The [activation-expiry follow-up](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-app/performance/2026-09-27-reader-expiry.md) reproduced a pending hint holding recruitment after its selected boot expired. The shared peer client now rechecks that session at its observed expiry while preserving the same request across valid renewal. The unchanged constrained @@ -367,7 +367,7 @@ earlier 34.830 s. Exact generated reads, owner identity and survivor drain remained intact. Refresh still took roughly five seconds, and neither sample establishes a supported recovery or freshness limit. -The [publication-notification follow-up](../crates/crab-cell-app/performance/2026-09-27-reader-publication.md) +The [publication-notification follow-up](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-app/performance/2026-09-27-reader-publication.md) passed the same constrained 3/5/10/20-node profile. Second-write readiness was 110–152 ms, using bounded advisory notifications and the existing signed recruitment path. The run also exposed and fixed raw handler time moving behind diff --git a/crab/Makefile b/crab/Makefile index f6a473d3d..7a3d1536f 100644 --- a/crab/Makefile +++ b/crab/Makefile @@ -355,12 +355,6 @@ crate-interface-check: crate-behavior-check: @$(PYTHON) scripts/check-crate-behavior.py --cargo "$(CARGO)" -cell-ltx-layout-check: - @$(PYTHON) scripts/check-cell-ltx-layout.py - -policy-entry-point-check: - @$(PYTHON) scripts/check-policy-entry-points.py - shipped-binary-version-check: @$(PYTHON) scripts/release/check-shipped-binary-versions.py --cargo "$(CARGO)" diff --git a/crab/scripts/cellule-qualification.py b/crab/scripts/cellule-qualification.py new file mode 100644 index 000000000..8dcb4de42 --- /dev/null +++ b/crab/scripts/cellule-qualification.py @@ -0,0 +1,85 @@ +#!/usr/bin/env python3 +"""Run the qualification validator from Crab's locked Cellule revision.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import tomllib + + +ROOT = Path(__file__).resolve().parents[2] + + +def cellule_root() -> Path: + workspace = tomllib.loads((ROOT / "Cargo.toml").read_text()) + dependency = workspace["workspace"]["dependencies"]["cellule-runtime"] + revision = dependency["rev"] + source = f"git+{dependency['git']}?rev={revision}#{revision}" + metadata = subprocess.run( + ["cargo", "metadata", "--locked", "--format-version", "1"], + cwd=ROOT, + check=True, + capture_output=True, + text=True, + ) + packages = json.loads(metadata.stdout)["packages"] + matches = [ + package for package in packages + if package["name"] == "cellule-runtime" and package["source"] == source + ] + if len(matches) != 1: + raise ValueError("the locked Cellule runtime does not match Cargo.toml") + manifest = Path(matches[0]["manifest_path"]) + root = manifest.parents[2] + if not (root / "Cargo.lock").is_file(): + raise ValueError("the locked Cellule checkout has no Cargo.lock") + return root + + +def target_dir() -> Path: + runner_temp = os.environ.get("RUNNER_TEMP") + if runner_temp: + return Path(runner_temp) / "cellule-qualification-target" + target_root = Path.home() / "Workspace" / "crabbuild-target" + if not target_root.is_dir() or not os.access(target_root, os.W_OK): + raise ValueError("the mounted Workspace volume is required for a Cellule build") + checkout = hashlib.sha256(os.fsencode(ROOT)).hexdigest()[:12] + return target_root / f"cellule-crab-{checkout}" + + +def main() -> int: + if len(sys.argv) < 2 or sys.argv[1] not in {"source-root", "build", "run"}: + print("usage: cellule-qualification.py source-root|build|run [receipt arguments...]", file=sys.stderr) + return 2 + source = cellule_root() + if sys.argv[1] == "source-root": + print(source) + return 0 + target = target_dir() + target.mkdir(parents=True, exist_ok=True) + command = [ + "cargo", sys.argv[1], "--locked", "--quiet", "--manifest-path", + str(source / "Cargo.toml"), "-p", "cellule-runtime", "--bin", + "qualification_receipt", + ] + if sys.argv[1] == "run": + command.extend(["--", *sys.argv[2:]]) + elif len(sys.argv) != 2: + print("build does not accept receipt arguments", file=sys.stderr) + return 2 + environment = os.environ.copy() + environment["CARGO_TARGET_DIR"] = str(target) + return subprocess.run(command, env=environment, check=False).returncode + + +if __name__ == "__main__": + try: + raise SystemExit(main()) + except (subprocess.CalledProcessError, ValueError) as error: + print(f"error: {error}", file=sys.stderr) + raise SystemExit(1) from error diff --git a/crab/scripts/check-architecture-gates.py b/crab/scripts/check-architecture-gates.py index 87c19004d..ab40d066f 100644 --- a/crab/scripts/check-architecture-gates.py +++ b/crab/scripts/check-architecture-gates.py @@ -18,11 +18,7 @@ "crab-storage": {"aws", "gcp", "azure", "fs"}, "crab-workflow": {"aws", "gcp", "azure", "fs", "http"}, } -OBJECT_STORE_DEV_IMPLEMENTATION_FEATURES = { - # This exercises the filesystem backend's unsupported conditional update. - # Runtime provider construction remains owned by crab-storage. - "crab-ltx": {"fs"}, -} +OBJECT_STORE_DEV_IMPLEMENTATION_FEATURES = {} XET_OWNER_PACKAGE = "crab-xet" XET_FORBIDDEN_PATTERNS = ("xet_core_structures", "xet-core-structures") XET_MODULE_REQUIRED_NORMAL_PACKAGES = { @@ -1748,12 +1744,7 @@ "pub use yaml::", } PRIVATE_INTERNAL_PACKAGES = { - "crab-cell-app", - "crab-cell-host", - "crab-cell-peer-http", - "crab-cell-runtime", "crab-http-server", - "crab-ltx", "crab-s3-gateway", "crab-vfs", "crab-workflow", @@ -1796,7 +1787,13 @@ "crab-s3-gateway": set(), } CELL_RUNTIME_SERVER_SOURCE_PATHS = ("crates/crab-http-server/src",) -CELL_RUNTIME_SERVER_IMPORT_PATTERN = "crab_ltx::" +LEGACY_CELL_PACKAGES = frozenset( + {"crab-cell-runtime", "crab-cell-app", "crab-cell-host", "crab-cell-peer-http", "crab-ltx"} +) +LEGACY_CELL_IMPORT_PATTERNS = ( + "crab_cell_runtime::", "crab_cell_app::", "crab_cell_host::", + "crab_cell_peer_http::", "crab_ltx::", +) CELL_RUNTIME_SERVER_CONSTRUCTOR_PATTERNS = ( "CellRuntime::new(", "CellRuntime::new_with_replica_host(", @@ -1817,72 +1814,7 @@ "node_log_transport", } ) -RETIRED_STANDALONE_LTX_SOURCE_PATHS = ( - "crates/crab-ltx/src", - "crates/crab-ltx/examples", -) -RETIRED_STANDALONE_LTX_FILENAMES = frozenset( - { - "schedule.rs", - "paged_vfs.rs", - "replica_roundtrip.rs", - "paged_read.rs", - "sparse_writer.rs", - "compact_history.rs", - "repository_replication_lifecycle.rs", - "rustfs_replication_scale_load.rs", - "rustfs_paged_read_scale_performance.rs", - } -) -RETIRED_STANDALONE_LTX_PATHS = ( - "crates/crab-ltx/src/paged/map.rs", -) -RETIRED_STANDALONE_LTX_SYMBOLS = re.compile( - r"\b(?:ReplicaHead|Replica|PagedDatabase|PagedConnection|CompactionSchedule)\b" -) -RETIRED_STANDALONE_LTX_MARKERS = re.compile( - r"(?:ltx/|head\.json|manifest\.json)" -) -CELL_RUNTIME_COORDINATION_KERNEL_PATH = "crates/crab-cell-runtime/src/coordination.rs" -# The actor adapter spans the actor module tree after the layout split; the -# gate reads every file below this path. -CELL_RUNTIME_COORDINATION_ACTOR_PATH = "crates/crab-cell-runtime/src/cell/actor" -CELL_RUNTIME_COORDINATION_REQUIRED_KERNEL_PATTERNS = ( - "pub(crate) enum CoordinationInput", - "pub(crate) enum CoordinationDecision", - "pub(crate) struct CoordinationState", - "pub(crate) fn step(&mut self, input: CoordinationInput)", -) -CELL_RUNTIME_COORDINATION_REQUIRED_ACTOR_PATTERNS = ( - "CoordinationState", - "coordination.step(CoordinationInput::", -) -CELL_RUNTIME_COORDINATION_FORBIDDEN_KERNEL_PATTERNS = ( - "async fn", - ".await", - "tokio::", - "object_store", - "rusqlite", - "reqwest::", - "std::fs", - "std::net", - "std::time", - "rand::", - "getrandom", - "spawn_blocking", - "Command::new(", -) WORKSPACE_DEPENDENCY_POLICY = { - "crab-cell-app": { - "normal": {"crab-cell-runtime"}, - # crab-cell-runtime is a dev edge so integration tests can enable its - # test-support feature without borrowing source files. - "dev": {"crab-cell-host", "crab-cell-runtime", "crab-ltx", "crab-storage"}, - }, - "crab-cell-host": {"normal": {"crab-cell-app", "crab-cell-runtime"}}, - "crab-cell-peer-http": {"normal": {"crab-cell-runtime"}}, - "crab-cell-runtime": {"normal": {"crab-ltx", "crab-storage"}}, - "crab-ltx": {"normal": {"crab-storage"}}, "crab-remote": { "normal": {"crab-auth", "crab-coordination", "crab-git", "crab-metadata", "crab-read", "crab-remote-git", "crab-storage", "crab-write", "crab-xet"}, }, @@ -1891,8 +1823,7 @@ "dev": {"crab-git", "crab-xet"}, }, "crab-write": {"normal": {"crab-coordination", "crab-types", "crab-git", "crab-metadata", "crab-remote-git", "crab-storage", "crab-xet"}}, - # The browser server composes Git and repository behavior with the external - # Cellule runtime. New edges to the legacy in-tree Cell crates are forbidden. + # The browser server composes Git and repository behavior with Cellule. "crab-http-server": { "normal": { "crab-coordination", @@ -2023,11 +1954,6 @@ "crab-xet": {}, } WORKSPACE_DEPENDENCY_PATHS = { - "crab-cell-app": "crates/crab-cell-app", - "crab-cell-host": "crates/crab-cell-host", - "crab-cell-peer-http": "crates/crab-cell-peer-http", - "crab-cell-runtime": "crates/crab-cell-runtime", - "crab-ltx": "crates/crab-ltx", "crab-write": "crates/crab-write", "crab-http-server": "crates/crab-http-server", "crab-auth": "crates/crab-auth", @@ -2492,17 +2418,16 @@ def production_struct_fields(text: str, name: str) -> set[str]: def check_cell_runtime_server_boundary(root: Path, metadata: dict) -> bool: - """Keep crab-http-server's production Cell ownership behind crab-cell-runtime.""" + """Keep the removed Cell crates out of the workspace and HTTP server.""" violations: list[str] = [] - package = package_by_name(metadata, "crab-http-server") - for dependency in package["dependencies"]: - if dependency["name"] != "crab-ltx": - continue - kind = dependency_kind(dependency) - if kind != "dev": - violations.append( - f"crab-http-server: {kind}-depends on crab-ltx; production Cell ownership belongs to crab-cell-runtime" - ) + for package in metadata["packages"]: + if package["name"] in LEGACY_CELL_PACKAGES: + violations.append(f"{package['name']}: removed Cell package remains in metadata") + for dependency in package["dependencies"]: + if dependency["name"] in LEGACY_CELL_PACKAGES: + violations.append( + f"{package['name']}: depends on removed {dependency['name']} package" + ) for relative_path in CELL_RUNTIME_SERVER_SOURCE_PATHS: path = root / relative_path @@ -2521,7 +2446,7 @@ def check_cell_runtime_server_boundary(root: Path, metadata: dict) -> bool: ): allowed_lines.update(range(1, len(text.splitlines()) + 1)) for number, line in enumerate(text.splitlines(), start=1): - if CELL_RUNTIME_SERVER_IMPORT_PATTERN in line and number not in allowed_lines: + if any(pattern in line for pattern in LEGACY_CELL_IMPORT_PATTERNS): violations.append(f"{relative}:{number}: {line.strip()}") if ( any(pattern in line for pattern in CELL_RUNTIME_SERVER_CONSTRUCTOR_PATTERNS) @@ -2538,7 +2463,7 @@ def check_cell_runtime_server_boundary(root: Path, metadata: dict) -> bool: ) if not violations: - print("ok: crab-http-server production Cell ownership stays behind crab-cell-runtime") + print("ok: removed Cell packages stay outside the workspace and server") return True print("error: crab-http-server escaped the canonical Cell runtime boundary:", file=sys.stderr) @@ -2547,107 +2472,6 @@ def check_cell_runtime_server_boundary(root: Path, metadata: dict) -> bool: return False -def check_standalone_ltx_hard_cut(root: Path) -> bool: - """Keep the retired epoch-head API out of callable LTX source and examples.""" - violations: list[str] = [] - for relative_path in RETIRED_STANDALONE_LTX_PATHS: - if (root / relative_path).exists(): - violations.append(f"{relative_path}: retired standalone LTX path") - for relative_path in RETIRED_STANDALONE_LTX_SOURCE_PATHS: - path = root / relative_path - if not path.exists(): - continue - candidates = [path] if path.is_file() else sorted(path.rglob("*.rs")) - for candidate in candidates: - if candidate.name in RETIRED_STANDALONE_LTX_FILENAMES: - violations.append(f"{rel(root, candidate)}: retired standalone LTX module") - text = candidate.read_text(encoding="utf-8") - state: dict[str, object] = {} - depth = 0 - catalog_head_base: int | None = None - for number, line in enumerate(text.splitlines(), start=1): - delta = rust_brace_delta(line, state) - if ( - candidate == root / "crates/crab-ltx/src/cell_layout.rs" - and re.match(r"\s*pub fn catalog_head_path\(", line) - and delta > 0 - ): - catalog_head_base = depth - # Catalog heads name immutable Cell catalog pages. Permit their - # head marker here while checking every other marker and symbol. - markers = line.replace("head.json", "") if catalog_head_base is not None else line - depth += delta - if catalog_head_base is not None and depth <= catalog_head_base: - catalog_head_base = None - if line.lstrip().startswith("//"): - continue - if RETIRED_STANDALONE_LTX_SYMBOLS.search(line) or RETIRED_STANDALONE_LTX_MARKERS.search( - markers - ): - violations.append(f"{rel(root, candidate)}:{number}: {line.strip()}") - - if not violations: - print("ok: standalone LTX epoch-head surfaces stay hard-removed") - return True - - print("error: retired standalone LTX surface regressed:", file=sys.stderr) - for violation in violations: - print(f" {violation}", file=sys.stderr) - return False - - -def check_cell_runtime_coordination_kernel(root: Path) -> bool: - """Keep volatile coordination decisions pure and actor-owned in production.""" - violations: list[str] = [] - kernel = root / CELL_RUNTIME_COORDINATION_KERNEL_PATH - actor = root / CELL_RUNTIME_COORDINATION_ACTOR_PATH - - if not kernel.exists(): - violations.append(f"{CELL_RUNTIME_COORDINATION_KERNEL_PATH}: missing coordination kernel") - else: - kernel_text = kernel.read_text(encoding="utf-8") - for pattern in CELL_RUNTIME_COORDINATION_REQUIRED_KERNEL_PATTERNS: - if pattern not in kernel_text: - violations.append( - f"{CELL_RUNTIME_COORDINATION_KERNEL_PATH}: missing {pattern!r}" - ) - allowed_lines = rust_test_only_lines(kernel_text) - for number, line in enumerate(kernel_text.splitlines(), start=1): - if number in allowed_lines: - continue - for pattern in CELL_RUNTIME_COORDINATION_FORBIDDEN_KERNEL_PATTERNS: - if pattern in line: - violations.append( - f"{CELL_RUNTIME_COORDINATION_KERNEL_PATH}:{number}: " - f"forbidden adapter dependency {pattern!r}" - ) - - if not actor.exists(): - violations.append(f"{CELL_RUNTIME_COORDINATION_ACTOR_PATH}: missing actor adapter") - elif actor.is_dir(): - actor_text = "\n".join( - candidate.read_text(encoding="utf-8") - for candidate in sorted(actor.rglob("*.rs")) - ) - for pattern in CELL_RUNTIME_COORDINATION_REQUIRED_ACTOR_PATTERNS: - if pattern not in actor_text: - violations.append(f"{CELL_RUNTIME_COORDINATION_ACTOR_PATH}: missing {pattern!r}") - else: - actor_text = actor.read_text(encoding="utf-8") - for pattern in CELL_RUNTIME_COORDINATION_REQUIRED_ACTOR_PATTERNS: - if pattern not in actor_text: - violations.append(f"{CELL_RUNTIME_COORDINATION_ACTOR_PATH}: missing {pattern!r}") - - if not violations: - print("ok: Cell coordination decisions stay in the pure kernel") - return True - - print("error: Cell coordination escaped the pure kernel boundary:", file=sys.stderr) - for violation in violations: - print(f" {violation}", file=sys.stderr) - return False - - def dependency_kind(dependency: dict) -> str: return dependency["kind"] or "normal" @@ -5091,8 +4915,6 @@ def main() -> int: check_package_release_policy(metadata), check_server_fixture_dependencies(metadata), check_cell_runtime_server_boundary(root, metadata), - check_standalone_ltx_hard_cut(root), - check_cell_runtime_coordination_kernel(root), check_workspace_dependency_policy(metadata), check_workspace_dependency_sources(root, metadata), check_workspace_xet_dependency_sources(root, metadata), diff --git a/crab/scripts/check-cell-ltx-layout.py b/crab/scripts/check-cell-ltx-layout.py deleted file mode 100644 index 5bf83510f..000000000 --- a/crab/scripts/check-cell-ltx-layout.py +++ /dev/null @@ -1,312 +0,0 @@ -#!/usr/bin/env python3 -"""Check the Cell and LTX crate layout rules from advisor plan 033. - -Rules: - 1. No `#[path]` attributes in the four crates. - 2. Every `#[cfg(test)]`/`#[test]` location under `src/` is listed in the - crate's `tests-allow-list.txt`. - 3. Every `tests-allow-list.txt` entry names an existing `src/` file, carries - a reason, and still holds tests or test modules, so a moved or emptied - test location cannot leave a stale entry behind. - 4. No `tests/.rs` that shadows `src/.rs`. - 5. Every test suite root has a matching module directory and at least one - test. - 6. Every module file inside a suite directory is declared by its parent module - file, so a split cannot leave a test file that the compiler never builds. - 7. Every `src/...` or `tests/...` path named by a crate guide exists, so the - guides keep owning the layout rules they describe. - 8. The runtime root surface equals `api-prelude.txt`. - 9. The runtime coordination kernel stays sans-I/O: no async, clock, or - storage, so the simulator and the model can replay the same transitions. - 10. The four crates depend only on each other and `crab-storage` (which the - Cellule synthesis renames to `cellule-store`), so the extraction cannot - acquire a Crab-specific coupling on the way out. - 11. Every `tests/...` path named in a crate's documentation exists, so the - delivery evidence map cannot point a reader at a file that never existed. -""" - -from __future__ import annotations - -import re -import sys -from pathlib import Path - -ROOT = Path(__file__).resolve().parents[2] -CRATES = ( - "crates/crab-cell-runtime", - "crates/crab-cell-app", - "crates/crab-cell-host", - "crates/crab-ltx", -) -SUITES = { - "crates/crab-cell-runtime": ( - "runtime", - "primitives", - "protocol", - "contracts", - "fleet", - "qualification", - ), - "crates/crab-cell-app": ("reference_application",), - "crates/crab-cell-host": ("node",), - "crates/crab-ltx": ("cell", "ltx", "host"), -} -TEST_ATTR = re.compile(r"^\s*#\[(?:tokio::)?test", re.M) -CFG_TEST = re.compile(r"^\s*#\[cfg\(test\)\]", re.M) -PATH_ATTR = re.compile(r"#\[path\s*=") -MODULE_DECL = re.compile(r"^\s*(?:pub(?:\([^)]*\))? )?mod ([a-z_][a-z_0-9]*)\s*;", re.M) -LINE_COMMENT = re.compile(r"//[^\n]*") -ROOT_RE_EXPORT = re.compile(r"^pub use ([^;]+);", re.M) -ROOT_CONST = re.compile(r"^\s*pub (?:const|struct|enum|trait|fn|type) ([A-Za-z_][A-Za-z0-9_]*)", re.M) - -# The four Cellule crates plus the shared store crate they may keep depending on. -EXTRACTION_DEPENDENCIES = frozenset( - { - "crab-storage", - "crab-ltx", - "crab-cell-runtime", - "crab-cell-app", - "crab-cell-host", - } -) -CRAB_DEPENDENCY = re.compile(r"^(crab-[a-z0-9-]+)", re.M) - -# A pure coordination kernel is what lets `coordination/sim.rs` and the TLA+ -# model replay production transitions; an I/O call here would silently move the -# decision out of the replayable surface. -SANS_IO_PATTERNS = ( - ("async", re.compile(r"\basync\b")), - ("await", re.compile(r"\.await\b")), - ("tokio", re.compile(r"\btokio::")), - ("storage", re.compile(r"\brusqlite\b|\bobject_store\b|\bstd::fs\b")), - ("clock", re.compile(r"\bSystemTime\b|\bInstant\b|\bstd::time\b")), -) - - -def sans_io_paths(crate_path: Path) -> list[Path]: - paths = [] - root = crate_path / "src" / "coordination.rs" - if root.is_file(): - paths.append(root) - directory = crate_path / "src" / "coordination" - if directory.is_dir(): - paths.extend(sorted(directory.rglob("*.rs"))) - return paths - - -def allow_list(crate_path: Path) -> dict[str, str]: - path = crate_path / "tests-allow-list.txt" - if not path.is_file(): - return {} - entries: dict[str, str] = {} - for line in path.read_text().splitlines(): - entry, _, reason = line.partition("#") - entry = entry.strip() - if entry: - entries[entry] = reason.strip() - return entries - - -def check_allow_list_entries(crate_path: Path, entries: dict[str, str]) -> list[str]: - problems: list[str] = [] - allow_path = (crate_path / "tests-allow-list.txt").relative_to(ROOT) - for entry, reason in sorted(entries.items()): - target = crate_path / "src" / entry - if not target.is_file(): - problems.append(f"{allow_path}: {entry} does not name an existing src file") - continue - if not reason: - problems.append(f"{allow_path}: {entry} needs a reason comment") - text = LINE_COMMENT.sub("", target.read_text()) - if ( - TEST_ATTR.search(text) is None - and CFG_TEST.search(text) is None - and MODULE_DECL.search(text) is None - ): - problems.append(f"{allow_path}: {entry} no longer holds tests or test modules") - return problems - - - -def check_suite_module_declarations(crate: str, crate_path: Path) -> list[str]: - """Every suite module file must be declared by the module that owns it.""" - problems: list[str] = [] - tests = crate_path / "tests" - for suite in SUITES.get(crate, ()): - suite_dir = tests / suite - suite_root = tests / f"{suite}.rs" - if not suite_dir.is_dir() or not suite_root.is_file(): - continue - for path in sorted(suite_dir.rglob("*.rs")): - if path.name == "mod.rs": - continue - if path.parent == suite_dir: - declaring = suite_root - else: - declaring = path.parent.with_suffix(".rs") - if not declaring.is_file(): - declaring = path.parent / "mod.rs" - if not declaring.is_file(): - problems.append( - f"{path.relative_to(ROOT)}: no module file declares it" - ) - continue - text = LINE_COMMENT.sub("", declaring.read_text()) - if path.stem not in MODULE_DECL.findall(text): - problems.append( - f"{path.relative_to(ROOT)}: {declaring.relative_to(ROOT)} does " - f"not declare `mod {path.stem};`" - ) - return problems - - - -GUIDE_PATH = re.compile(r"`((?:src|tests)/[^`]+)`") -DOCUMENTED_TEST_PATH = re.compile(r"`(tests/[^`\s]+?\.rs)`") - - -def check_guide_paths(crate_path: Path) -> list[str]: - """Every crate-relative path a crate guide names must exist.""" - guide = crate_path / "AGENTS.md" - if not guide.is_file(): - return [] - problems: list[str] = [] - for token in GUIDE_PATH.findall(guide.read_text()): - token = token.strip() - if "<" in token or "{" in token: - continue - if not (crate_path / token).exists(): - problems.append( - f"{guide.relative_to(ROOT)}: {token} is not present in the crate" - ) - return problems - - -def check_extraction_dependencies(crate_path: Path) -> list[str]: - """Every `crab-*` dependency stays inside the Cellule extraction set.""" - manifest = crate_path / "Cargo.toml" - if not manifest.is_file(): - return [] - problems: list[str] = [] - for name in sorted(set(CRAB_DEPENDENCY.findall(manifest.read_text()))): - if name not in EXTRACTION_DEPENDENCIES: - problems.append( - f"{manifest.relative_to(ROOT)}: {name} is outside the Cellule dependency set" - ) - return problems - - -def check_documented_test_paths(crate_path: Path) -> list[str]: - """Every crate-relative test path a crate's docs name must exist. - - `src/...` paths are deliberately not checked: a crate's docs may describe - another crate's sources, while `tests/...` paths are always crate-relative. - """ - problems: list[str] = [] - for path in sorted(crate_path.rglob("*.md")): - if "target" in path.parts: - continue - for token in DOCUMENTED_TEST_PATH.findall(path.read_text()): - token = token.strip() - if "<" in token or "{" in token: - continue - if not (crate_path / token).exists(): - problems.append( - f"{path.relative_to(ROOT)}: {token} is not present in the crate" - ) - return problems - - -def check(crate: str) -> list[str]: - crate_path = ROOT / crate - problems: list[str] = [] - entries = allow_list(crate_path) - allowed = set(entries) - problems.extend(check_allow_list_entries(crate_path, entries)) - problems.extend(check_suite_module_declarations(crate, crate_path)) - problems.extend(check_guide_paths(crate_path)) - problems.extend(check_documented_test_paths(crate_path)) - problems.extend(check_extraction_dependencies(crate_path)) - search_roots = [crate_path / "src"] - if (crate_path / "tests").is_dir(): - search_roots.append(crate_path / "tests") - for path in sorted({file for root in search_roots for file in root.rglob("*.rs")}): - text = path.read_text() - if PATH_ATTR.search(LINE_COMMENT.sub("", text)): - problems.append(f"{path.relative_to(ROOT)}: #[path] attribute is not allowed") - if "/src/" not in f"/{path.relative_to(ROOT)}": - continue - if TEST_ATTR.search(text) or CFG_TEST.search(text): - relative = str(path.relative_to(crate_path / "src")) - if relative not in allowed: - problems.append( - f"{path.relative_to(ROOT)}: in-src tests are not in tests-allow-list.txt" - ) - for suite in SUITES.get(crate, ()): - root_file = crate_path / "tests" / f"{suite}.rs" - if not root_file.is_file(): - problems.append(f"{root_file.relative_to(ROOT)}: missing suite root") - continue - if not TEST_ATTR.search(root_file.read_text()) and (crate_path / "tests" / suite).is_dir(): - module_dir = crate_path / "tests" / suite - if not any(TEST_ATTR.search(child.read_text()) for child in module_dir.rglob("*.rs")): - problems.append(f"{root_file.relative_to(ROOT)}: suite has no tests") - tests_dir = crate_path / "tests" - if tests_dir.is_dir(): - suite_names = set(SUITES.get(crate, ())) - for path in tests_dir.glob("*.rs"): - if path.stem in suite_names: - continue - if (crate_path / "src" / path.name).exists(): - problems.append( - f"{path.relative_to(ROOT)}: test file shadows a source module name" - ) - prelude_path = crate_path / "api-prelude.txt" - if prelude_path.is_file(): - expected = {line.strip() for line in prelude_path.read_text().splitlines() if line.strip()} - actual: set[str] = set() - lib = (crate_path / "src/lib.rs").read_text() - for statement in ROOT_RE_EXPORT.findall(lib): - statement = " ".join(statement.split()) - if "{" in statement: - inner = statement[statement.index("{") + 1 : statement.rindex("}")] - for part in inner.split(","): - part = part.strip() - if part: - actual.add(part.split(" as ")[-1].strip()) - else: - actual.add(statement.split("::")[-1].strip()) - if actual != expected: - problems.append( - f"{prelude_path.relative_to(ROOT)}: root surface differs " - f"(missing {sorted(expected - actual)}, extra {sorted(actual - expected)})" - ) - if crate == "crates/crab-cell-runtime": - for path in sans_io_paths(crate_path): - text = LINE_COMMENT.sub("", path.read_text()) - for label, pattern in SANS_IO_PATTERNS: - found = pattern.search(text) - if found is None: - continue - line = text[: found.start()].count("\n") + 1 - problems.append( - f"{path.relative_to(ROOT)}:{line}: coordination kernel must stay " - f"sans-I/O (found {label})" - ) - return problems - - -def main() -> int: - problems: list[str] = [] - for crate in CRATES: - problems.extend(check(crate)) - if problems: - for problem in problems: - print(f"error: {problem}") - return 1 - print("ok: Cell and LTX crate layout checks pass") - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/crab/scripts/check-doc-rust-fences.py b/crab/scripts/check-doc-rust-fences.py deleted file mode 100644 index eeaaefe94..000000000 --- a/crab/scripts/check-doc-rust-fences.py +++ /dev/null @@ -1,104 +0,0 @@ -#!/usr/bin/env python3 -"""Check that Rust fences in the Cell and LTX documentation parse. - -A reader copies documentation examples, so a fence that is not valid Rust is a -defect even when no test compiles it: the docs are `rust,ignore` precisely -because they need a provider, not because they may be syntactically broken. - -Rules: - 1. Every ```rust fence under the four crates parses after being wrapped in - `fn main() { ... }`, so a statement snippet parses while a broken one fails. - 2. A deliberately incomplete snippet is listed in ALLOW with the reason and - the first line of the fence, so the exception is reviewed, not assumed. -""" - -from __future__ import annotations - -import subprocess -import sys -import tempfile -from pathlib import Path - -ROOT = Path(__file__).resolve().parents[2] -CRATES = ( - "crates/crab-cell-runtime", - "crates/crab-cell-app", - "crates/crab-cell-host", - "crates/crab-ltx", -) - -# (crate-relative path, first non-empty line of the fence): reason. -ALLOW = { - ( - "crates/crab-cell-runtime/docs/failover-and-followers.md", - "let observed = /* latest VersionedControl loaded from authority */;", - ): "the comment is the placeholder for a value only the caller can load", -} - - -def fences(text: str) -> list[tuple[str, int, str]]: - """Returns one (language, first line number, body) entry per fenced block.""" - found: list[tuple[str, int, str]] = [] - body: list[str] | None = None - language = "" - start = 0 - for number, line in enumerate(text.splitlines(), 1): - if line.startswith("```"): - if body is None: - body, language, start = [], line.strip("`").strip(), number + 1 - else: - found.append((language, start, "\n".join(body))) - body = None - elif body is not None: - body.append(line) - return found - - -def parses(body: str) -> str | None: - """Returns the first parse error, or None when the wrapped snippet parses.""" - wrapped = "fn main() {\n" + body + "\n}\n" - with tempfile.NamedTemporaryFile("w", suffix=".rs") as file: - file.write(wrapped) - file.flush() - result = subprocess.run( - ["rustfmt", "--edition", "2024", "--emit", "stdout", file.name], - capture_output=True, - text=True, - ) - if result.returncode == 0: - return None - return next( - (line for line in result.stderr.splitlines() if line.startswith("error")), - "rustfmt rejected the snippet", - ) - - -def main() -> int: - problems: list[str] = [] - checked = 0 - for crate in CRATES: - crate_path = ROOT / crate - for path in sorted(crate_path.rglob("*.md")): - if "target" in path.parts: - continue - relative = str(path.relative_to(ROOT)) - for language, start, body in fences(path.read_text()): - if not language.startswith("rust"): - continue - first = next((line for line in body.splitlines() if line.strip()), "") - if (relative, first) in ALLOW: - continue - checked += 1 - error = parses(body) - if error is not None: - problems.append(f"{relative}:{start}: {error}") - for problem in problems: - print(f"error: {problem}", file=sys.stderr) - if problems: - return 1 - print(f"ok: {checked} documented Rust snippets parse") - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/crab/scripts/check-policy-entry-points.py b/crab/scripts/check-policy-entry-points.py deleted file mode 100755 index 2652c9fe5..000000000 --- a/crab/scripts/check-policy-entry-points.py +++ /dev/null @@ -1,219 +0,0 @@ -#!/usr/bin/env python3 -"""Check that Cell runtime policy entry points are wired or explicitly deferred. - -A policy seam that only tests reach is invisible in review: the tests stay -green while the behaviour it implements never runs, and the code reads as if the -policy were active. Three such seams have already existed in this tree -(`observe_pressure`, `evict_idle`, `takeover_unpublished`), so this check keeps -an explicit inventory of every policy entry point in the Cell stack. - -Rules: - 1. Every public function in the Cell stack whose name starts with a policy - prefix is a policy entry point. - 2. A `wired` entry point must have at least one call site outside tests. - 3. A `deferred` entry point must have no call site outside tests, and must - record why it is not wired yet. - 4. An entry point with no inventory entry and no production caller fails the - check, so a new seam cannot be added unnoticed. - -The scan is name-based and crate-scoped on purpose: it is a review gate, not a -compiler, and the inventory carries the judgment. -""" - -from __future__ import annotations - -import re -import sys -from pathlib import Path - -ROOT = Path(__file__).resolve().parents[2] - -# Cell stack only. Unrelated crates use the same verbs for their own types -# (`CatalogWriter::sweep_unreferenced`, cache evictors), which would otherwise -# read as production callers of a runtime seam. -CRATES = ( - "crates/crab-cell-runtime", - "crates/crab-cell-app", - "crates/crab-cell-host", - "crates/crab-http-server", -) - -# Bound policy surface: lifecycle and fleet seams that decide what the runtime -# does with ownership, residency, or pressure. Ordinary accessors stay out. -POLICY_PREFIXES = ( - "acquire_idle", - "activate_", - "evict", - "observe_pressure", - "quiesce", - "rebalance", - "release_idle", - "shed", - "sweep", - "takeover_", -) - -# Only the crate-external surface counts: a `pub(crate)` helper cannot be a -# policy entry point for another crate. -DEFINITION = re.compile(r"^\s*pub\s+(?:async\s+)?fn\s+(\w+)", re.M) -CFG_TEST = "#[cfg(test)]" - -# (crate-relative path, symbol): (status, reason) -INVENTORY = { - ( - "crates/crab-cell-runtime/src/cell/actor/runtime.rs", - "observe_pressure", - ): ( - "deferred", - "the actor samples this node's own reservation ledger on its tick, so " - "the classifier and its shedding path are live; this entry point stays " - "for a host that measures cgroup or host pressure, which no caller " - "supplies yet", - ), - ( - "crates/crab-cell-runtime/src/cell/actor/runtime.rs", - "evict_idle", - ): ( - "deferred", - "shedding runs through the pressure classifier, which starts the same " - "bounded eviction; this entry point remains the explicit operator sweep " - "and has no production caller", - ), - ( - "crates/crab-cell-runtime/src/cell/actor/acquire.rs", - "takeover_unpublished", - ): ( - "deferred", - "the unleased initializer refuses to fence an existing unpublished owner " - "(it cannot mint a takeover proof) and the router fails closed on a " - "rootless control record, so only tests exercise this takeover path", - ), - ( - "crates/crab-cell-runtime/src/primitives/blob/store.rs", - "sweep_unreferenced", - ): ( - "deferred", - "no product collector exists yet; abandoned uploads follow the provider " - "lifecycle contract", - ), -} - - -def strip_test_items(text: str) -> str: - """Removes `#[cfg(test)]` items so in-src tests do not count as callers.""" - while True: - index = text.find(CFG_TEST) - if index < 0: - return text - start = text.find("{", index) - if start < 0: - return text[:index] - depth = 0 - cursor = start - while cursor < len(text): - if text[cursor] == "{": - depth += 1 - elif text[cursor] == "}": - depth -= 1 - if depth == 0: - break - cursor += 1 - text = text[:index] + text[cursor + 1 :] - - -def is_test_path(path: Path) -> bool: - parts = path.parts - return ( - any(part == "tests" or part.endswith("_tests") for part in parts) - or path.name == "tests.rs" - or path.name.endswith("_tests.rs") - ) - - -def production_sources(crate: Path) -> list[Path]: - return sorted( - path - for path in (crate / "src").rglob("*.rs") - if not is_test_path(path.relative_to(crate)) - ) - - -def definitions(crates: tuple[Path, ...]) -> dict[tuple[str, str], Path]: - found: dict[tuple[str, str], Path] = {} - for crate in crates: - for path in production_sources(crate): - text = strip_test_items(path.read_text()) - for name in DEFINITION.findall(text): - if name.startswith(POLICY_PREFIXES): - found[(str(path.relative_to(ROOT)), name)] = path - return found - - -def call_sites(crates: tuple[Path, ...], name: str) -> list[str]: - pattern = re.compile(r"\b" + re.escape(name) + r"\s*\(") - definition = re.compile(r"fn\s+" + re.escape(name) + r"\s*\(") - hits = [] - for crate in crates: - for path in production_sources(crate): - text = strip_test_items(path.read_text()) - for number, line in enumerate(text.splitlines(), 1): - if line.lstrip().startswith("//"): - continue - if pattern.search(line) and not definition.search(line): - hits.append(f"{path.relative_to(ROOT)}:{number}") - return hits - - -def check() -> list[str]: - crates = tuple(ROOT / crate for crate in CRATES) - found = definitions(crates) - problems: list[str] = [] - for entry, (status, reason) in sorted(INVENTORY.items()): - if status not in ("wired", "deferred"): - problems.append(f"{entry[0]}: {entry[1]} has unknown status {status!r}") - continue - if not reason: - problems.append(f"{entry[0]}: {entry[1]} records no reason") - if entry not in found: - problems.append( - f"{entry[0]}: inventory lists {entry[1]}, which is not a public policy entry point" - ) - continue - hits = call_sites(crates, entry[1]) - if status == "wired" and not hits: - problems.append( - f"{entry[0]}: {entry[1]} is inventoried as wired but no caller outside tests reaches it" - ) - if status == "deferred" and hits: - problems.append( - f"{entry[0]}: {entry[1]} is inventoried as deferred but {hits[0]} calls it" - ) - for entry, path in sorted(found.items()): - if entry in INVENTORY: - continue - hits = call_sites(crates, entry[1]) - if not hits: - problems.append( - f"{path.relative_to(ROOT)}: {entry[1]} is an unwired policy entry point; " - "wire it or add it to INVENTORY with a status and reason" - ) - return problems - - -def main() -> int: - problems = check() - if problems: - for problem in problems: - print(f"error: {problem}", file=sys.stderr) - return 1 - wired = sum(1 for status, _ in INVENTORY.values() if status == "wired") - deferred = len(INVENTORY) - wired - print( - f"ok: every Cell runtime policy entry point is wired or deferred " - f"({wired} wired, {deferred} deferred)" - ) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/crab/scripts/test_check_architecture_gates.py b/crab/scripts/test_check_architecture_gates.py index a61fda68c..5782e2e72 100644 --- a/crab/scripts/test_check_architecture_gates.py +++ b/crab/scripts/test_check_architecture_gates.py @@ -52,25 +52,6 @@ def test_xet_adapter_remains_the_runtime_owner(self): class StorageScopeTests(unittest.TestCase): - def test_object_store_feature_ownership_distinguishes_test_fixtures(self): - metadata = {"packages": [{ - "name": "crab-ltx", - "dependencies": [ - { - "name": "object_store", "kind": None, - "uses_default_features": False, "features": [], - }, - { - "name": "object_store", "kind": "dev", - "uses_default_features": False, "features": ["fs"], - }, - ], - }]} - with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()): - self.assertTrue(GATES.check_object_store_features(metadata)) - metadata["packages"][0]["dependencies"][0]["features"] = ["fs"] - self.assertFalse(GATES.check_object_store_features(metadata)) - def test_dependency_prefixes_ignore_words_embedded_in_test_names(self): metadata = {"packages": [{ "name": "crab-storage", @@ -103,255 +84,29 @@ def test_dependency_prefixes_ignore_words_embedded_in_test_names(self): self.assertEqual(result, expected) -class CellRuntimeBoundaryTests(unittest.TestCase): - def metadata(self, dependency_kind="dev"): - return { - "packages": [{ - "name": "crab-http-server", - "dependencies": [{ - "name": "crab-ltx", - "kind": dependency_kind, - "optional": False, - "features": ["replica"], - }], - }], - } - - def check_source( - self, - text, - relative="crates/crab-http-server/src/lib.rs", - dependency_kind="dev", - ): +class CelluleBoundaryTests(unittest.TestCase): + def check_source(self, source: str, dependencies: tuple[str, ...] = ()) -> bool: with tempfile.TemporaryDirectory() as directory: root = Path(directory) - source = root / relative - source.parent.mkdir(parents=True) - source.write_text(text, encoding="utf-8") - with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()): - return GATES.check_cell_runtime_server_boundary( - root, - self.metadata(dependency_kind), - ) - - def test_production_import_is_rejected(self): - self.assertFalse(self.check_source("use crab_ltx::CellReplica;\nfn route() {}\n")) - - def test_normal_and_build_dependencies_are_rejected(self): - for dependency_kind in (None, "build"): - with self.subTest(dependency_kind=dependency_kind): - self.assertFalse( - self.check_source("fn route() {}\n", dependency_kind=dependency_kind) - ) - - def test_cfg_test_module_and_nested_test_module_are_admitted(self): - source = """fn route() {} - -#[cfg(test)] -mod tests { - mod nested { - use crab_ltx::CellReplica; - } -} - -fn later_production_code() {} -""" - self.assertTrue(self.check_source(source)) - - def test_test_file_is_admitted_but_production_after_cfg_block_is_not(self): - self.assertTrue( - self.check_source( - "use crab_ltx::CellReplica;\n", - relative="crates/crab-http-server/src/cells/scheduler/tests.rs", - ) - ) - - def test_production_server_component_fields_are_rejected(self): - for field in ( - "cell_runtime", - "catalog", - "scheduler_status", - "cell_capacity", - "repository_cells", - "peer_receiver", - "follower_store", - "node_log_transport", - ): - with self.subTest(field=field): - self.assertFalse( - self.check_source( - "pub(crate) struct Server {\n" - f" {field}: usize,\n" - "}\n", - relative="crates/crab-http-server/src/server.rs", - ) - ) - - def test_test_only_server_component_fields_are_admitted(self): - self.assertTrue( - self.check_source( - "pub(crate) struct Server {\n" - " #[cfg(test)]\n" - " repository_cells: usize,\n" - "}\n", - relative="crates/crab-http-server/src/server.rs", - ) - ) - self.assertFalse( - self.check_source( - "#[cfg(test)]\nmod tests { use crab_ltx::CellReplica; }\n" - "use crab_ltx::Db;\n", - ) - ) - - -class CellCoordinationKernelTests(unittest.TestCase): - def check_kernel(self, kernel, actor): - with tempfile.TemporaryDirectory() as directory: - root = Path(directory) - kernel_path = root / GATES.CELL_RUNTIME_COORDINATION_KERNEL_PATH - actor_path = root / GATES.CELL_RUNTIME_COORDINATION_ACTOR_PATH - kernel_path.parent.mkdir(parents=True, exist_ok=True) - actor_path.parent.mkdir(parents=True, exist_ok=True) - kernel_path.write_text(kernel, encoding="utf-8") - actor_path.write_text(actor, encoding="utf-8") - with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()): - return GATES.check_cell_runtime_coordination_kernel(root) - - def test_pure_kernel_and_actor_adapter_are_admitted(self): - kernel = """pub(crate) enum CoordinationInput {} -pub(crate) enum CoordinationDecision {} -pub(crate) struct CoordinationState; -impl CoordinationState { - pub(crate) fn step(&mut self, input: CoordinationInput) {} -} -""" - self.assertTrue( - self.check_kernel( - kernel, - "use crate::coordination::CoordinationState;\n" - "active.coordination.step(CoordinationInput::Fence);\n", - ) - ) - - def test_kernel_rejects_async_and_provider_adapters(self): - kernel = """pub(crate) enum CoordinationInput {} -pub(crate) enum CoordinationDecision {} -pub(crate) struct CoordinationState; -impl CoordinationState { - pub(crate) fn step(&mut self, input: CoordinationInput) {} - async fn read() { object_store::get().await; } -} -""" - self.assertFalse( - self.check_kernel( - kernel, - "use crate::coordination::CoordinationState;\n" - "active.coordination.step(CoordinationInput::Fence);\n", - ) - ) - - def test_actor_must_retain_the_kernel_adapter_call(self): - kernel = """pub(crate) enum CoordinationInput {} -pub(crate) enum CoordinationDecision {} -pub(crate) struct CoordinationState; -impl CoordinationState { - pub(crate) fn step(&mut self, input: CoordinationInput) {} -} -""" - self.assertFalse(self.check_kernel(kernel, "fn actor() {}\n")) - - -class StandaloneLtxHardCutTests(unittest.TestCase): - def check_source(self, text, relative="crates/crab-ltx/src/lib.rs"): - with tempfile.TemporaryDirectory() as directory: - root = Path(directory) - source = root / relative - source.parent.mkdir(parents=True) - source.write_text(text, encoding="utf-8") + path = root / "crates/crab-http-server/src/lib.rs" + path.parent.mkdir(parents=True) + path.write_text(source, encoding="utf-8") + metadata = {"packages": [{ + "name": "crab-http-server", + "dependencies": [{"name": name} for name in dependencies], + }]} with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()): - return GATES.check_standalone_ltx_hard_cut(root) - - def test_tenant_catalog_head_is_admitted_only_in_its_layout_method(self): - source = '''impl CellStorageLayout { - pub fn catalog_head_path(&self, tenant: &[u8; 16], shard: u8) -> Path { - Path::from(format!( - "{}/{}/{shard:02x}/head.json", - self.catalog_tenants_prefix(), - encode_hex(tenant) - )) - } -} -''' - path = "crates/crab-ltx/src/cell_layout.rs" - self.assertTrue(self.check_source(source, path)) - for relative, text in ( - ("crates/crab-ltx/src/other/cell_layout.rs", source), - (path, source.replace("catalog_head_path", "standalone_head_path")), - (path, source + 'fn head() -> &\'static str { "head.json" }\n'), - (path, source.replace("head.json", "manifest.json")), - (path, source.replace('"{}/{}/{shard:02x}/head.json",', - '"catalog/{shard:02x}/head.json", Replica,')), - ): - with self.subTest(relative=relative, source=text): - self.assertFalse(self.check_source(text, relative)) + return GATES.check_cell_runtime_server_boundary(root, metadata) - def test_retired_epoch_head_symbols_are_rejected(self): - for symbol in ( - "Replica", - "ReplicaHead", - "PagedDatabase", - "PagedConnection", - "CompactionSchedule", - ): - with self.subTest(symbol=symbol): - self.assertFalse(self.check_source(f"pub struct {symbol};\n")) + def test_cellule_import_is_admitted(self): + self.assertTrue(self.check_source("use cellule_runtime::CellRuntime;\n")) - def test_cell_scoped_surfaces_are_admitted(self): - self.assertTrue( - self.check_source( - "pub struct CellReplica;\n" - "pub struct CellPagedDatabase;\n" - "pub struct CellWritableDatabase;\n" - ) - ) + def test_removed_workspace_dependency_is_rejected(self): + self.assertFalse(self.check_source("", ("crab-cell-runtime",))) + self.assertFalse(self.check_source("", ("crab-ltx",))) - def test_retired_names_in_rust_comments_are_not_callable_surfaces(self): - self.assertTrue( - self.check_source( - "/// Replica metadata was not valid JSON.\n" - " // The old head.json path is no longer used.\n" - ) - ) - - def test_canonical_replica_module_is_admitted_but_storage_markers_are_rejected(self): - with tempfile.TemporaryDirectory() as directory: - root = Path(directory) - source = root / "crates/crab-ltx/src/replica.rs" - source.parent.mkdir(parents=True) - source.write_text("pub struct CellReplica;\n", encoding="utf-8") - with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr( - io.StringIO() - ): - self.assertTrue(GATES.check_standalone_ltx_hard_cut(root)) - - self.assertFalse(self.check_source("const PREFIX: &str = \"ltx/\";\n")) - self.assertFalse(self.check_source("const HEAD: &str = \"head.json\";\n")) - self.assertFalse(self.check_source("const MANIFEST: &str = \"manifest.json\";\n")) - - def test_retired_directories_and_examples_are_rejected(self): - with tempfile.TemporaryDirectory() as directory: - root = Path(directory) - paged_map = root / "crates/crab-ltx/src/paged/map.rs" - paged_map.parent.mkdir(parents=True) - paged_map.write_text("pub fn page_map() {}\n", encoding="utf-8") - example = root / "crates/crab-ltx/examples/replica_roundtrip.rs" - example.parent.mkdir(parents=True) - example.write_text("fn main() {}\n", encoding="utf-8") - with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr( - io.StringIO() - ): - self.assertFalse(GATES.check_standalone_ltx_hard_cut(root)) + def test_removed_import_is_rejected_even_in_tests(self): + self.assertFalse(self.check_source("#[cfg(test)] mod tests { use crab_ltx::CellReplica; }\n")) if __name__ == "__main__": diff --git a/crates/AGENTS.md b/crates/AGENTS.md index 8188c2373..95fa939c7 100644 --- a/crates/AGENTS.md +++ b/crates/AGENTS.md @@ -24,12 +24,8 @@ Scoped rules for `crates/`. Root `AGENTS.md` also applies. - `crab-staging` — local segment staging, chunk indexes, prepared push plans, multipart resume, compaction, and recovery. - `crab-coordination` — push locks, write coordination, and feature-gated DynamoDB, Spanner, and Cosmos DB active-active backends. - `crab-lfs` — Git LFS object layout, storage access, and integrity checks; pointer parsing remains in `crab-git`. -- `crab-ltx` — legacy qualification source for the LTX mechanics now owned by Cellule. Provenance remains in `crab-ltx/UPSTREAM.md` until its release gates move. -- `crab-cell-runtime` — legacy qualification source for the Cell runtime now owned by Cellule. -- `crab-cell-peer-http` — legacy qualification source for the peer transport now owned by Cellule. - `crab-http-server` consumes the pinned Cellule runtime, application, host, - store, and LTX crates. The in-tree Cell and LTX crates still serve legacy - qualification workflows; do not add a new product dependency on them. + store, and LTX crates. Cellule owns their source and qualification contracts. ### Read, Cache, and Virtual Filesystems @@ -84,27 +80,13 @@ crab-types - Server crates are composition boundaries. Do not move server policy or broad dependency sets into lower libraries. - Preserve source errors across crate boundaries. Map errors only where the receiving layer adds a real contract or user-facing decision. -## Cell and LTX Layout - -`crab-cell-runtime`, `crab-cell-app`, `crab-cell-host`, and `crab-ltx` remain -for Crab's legacy qualification workflows. Cellule now owns the live runtime; -keep these source-local gates intact until their release evidence moves there: - -- `src/` is production code; the test surface is `tests/.rs` plus its - `tests//` modules, one binary per capability suite. Never name a suite - after a source file it happens to exercise. -- Shared fixtures live in `tests/support/`; a suite declares `mod support;` and - reaches it through the crate root. No `#[path]` attributes. -- A test that needs private or `pub(crate)` state stays in its module and is - listed with a reason in that crate's `tests-allow-list.txt`; everything else - belongs in a suite. -- The runtime and LTX root surfaces are frozen in `api-prelude.txt`; a new root - re-export edits `src/lib.rs` and that file in the same commit. -- These four crates depend only on each other and `crab-storage`; do not add - product dependencies or extend their public contracts. -- `crab/scripts/check-cell-ltx-layout.py` checks the layout and - `crab/scripts/check-policy-entry-points.py` the actor policy seams; the - per-crate `AGENTS.md` guides own the test map and read-first routes. +## Cellule Boundary + +`crab-http-server` embeds the pinned Cellule runtime. Keep product HTTP, +authorization, cloud configuration, and release policy here; change reusable +Cell and LTX mechanics in [Cellule](https://github.com/crabbuild/cellule). +Crab's release workflow verifies the exact pinned Cellule qualification +contracts while binding evidence to the server image and source revision. ## Feature Flags diff --git a/crates/crab-cell-app/AGENTS.md b/crates/crab-cell-app/AGENTS.md deleted file mode 100644 index 7807ad46a..000000000 --- a/crates/crab-cell-app/AGENTS.md +++ /dev/null @@ -1,15 +0,0 @@ -# AGENTS.md - -Scoped rules for `crates/crab-cell-app/`. Root and `crates/` guidance apply. - -- Keep this crate dependency-light: runtime contracts only; no HTTP, provider, - storage construction, credentials, or node lifecycle policy. -- Stable IDs, role, shard count, and descriptor bytes are application contracts. -- Do not expose raw SQLite, authority, replica, local paths, or arbitrary - operation IDs through the author handle. -- Layout: the integration suites are `tests/reference_application.rs` (with - `tests/reference_application/`) and `tests/contracts.rs`; `src/tests.rs` is the - only in-src test location and is listed in `tests-allow-list.txt`. -- Run `python3 crab/scripts/check-cell-ltx-layout.py` after layout changes. -- Run the crate tests, format, Clippy, and dependency-tree checks after API - changes. diff --git a/crates/crab-cell-app/CLAUDE.md b/crates/crab-cell-app/CLAUDE.md deleted file mode 120000 index 47dc3e3d8..000000000 --- a/crates/crab-cell-app/CLAUDE.md +++ /dev/null @@ -1 +0,0 @@ -AGENTS.md \ No newline at end of file diff --git a/crates/crab-cell-app/Cargo.toml b/crates/crab-cell-app/Cargo.toml deleted file mode 100644 index fd487345f..000000000 --- a/crates/crab-cell-app/Cargo.toml +++ /dev/null @@ -1,27 +0,0 @@ -[package] -name = "crab-cell-app" -version = "0.1.0" -edition.workspace = true -license.workspace = true -repository.workspace = true -rust-version.workspace = true -publish = false -description = "Dependency-light author API for Crab Cell applications" - -[dependencies] -blake3.workspace = true -crab-cell-runtime.workspace = true - -[dev-dependencies] -async-trait.workspace = true -crab-cell-host.workspace = true -crab-cell-runtime = { workspace = true, features = ["test-support"] } -crab-ltx = { workspace = true, features = ["replica"] } -crab-storage.workspace = true -ed25519-dalek.workspace = true -futures-util.workspace = true -object_store.workspace = true -tempfile.workspace = true -tokio = { workspace = true, features = ["io-util", "macros", "net", "rt-multi-thread"] } -tokio-util.workspace = true -tracing-subscriber = "0.3" diff --git a/crates/crab-cell-app/PERFORMANCE.md b/crates/crab-cell-app/PERFORMANCE.md deleted file mode 100644 index 57c185ae8..000000000 --- a/crates/crab-cell-app/PERFORMANCE.md +++ /dev/null @@ -1,528 +0,0 @@ -# Cell primitive end-to-end performance - -## Entity targeting correctness - -The entity-ledger gate provisions twelve independent SQL Cells, four on each -of three public `CellNode` hosts. Generated clients use a signed TCP balancer -to prepare commands, resolve durable outcomes, replay duplicates, and read -each Cell at its returned receipt. Reusing one request identity across all -twelve Cells must produce twelve independent records. Stored authority must -confirm 4/4/4 ownership and published roots covering the receipts; each gateway -must execute local and forwarded calls. All node sessions renew and withdraw. - -Run against an isolated RustFS bucket with a fresh prefix: - -```sh -AWS_ACCESS_KEY_ID=crab AWS_SECRET_ACCESS_KEY=crab \ -CRAB_CELL_TEST_ENDPOINT=http://127.0.0.1:9000 \ -CRAB_CELL_TEST_BUCKET=crab-reference-app \ -CRAB_CELL_TEST_PREFIX=entity-ledgers-unique-run \ -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/" \ - cargo test -p crab-cell-app --locked --test reference_application \ - entity_ledgers_are_isolated_across_three_rustfs_hosts -- --ignored --nocapture -``` - -The Compose workflow runs this gate as `driver entities`, retaining -`entities.log`, binary identity, and kernel counters. Its three hosts share -one process and have separate SQLite/cache directories. This is a correctness -check, not a throughput measurement or the many-Cell 3/5/10/20 process profile. - -## Scheduled writable entity processes - -After preparing the source archive and Linux binary as described under -[three constrained Compose nodes](#three-constrained-compose-nodes), run: - -```sh -python3 "$CRAB_REFERENCE_STATE/source/crates/crab-cell-app/qualification/scale.py" \ - --state "$CRAB_REFERENCE_STATE" --project entity-fleet-unique-run \ - --workload entities -``` - -Every node has four writable entity Cells and a private SQLite/WAL/cache volume. -The fleet grows from 3 to 5, 10, and 20 constrained containers; its workload grows -from 12 to 20, 40, and 80 Cells. Existing Cells retain their owners. This measures -adding owners and workload, not automatic redistribution of a fixed dataset. -Gateways publish the expanded routing table before each stage starts, and the -driver independently checks each stored owner, epoch, incarnation, and target. - -Each stage runs three shapes at 1, 4, and 16 scheduled actions per node per second, -with global client concurrency bounded at 4, 16, and 64 respectively. Each point -offers ten seconds of arrivals, then drains admitted calls. A write action -includes its receipt-bound readback; a read action queries the current owner. -Uniform traffic cycles through all Cells. Hot traffic directs 80% of writes to -one Cell and distributes the rest across the other Cells. Skewed traffic sends -80% reads to four Cells and distributes 20% writes across the fleet. - -The driver records every scheduled arrival, including late starts, client -saturation, pre-dispatch failures, ambiguous outcomes and resolution, write -receipts, query receipts, and values. Completed samples flush as they arrive so -an interrupted run retains partial evidence. The independent Python verifier -checks exact per-Cell read prefixes and final counts against all acknowledged -write sequences. It reports service and arrival latency, achieved throughput -including drain time, and whether each point served every planned arrival. -An integrity pass does not turn an overloaded point into supported capacity. - -`evidence/entity-scaling/control` retains raw TSVs for all 36 points and one-second -node samples: cgroup CPU/throttling and memory, logical local file bytes, active -Cells, admitted SQL/primitive/hydration jobs, retained/disk reservations, gateway -calls, and cumulative object-store operations/bytes. Object counts are logical -backend operations; provider-internal retries are not counted separately. -Per-window resource deltas state their actual sample bounds. Admitted job counts -are not queue-depth measurements. Raw object durability waits are retained per -node. The current host profile uses object durability and owner reads; it does -not measure follower durability or sparse replica-query capacity. - -Action durations use `Instant`. Resource samples and windows align using the -shared Linux boot clock from [`/proc/uptime`](https://github.com/torvalds/linux/blob/master/fs/proc/uptime.c), -whose centisecond precision is sufficient for one-second resource samples. -Wall timestamps are retained with their observed adjustment, but do not govern -duration or resource-window validation. Node logs retain runtime warnings, -including SQL deadlines and the error that causes publication, compaction, -renewal, or post-commit execution to fence a Cell. - -The Compose workflow runs this profile after reader qualification. Its resource -limits remain 1 CPU/1 GiB per node, while all containers share the recorded Docker -host. Owner-loss recovery, continuous container/schema rollout, workflow and -read-model traffic, actual queue depth, repeated capacity runs, and isolated -multi-host fault domains remain separate qualification gates. - -### Qualification status (2026-09-27) - -The first Colima/RustFS run used source -`0d4bc19e1c19e63eb226d68e8bed18361d37fef5`. All nine three-node windows -completed, including receipt readback and published-root checks for twelve -Cells. The run then failed during the five-node hot workload at 16 scheduled -actions per node per second: a mutation for entity 13 remained unresolved and -its owner's active Cell count fell from four to three. Ten- and twenty-node -entity stages were not reached. The fencing cause remains unproven. - -The same run exposed an independent verifier error: its ten-second duration -check used wall time during a clock correction. A regression now checks the -monotonic window while retaining the wall-clock adjustment. Source -`62e45a44caed99e4ddf9500f4c41a230de9b6f2a` also adds fencing-cause warnings -and retains RustFS file logs. That source passed 23 native reference tests, -16 Python verifier tests, strict app/runtime Clippy and the Linux release build. -Its diagnostic Compose repeat has not been run. These results do not establish -a passing 3/5/10/20 writable-entity profile or supported throughput limits. - -## Additive application release correctness - -The public-host rollout test compiles a successor SQL module with a new typed -receipt-payload query and retains the predecessor code. An old generated client -uses TCP to invoke the successor host while the successor client writes the -same retained Cell locally. After both writes acknowledge, an operator publishes -the code-only migration. A prepared old capability and a predecessor client -must fail before execution. The upgraded generated client then reads the added -query over signed TCP, replays an old request with its original receipt, and -writes another receipt. A fresh host restores the published root in a separate -SQLite directory, replays the request again without duplication, and writes -successfully. Four visible receipts must remain. - -Run the real-provider version against an existing isolated RustFS bucket and a -fresh prefix: - -```sh -AWS_ACCESS_KEY_ID=crab AWS_SECRET_ACCESS_KEY=crab \ -CRAB_CELL_TEST_ENDPOINT=http://127.0.0.1:9000 \ -CRAB_CELL_TEST_BUCKET=crab-reference-app \ -CRAB_CELL_PERF_PROCESS_ROOT=additive-rollout-unique-run \ -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/" \ - cargo test -p crab-cell-app --locked --test reference_application \ - three_node_host_rustfs_additive_code_rollout -- --ignored --nocapture -``` - -The Compose smoke runs this same gate in a separate constrained driver container -before stopping RustFS, retaining `rollout.log`, its binary hash, and kernel -resource counters. The three initial hosts share that process; one host is -retired before its replacement starts. The schema stays at version one, and -traffic pauses for code publication and recovery. This gate does not establish -rolling-container availability, sustained throughput, or a migration latency -bound. Independent-process scale and fault qualification remain separate. - -### Local GA RustFS receipt, 2026-09-27 - -Source `ea51218b45c7401fac600f2e35aaa75aff30dc35` passed with the digest-pinned -Rust 1.97 and RustFS 1.0.0 images in `qualification/compose.yaml`. A fresh Linux -release build took 3m43s. The binary SHA-256 was -`0dfdedef7b47a28833ba608915149e3f2f7f915b7a1a8227efaf58ef0130fd52`. -Colima supplied four CPUs and 8,307,101,696 bytes of VM memory; each of the -three node containers and both successive driver containers enforced one CPU, -1,073,741,824 bytes of memory and no swap. RustFS shared that VM. - -The ordinary three-node smoke passed in 14.93s: 180 complete primitive actions, -one duplicate generated command, twelve replica reads split six per reader, -automatic recruitment/refresh and target-zero eviction. Balancer ingress was -524/524/524; every node performed local and forwarded work. The separately -invoked rollout gate passed in 0.48s with four visible application receipts, -two exact duplicate replays and fresh-host recovery. This short smoke is not -an offered-rate capacity or availability result. - -All three node processes withdrew their sessions. Nodes, drivers and bucket -initialization exited zero; RustFS was stopped after evidence capture. Kernel -counters recorded no OOM or CPU throttling. Node memory peaks were -12,853,248–29,618,176 bytes; rollout driver peak was 27,291,648 bytes. Raw logs, -kernel samples, image/container inspections and source/binary identities are -retained under `additive-rollout-ea51218-20260927/evidence` in the external -qualification state directory. Evidence SHA-256 values: - -| Artifact | SHA-256 | -| --- | --- | -| `driver.log` | `23bb78bfed1311c8af57e9775af6e743e0f0d35f990058b8c064d5cb2bfd17de` | -| `rollout.log` | `ac481febb24d076295fed3c8582ed35ec917a3de3475ac210ef07481cba5a2b0` | -| `containers.json` | `ed656bffd21f614cc83ffe14f56f3b4b6941a09791eab99c9968beb6d4d1d4ef` | -| `verification.json` | `9ad897b570edca4ea05e3a2573431cd94ee3c8a0af46e35f91f1687ab47eac7c` | - -## Generated client action and recovery slices - -The ignored `reference_public_host_action_performance` test runs the generated -reference SQL client on three `CellNode` hosts. Each local or forwarded action -prepares one typed command, waits for its published receipt, and verifies the -row count through a typed query at that receipt. The forwarded lane crosses a -signed peer gateway and then the owner over loopback TCP. The test reports -separate full-action and `execute()`-to-durable-ack distributions. The latter -includes handler execution, transport on the forwarded lane, and publication; -it does not isolate object-store waiting from those costs. The runtime's -`durability_proof` telemetry separately records post-commit submission through -object durability proof, including publication queue time. Recovery timing runs -from owner fencing through authority takeover, exact-root restore, and the -first verified read. It is one recovery sample, so no percentile is reported. - -```bash -CRAB_CELL_PERF_ITERATIONS=100 \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-my-worktree \ - cargo test -p crab-cell-app --test reference_application \ - public_host::reference_public_host_action_performance \ - --release --locked -- --ignored --nocapture -``` - -The local store is in memory, routing is static, requests are serial within -each lane, and one process hosts all three nodes. These results measure the -reference action path and do not establish a supported Cell count, aggregate -throughput, cloud durability latency, or takeover SLO. -The two local release runs are recorded in -[`performance/2026-09-25-public-host-action.md`](performance/2026-09-25-public-host-action.md). - -The ignored `reference_primitive_end_to_end_performance` test measures complete, -serial user actions through a compiled `crab-cell-app` handle, the local Cell -router, SQLite actors, LTX publication, and read-back verification. Each result -is one verified action, not one primitive method call. The benchmark creates -seven Cells before the timer starts and uses a temporary local directory plus -the in-memory object store. Cron effect delivery traverses signed peer request -encoding, verification, dispatch, and the destination Cell through a loopback -transport. No HTTP listener, network, cloud object store, failover, or concurrent -client load is measured. -The Workflow handler is the reference application's in-process echo handler; -its result verifies durable activity completion but does not simulate an -external service call. - -| Result | Timed action and observed side effect | -| --- | --- | -| `sql_order_insert_read` | Insert an order with a parameterized statement; read the total at its commit receipt. | -| `kv_cart_put_get` | Save a cart value; read the exact bytes at its commit receipt. | -| `blob_attachment_upload_read_32k` | Begin, upload and commit a 32 KiB binary attachment; read and compare all bytes. | -| `queue_notification_send_claim_ack` | Send a notification, claim and validate its lease, then acknowledge it. | -| `workflow_fulfillment_activity` | Start a fulfillment workflow, execute its registered native activity, then read terminal state. | -| `cron_invoice_schedule_deliver` | Register a due invoice schedule, wait for its due time, run maintenance, deliver its published effect through the peer dispatcher, then read the SQL receipt at the destination. | - -Run from the repository root with a target directory unique to the checkout on -the mounted Workspace volume: - -```bash -CRAB_CELL_PERF_ITERATIONS=100 \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-my-worktree \ - cargo test -p crab-cell-app --test reference_application \ - performance::reference_primitive_end_to_end_performance \ - --release --locked -- --ignored --nocapture -``` - -The iteration count is 30 by default and accepts 1 through 1,000. Results print -one `PERF` line per action with sample count, total elapsed time, verified -actions per second, and nearest-rank p50/p95/p99/max action latency. Execution -is serial and includes all method calls, verification, and the 6 ms intentional -Cron due-time wait. It excludes Cell bootstrap and shutdown. Capture the source -revision, Rust profile, CPU, storage, iteration count, and raw output with every -report. These local numbers are development evidence, not a cloud capacity or -release qualification result. Two measured runs are recorded in -[`performance/2026-09-21-local.md`](performance/2026-09-21-local.md). - -## Three-runtime fleet workload - -The second ignored test places the seven reference Cells across three -independent runtimes. Six primitive lanes run concurrently through signed peer -requests over loopback TCP, with 100 verified actions per lane when the -iteration count is 100. It reports each lane and the fleet's combined action -latency and throughput: - -```bash -CRAB_CELL_PERF_ITERATIONS=100 \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-my-worktree \ - cargo test -p crab-cell-app --test reference_application \ - performance::reference_three_node_fleet_end_to_end_performance \ - --release --locked -- --ignored --nocapture -``` - -The runtimes share one process and an in-memory object store. Static Cell-ID -routing uses the local TCP stack; this run does not include product ingress, -dynamic placement, mTLS, or provider network latency. The measured topology, -workload, and two runs are in -[`performance/2026-09-21-three-node-local.md`](performance/2026-09-21-three-node-local.md). - -## Three-process RustFS workload - -The process benchmark runs the six concurrent action lanes through three -public `CellNode` hosts. Each process owns its SQLite workers, WAL/cache -files, signed renewable node session, and ordered shutdown. RustFS stores -Cell authority, LTX objects, and session records. The parent binds public -application handles; generated stable-ID commands additionally prove that -repeating a committed mutation preserves its receipt and creates one visible -effect. Every primitive lane checks its visible result. - -After the measured action lanes, the generated-client proof admits two SQL -snapshot readers on the other node processes through the host-owned -`ReadReplicaManager`. It verifies missing readers do not fall back to the owner, -then issues a new owner mutation twice. The supervisor must discover its exact -newer receipt without a fixture refresh hint; both readers return one additional effect. -Changing the desired-reader policy to zero must evict both views automatically. -Each node proves its retained manager rejects activation and resolution after -host shutdown. Initial admission uses signed owner hints from the host-owned -recruiter; marker files only observe readiness. - -Successful direct peer replies are counted by selected physical node: six -per reader, zero on the writer. Each receiver executes the explicit query -against its local admitted snapshot, including gateway receivers. The fixture -observes each reader's receipt before testing the new position, so background -refresh timing cannot make a stale-position assertion flaky. Deterministic -stale/minimum-receipt checks remain in the runtime suite and the earlier -source-bound Compose report. These steps are outside the action timer and do -not establish replica throughput. - -Provide an isolated bucket, a prefix, and explicit credentials: - -```bash -AWS_ACCESS_KEY_ID=crab AWS_SECRET_ACCESS_KEY=crab \ -CRAB_CELL_TEST_ENDPOINT=http://127.0.0.1:9000 \ -CRAB_CELL_TEST_BUCKET=crab-reference-app \ -CRAB_CELL_TEST_PREFIX=reference-performance \ -CRAB_CELL_PERF_ITERATIONS=30 \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-my-worktree \ - cargo test -p crab-cell-app --test reference_application \ - process_performance::reference_balanced_three_process_fleet_end_to_end_performance \ - --release --locked -- --ignored --nocapture -``` - -Each invocation adds a unique object prefix. The balanced variant sends every -request to a round-robin TCP gateway. Each receiving node verifies the peer -signature, dispatches locally or forwards once to the owner. The test requires -local and forwarded results on every node. The direct variant, -`reference_three_process_fleet_end_to_end_performance`, omits the gateway. -Both measure complete actions; setup and shutdown remain outside the action -timer. Node object-proof wait samples include the whole process lifetime. - -The historical September 21 filesystem measurements remain in -[`performance/2026-09-21-three-process-local.md`](performance/2026-09-21-three-process-local.md) -and [`performance/2026-09-21-balanced-three-process-local.md`](performance/2026-09-21-balanced-three-process-local.md). -They do not describe the current RustFS path. - -### Reader recruitment and process loss - -The ignored `owner_replaces_killed_reader_through_public_hosts` case uses the -same GA RustFS environment as the process runs. It starts three public hosts, -exercises the reference primitives, admits two readers through signed owner -hints, then starts two more independent processes. It kills one selected -reader, waits for two current readers, and verifies twelve generated queries -against the acknowledged receipt. Writer session, epoch, and incarnation must -remain unchanged. The four survivors must drain successfully. - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/" \ - CRAB_CELL_PERF_ITERATIONS=5 cargo test -p crab-cell-app --locked \ - --test reference_application owner_replaces_killed_reader_through_public_hosts \ - -- --ignored --nocapture -``` - -This native fault run does not impose container CPU or memory limits. The -three-node Compose proof below independently verifies the shared recruitment -path under those limits. Neither run establishes sustained capacity or a -recovery SLO. - -## Three constrained Compose nodes - -[`qualification/compose.yaml`](qualification/compose.yaml) pins GA RustFS 1.0.0 -and the build image by digest. It runs three independent nodes plus a driver -containing the TCP balancer. Each node is limited to one CPU, 1 GiB memory, -zero swap, and 256 processes. Each has a private disk-backed Docker volume for -SQLite/WAL/cache; the shared evidence directory holds control markers and -reports. The `crab`/`crab` credentials are local fixture credentials. No service -publishes a host port. -The disposable evidence directory is shared and writable by the host and all -fixture containers; its sticky bit protects entries owned by another UID. -This also permits capability-free container root to create logs on a Linux -runner-owned bind mount. Source and binary mounts remain read-only. - -The worker pool retains its 32-writer ceiling and explicitly reserves 32 MiB -for native admission, covering a reader's overlapping refresh snapshots. -The default 32-writer budget alone is 2 MiB and correctly refuses a 12 MiB -snapshot; raising the writer count is not required to provision reader memory. - -Use a fresh Compose project and state directory per run. The source archive -must contain the committed change being measured. The selected Docker/Colima -VM must mount the external state directory and have space for the build: - -```bash -export CRAB_REFERENCE_STATE="$HOME/Workspace/crabbuild-target/crab-my-worktree/reference-$(git rev-parse --short HEAD)-$(date +%s)" -export CRAB_REFERENCE_PROJECT="crab-reference-$(date +%s)" -mkdir -p "$CRAB_REFERENCE_STATE/source" "$CRAB_REFERENCE_STATE/target-linux" "$CRAB_REFERENCE_STATE/evidence" -chmod 1777 "$CRAB_REFERENCE_STATE/evidence" -git archive HEAD | tar -x -C "$CRAB_REFERENCE_STATE/source" -git rev-parse HEAD > "$CRAB_REFERENCE_STATE/evidence/source-revision.txt" -compose() { - docker compose -p "$CRAB_REFERENCE_PROJECT" \ - -f "$CRAB_REFERENCE_STATE/source/crates/crab-cell-app/qualification/compose.yaml" "$@" -} -compose run --rm build -compose up -d node-0 node-1 node-2 -compose run --name "$CRAB_REFERENCE_PROJECT-driver" driver -compose ps -a -compose logs --no-color > "$CRAB_REFERENCE_STATE/evidence/compose.log" -docker inspect $(compose ps -aq) > "$CRAB_REFERENCE_STATE/evidence/containers.json" -``` - -The wrapper verifies the actual cgroup CPU/memory/swap limits and exactly one -executed test. It retains the binary hash, per-node kernel CPU and peak-memory -counters, active Cell and retained-byte observations, object-proof wait -samples, action timings, duplicate-delivery proof, and gateway counts. Require -all three nodes and the driver to exit zero. A driver failure leaves nodes -available for diagnosis; use `compose stop` after collecting evidence. Remove -only this project's containers/volumes with `compose down -v` when their -artifacts are no longer needed. - -The [initial 2026-09-27 run](performance/2026-09-27-three-node-compose.md) -records the owner action path. The -[generated replica-read follow-up](performance/2026-09-27-replica-compose.md) -records two non-owner readers, stale/minimum-receipt checks, controlled refresh, -source and binary identity, and container resource evidence on GA RustFS. -The [host-owned reader run](performance/2026-09-27-host-readers-compose.md) -then proves automatic refresh, target-zero eviction, and terminal drain through -the manager shared with the product server. -The [public-host recruitment run](performance/2026-09-27-reader-recruitment.md) -adds automatic signed placement on three constrained nodes and a separate -native three-to-five-process test that replaces a killed reader. The latter -does not impose per-process resource limits. - -This is an application integration smoke with six closed-loop lanes and seven -Cells assigned 3/2/2. Ingress counts are even; owner load follows the action -mix. It does not establish a supported throughput, 5/10/20-node application -capacity, mTLS, failure-domain isolation, replica-query throughput, or recovery -during arrivals. The issue-service fleet qualification -covers its separate product ingress and scaling paths. - -### Constrained reader scaling and loss - -After building the archived source with the procedure above, run the controller -from that same archive with a fresh Compose project: - -```bash -python3 "$CRAB_REFERENCE_STATE/source/crates/crab-cell-app/qualification/scale.py" \ - --state "$CRAB_REFERENCE_STATE" --project "$CRAB_REFERENCE_PROJECT-scale" -``` - -The controller writes a resolved Compose configuration into -`evidence/scaling/`, starts the driver, and follows its bounded scale/fault -requests. Every node and the driver inherit the one CPU / 1 GiB / zero-swap -profile. The driver grows one fleet through 3, 5, 10 and 20 live nodes. The -seven writer Cells remain on the original three owners; new nodes participate -as gateways and admitted readers for the reference SQL Cell. - -At each size, generated commands publish two new receipts with duplicate -delivery checks. Both initial recruitment and a later refresh must become -ready automatically. The driver verifies 30 exact queries per selected reader, -records serial read latency, and separately sends owner reads through a TCP -balancer whose entry counts must differ by at most one. Replica requests go -directly to the selected readers through the signed peer transport; this -profile does not place a second balancer in that path. - -Each stage also measures sixty seconds of concurrent refresh and queries. -One generated writer schedules five unique mutations per second through the -balancer. It records missed arrivals instead of issuing a catch-up burst. -Eight closed-loop readers run concurrently: four require the latest -acknowledged receipt and four permit older snapshots. Typed `ReplicaBehind` -responses are counted separately; other query errors fail the run. Every -successful snapshot count must match the acknowledged command history at its -actual receipt. All selected readers must serve queries, and the writer must -remain excluded. The final owner read and all selected readers must cover the -last acknowledgement before the next stage. - -Raw per-command and per-reader TSV files live in `evidence/scaling/control/`. -The controller independently checks their scheduled arrivals, exact counts, -minimum receipts, lag, and completeness, and binds them by SHA-256 in -`verification.json`. `fully_served_writes` is false when any scheduled write -was missed, even if all admitted work remains correct. Query latency excludes -typed behind responses; their count remains visible. This is a fixed mixed -workload, not a saturation curve or a freshness guarantee. - -At five nodes, three readers and one spare are eligible. A separate sixty-second -window keeps the same five scheduled writes/second and eight read lanes active. -Ten seconds into that window, the controller kills a selected reader container -and proves exit 137 without an OOM event. The writer uses the original three -gateways throughout this window; gateway removal and owner failure remain -separate qualifications. Replica reads keep using public selection and its -bounded attempts across the selected readers. Unexpected errors fail the run. - -The driver requires an automatically recruited replacement to serve workload -queries before second fifty, leaving at least ten seconds of subsequent load. -After the window, every reader must cover the final acknowledged receipt, and -twelve exact queries check the replacement set. Raw `reader_loss-5-*.tsv` files -and `reader-loss.tsv` retain the workload and fault timeline. The independent -verifier requires successful writes and queries from every read lane wholly -before, during, and after replacement. It reports phase counts and latency, -checks receipt/value correspondence, and preserves missed scheduled writes. -A request spanning the outage cannot count as service during replacement. - -The fleet temporarily has four survivors before growing to ten. The killed -boot is never restarted with the fixture's deterministic identity: growth -creates a new node/session, yielding twenty live nodes out of twenty-one -created containers. Writer ownership must remain unchanged. Target zero then -evicts every view, and all surviving hosts must withdraw and drain. - -The controller requires successful exact tests, distinct scratch volumes, -matching binary hashes, actual Docker/cgroup limits and no OOM events. It -retains logs, fault events, resource counters and `verification.json`, then -stops only its project. Failed runs retain their containers and evidence. -An optional repeated `--compose-file` supplies explicit image/cache overrides -to the same source configuration; the resolved result is retained. - -This combines a scaling/failure smoke and bounded mixed-workload measurements. -One replicated SQL Cell and one killed reader do not establish many-Cell -capacity, gateway-loss handling, arbitrary fault availability or an owner-loss SLO. A one-CPU cap -does not reserve a physical core; record Docker VM resources and contention -before comparing latency across sizes. The Compose CI runs this profile after -the three-node lifecycle smoke using the same compiled binary. -The [2026-09-27 run](performance/2026-09-27-reader-scaling.md) records all four -sizes, 990 measured generated reads, a 34.830-second reader replacement sample, -roughly five-second refresh observations, and successful drain of all twenty -survivors. Query latency and freshness are reported separately. -The [reader-loss-under-load run](performance/2026-09-27-reader-loss-during-load.md) -records two reproduced availability failures, their shared routing and host -recruitment fixes, and a passing raw-data progress check on GA RustFS. All eight -lanes made successful queries during replacement. Its missed writes and host -contention preclude a supported throughput claim. - -The [mixed-read/write run](performance/2026-09-27-mixed-readers.md) records -691,427 correct replica reads across four sixty-second windows, with explicit -behind responses and 144 missed scheduled writes. It establishes concurrent -snapshot correctness for that workload, not a supported mixed-load capacity. -The same report records a fresh current-main integration run with 703,981 -exact reads, 1,172 acknowledged writes and 28 missed arrivals. Only its -twenty-node window fully served the offered writes; the capacity limit remains -unqualified. - -## Native RustFS sanity check - -On 2026-09-22, the ignored typed primitive smoke was also run against a fresh -native RustFS instance and an isolated bucket/prefix. The run reached the -object-store-backed workload but failed the PR profile's measured latency -envelope after 377.94 seconds; RustFS reported roughly five-second commits for -small control objects on the mounted workspace volume. No receipt was emitted -and this is not provider qualification evidence. It is retained here as a -failed environment check so a slow local backend cannot be mistaken for a -passing production profile. diff --git a/crates/crab-cell-app/README.md b/crates/crab-cell-app/README.md deleted file mode 100644 index 76ce3d984..000000000 --- a/crates/crab-cell-app/README.md +++ /dev/null @@ -1,92 +0,0 @@ -# crab-cell-app - -Author-facing composition for statically linked Cell applications. The crate -compiles a runtime registry together with a bounded, deterministic topology -descriptor, so an application hands the host one `CompiledApplication` instead -of assembling modules, namespaces, and cell types itself. Node lifecycle, -storage providers, authority, and HTTP policy stay outside this boundary in -`crab-http-server`. - -`ApplicationHandle::new` checks the author type's application name and the client's compiled -registry digest against the hosted application before any typed call can start. -`CellNode::application_handle` returns the same binding error to its caller. - -`cell_client!` generates scoped Rust accessors and typed command, prepare, query, -and pending-resolution methods from explicit stable namespace and operation IDs. -Its constructor checks each binding against the compiled registry. Accessors -accept only their declared `CellKey` type and derive canonical shard targets -from its stable encoding and the application's `CellType` descriptor. -The reference application exercises generated calls through `CellNode` hosts; -the macro documentation includes valid and compile-fail examples. - -## Explicit replica reads - -Typed queries default to `crab_cell_runtime::client::ReadPolicy::CurrentOwner`. -Authors can pass `handle.with_read_policy(ReadPolicy::Replica)` to a generated -client constructor. Its queries then return an admitted snapshot's actual -receipt; an optional minimum still checks Cell, incarnation, and sequence. -Missing readers return `ReplicaUnavailable`, and a lagging view returns -`ReplicaBehind`. Neither outcome falls back to the owner. Commands, mutation -resolution, state streams, and Queue/Effects/Workflow activity lease validation -retain owner ordering. A stale view cannot prove a claim is still valid for -external work. - -The host first wires `CellClient::with_read_replicas` with a shared -`ReplicaReadRouter`, an authenticated `ReplicaPeerClient`, and optionally its -local admitted-view resolver. The author handle receives none of those storage -or transport capabilities. The router uses S3 desired counts and signed live -membership, shares outstanding-attempt counts across client clones, and keeps -selection and retries within one five-second deadline. The object durability -profile is required for the all-node-loss guarantee. - -## Module map - -| Module | Responsibility | -| --- | --- | -| `lib.rs` | Application builder, cell-type bindings, compiled application, and the author handle | -| `client_macro.rs` | Generated scoped client and operation methods | -| `tests.rs` | In-src tests for the private registry validation path, listed in [`tests-allow-list.txt`](tests-allow-list.txt) | - -## Tests - -`tests/` holds two binaries: `reference_application` (application binding, -commits, fleet behavior, primitives, and process performance, with shared -fixtures in `tests/reference_application/harness.rs`) and `contracts` (the -descriptor, digest, and identity contracts). -The reference suite also runs three `CellNode` hosts to test ambiguous results, -duplicate delivery, owner loss, and an additive application release through -the public APIs. The release adds a generated query, retains predecessor code -during client overlap, publishes the new code, fences old capabilities, and -recovers exact receipts on a fresh host. Its RustFS variant also runs in the -Compose workflow; this rollout gate hosts its runtimes in one process and -does not qualify continuous traffic across rolling container replacements. -The public-host reader checks also verify publication-triggered recruitment -and cancellation of a storage-stalled activation during drain. - -`CellType::new` declares a fixed-shard namespace. For independently addressed -entity Cells, declare one namespace shard and call -`CellType::with_entity_partitions`. Generated accessors and -`ApplicationHandle::target_for_scope` derive the canonical 33-byte entity -partition from the typed key. `CellType::entity_partition` exposes the same -derivation for host provisioning; the application handle validates that encoding. -The entity mode is part of the immutable application descriptor. It does not -split SQLite state or remove Cell catalog and host admission bounds. -Explicit SQL and effect capabilities accept entity targets. Namespace-level -KV, Blob, Queue, Cron, Workflow, and activity capabilities require fixed shards -and reject entity declarations before a call can start. - -The entity-ledger fixture provisions twelve SQL Cells across three public -hosts, uses generated operations through the signed peer balancer, and verifies -independent request ledgers, exact duplicate receipts, visible readback, and -4/4/4 ownership read from object-store authority. Its ignored RustFS variant -runs in the Compose workflow. These hosts share one process; this fixture is -a correctness gate before the many-Cell process scaling workload. - -```sh -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/ \ - cargo test -p crab-cell-app --locked -``` - -See `AGENTS.md` for contributor rules and -`crates/crab-cell-runtime/docs/application-framework.md` for the framework -contract this crate implements. diff --git a/crates/crab-cell-app/examples/authoring.rs b/crates/crab-cell-app/examples/authoring.rs deleted file mode 100644 index e43d432a7..000000000 --- a/crates/crab-cell-app/examples/authoring.rs +++ /dev/null @@ -1,95 +0,0 @@ -//! Compiles one application the way an author does. -//! -//! Run with `cargo run -p crab-cell-app --example authoring`. A host takes the -//! finished `CompiledApplication`; building it here keeps the documented -//! authoring flow compiling against the published contract. - -use std::sync::{Arc, OnceLock}; - -use crab_cell_app::{ApplicationBuilder, CellType, CompiledApplication}; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::identity::{Digest, NamespaceId}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, MigrationDescriptor, ModuleDescriptor, NamespaceDescriptor, - RegistryBuilder, -}; - -/// The module an application registers before a host can serve it. -struct Repository; - -/// The namespace this module owns. A descriptor borrows its lists for the -/// process, so they live in `static` items rather than in the initializer. -static NAMESPACES: [NamespaceDescriptor; 1] = [NamespaceDescriptor { - id: NamespaceId::from_bytes([2; 16]), - name: "repository", - role: CatalogRole::Sql, - shards: 1, - effect_targets: &[], - dead_letter: None, -}]; - -/// The single statement that installs the module's schema. -const MIGRATION_SQL: &str = "-- repository migration v1"; - -impl CellModule for Repository { - const NAME: &'static str = "repository"; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: OnceLock = OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: "repository", - source_digest: Digest::from_bytes([1; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - // A module must install its own schema, and the registry pins each - // migration to the exact statement: the digest is its BLAKE3 hash. - migrations: Box::leak( - vec![MigrationDescriptor { - version: 1, - sql: MIGRATION_SQL, - digest: Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - }] - .into_boxed_slice(), - ), - commands: &[], - queries: &[], - workflow_definitions: &[], - activity_types: &[], - namespaces: &NAMESPACES, - }) - } - - /// Native bindings for this module's commands and queries live here. This - /// example registers none, so the module declares its namespace only. - fn register(self, _registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - Ok(()) - } -} - -fn main() -> crab_cell_runtime::Result<()> { - let mut builder = ApplicationBuilder::new( - "repository", - BuildDescriptor { - source_revision: "authoring-example".into(), - cargo_lock_digest: Digest::from_bytes([3; 32]), - }, - )?; - builder.register(Repository)?; - builder.cell_type(CellType::new( - "repository", - "repository", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - )?)?; - - let application: Arc = Arc::new(builder.finish()?); - println!( - "compiled {} with {} cell type(s), digest {:?}", - application.name(), - application.cell_types().len(), - application.descriptor_digest() - ); - Ok(()) -} diff --git a/crates/crab-cell-app/performance/2026-09-21-balanced-three-process-local.md b/crates/crab-cell-app/performance/2026-09-21-balanced-three-process-local.md deleted file mode 100644 index 4363437cc..000000000 --- a/crates/crab-cell-app/performance/2026-09-21-balanced-three-process-local.md +++ /dev/null @@ -1,73 +0,0 @@ -# Cell balanced three-process fleet performance — 2026-09-21 - -Measured code: `752ad2c0067`. - -The load generator and load balancer ran in one process, with three separate -Cell owner processes on one Apple M2 Max host (Darwin 25.5.0 arm64, Rust -1.97.0). Each owner had its own node session, SQLite worker pool, and database -directory. Seven Cells were placed across the owners as in the -[direct-owner report](2026-09-21-three-process-local.md). All processes shared -the test-only filesystem CAS object store. - -The load generator sent signed Cell requests to a loopback TCP balancer. The -balancer forwarded each request to one of the three owners in round-robin -order without changing the signed bytes. Every owner verified the request and -served a locally owned Cell or forwarded it once over loopback TCP to the -owner in a static Cell-ID map. The owner then dispatched through the Cell actor. -The client-to-balancer, balancer-to-entry, and any entry-to-owner round trips -are inside the action timer. Every peer request opens a new TCP connection. -The balancer ran as a task in the load-generator process, not as an external -service. - -The six actions in `PERFORMANCE.md` ran as concurrent lanes, with 100 serial -actions per lane. Each action included its visible result check, including SQL -read-back for Cron effect delivery. Cron included the intentional 6 ms due-time -wait. Bootstrap and shutdown were outside the timer. Fleet throughput divides -600 verified actions by the wall time until all lanes finish; fleet latency -percentiles pool all 600 samples. At most six actions were in flight, so this -finite workload is not a saturation or production capacity measurement. - -Run from the repository root with an external target directory unique to the -checkout: - -```bash -CRAB_CELL_PERF_ITERATIONS=100 \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-89be5c6d \ - cargo test -p crab-cell-app --test reference_application \ - process_performance::reference_balanced_three_process_fleet_end_to_end_performance \ - --release --locked -- --ignored --nocapture -``` - -| Verified action | Run | actions/s | p50 ms | p95 ms | p99 ms | max ms | -| --- | ---: | ---: | ---: | ---: | ---: | ---: | -| SQL order insert and read | 1 | 41.77 | 14.886 | 57.259 | 210.677 | 245.825 | -| SQL order insert and read | 2 | 52.85 | 16.102 | 36.387 | 59.547 | 71.671 | -| KV cart put and get | 1 | 49.25 | 12.244 | 40.209 | 228.303 | 355.001 | -| KV cart put and get | 2 | 67.87 | 12.582 | 24.601 | 49.140 | 70.371 | -| Blob attachment upload and read, 32 KiB | 1 | 22.84 | 33.561 | 84.616 | 234.761 | 306.165 | -| Blob attachment upload and read, 32 KiB | 2 | 22.86 | 36.549 | 77.706 | 92.300 | 454.294 | -| Queue notification send, claim, acknowledge | 1 | 23.66 | 32.874 | 74.543 | 241.752 | 299.029 | -| Queue notification send, claim, acknowledge | 2 | 25.91 | 34.079 | 68.093 | 82.974 | 112.842 | -| Workflow start, native activity, terminal read | 1 | 21.96 | 36.825 | 88.341 | 256.105 | 268.218 | -| Workflow start, native activity, terminal read | 2 | 22.57 | 35.291 | 82.797 | 111.565 | 415.548 | -| Cron schedule, tick, effect delivery, SQL read | 1 | 15.67 | 49.818 | 137.387 | 327.909 | 358.473 | -| Cron schedule, tick, effect delivery, SQL read | 2 | 15.66 | 56.400 | 113.502 | 160.807 | 431.613 | -| **Fleet, all six actions** | **1** | **94.01** | **29.647** | **94.869** | **256.105** | **358.473** | -| **Fleet, all six actions** | **2** | **93.94** | **30.429** | **83.836** | **114.228** | **454.294** | - -Each run sent 5,000 peer requests through the balancer. Entry selection was -1,667 / 1,667 / 1,666 requests across nodes 0 / 1 / 2. The receiving nodes -forwarded 831 / 1,542 / 1,016 requests in run 1 and 822 / 1,533 / 1,019 in -run 2. Every node served local requests and forwarded remote requests. The -different per-node forwarding counts reflect the unequal seven-Cell placement -and the primitive mix, not uneven balancer selection. - -Run 1 had a substantially higher latency tail than run 2 under uncontrolled -desktop load. A nearby direct-owner run was also slower than these balanced -runs, so these observations cannot isolate load-balancer or forwarding cost. -This test does not measure separate machines, an external balancer, provider -network latency, product ingress, mTLS, dynamic owner discovery or placement, -follower durability, node loss, or millions of instantiated Cells. The Workflow -activity handler remains the in-process reference echo handler. A deployed -three-machine test and a separate high-cardinality control-plane test are -needed to qualify those claims. diff --git a/crates/crab-cell-app/performance/2026-09-21-local.md b/crates/crab-cell-app/performance/2026-09-21-local.md deleted file mode 100644 index 1f9413ef1..000000000 --- a/crates/crab-cell-app/performance/2026-09-21-local.md +++ /dev/null @@ -1,44 +0,0 @@ -# Cell primitive local end-to-end performance — 2026-09-21 - -Measured code: `a87ac6c2b20`. -Host: Apple M2 Max, macOS Darwin 25.5.0 arm64, Rust 1.97.0. -Profile: Cargo `--release`; serial clients; 100 verified actions per primitive -per run. Each action uses the compiled typed application, local Cell router, -SQLite actor, LTX publication, and a read or durable outcome check. Storage is -an in-memory object store, including the Blob part artifact path. Cron uses a -signed in-process peer round trip. The timer excludes seven-Cell bootstrap and -shutdown; it includes all calls for each action and the intentional 6 ms Cron -due-time wait. - -Run from the repository root: - -```bash -CRAB_CELL_PERF_ITERATIONS=100 \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-89be5c6d \ - cargo test -p crab-cell-app --test reference_application \ - performance::reference_primitive_end_to_end_performance \ - --release --locked -- --ignored --nocapture -``` - -| Action | Run | actions/s | p50 ms | p95 ms | p99 ms | max ms | -| --- | ---: | ---: | ---: | ---: | ---: | ---: | -| SQL order insert and read | 1 | 767.52 | 0.844 | 1.171 | 14.907 | 15.965 | -| SQL order insert and read | 2 | 795.10 | 0.810 | 1.034 | 15.274 | 17.582 | -| KV cart put and get | 1 | 763.54 | 0.901 | 1.125 | 14.280 | 14.656 | -| KV cart put and get | 2 | 771.10 | 0.826 | 1.091 | 14.598 | 18.465 | -| Blob attachment upload and read, 32 KiB | 1 | 199.38 | 3.049 | 19.488 | 25.457 | 31.857 | -| Blob attachment upload and read, 32 KiB | 2 | 195.64 | 3.116 | 20.469 | 30.228 | 30.629 | -| Queue notification send, claim, acknowledge | 1 | 189.21 | 3.203 | 18.492 | 24.517 | 41.870 | -| Queue notification send, claim, acknowledge | 2 | 206.59 | 3.098 | 18.373 | 19.350 | 26.598 | -| Workflow start, native activity, terminal read | 1 | 166.49 | 3.939 | 21.612 | 24.113 | 28.136 | -| Workflow start, native activity, terminal read | 2 | 170.51 | 4.003 | 19.760 | 21.646 | 23.099 | -| Cron schedule, tick, effect delivery, SQL read | 1 | 55.78 | 14.321 | 31.186 | 39.472 | 78.567 | -| Cron schedule, tick, effect delivery, SQL read | 2 | 46.22 | 14.180 | 70.304 | 104.733 | 115.503 | - -These are development measurements on an uncontrolled desktop. The two runs -show substantial variation, so the figures should not be used as capacity -limits or performance regressions. The Workflow activity's reference handler -returns its input and does not call an external service. No HTTP transport, -cloud provider, multi-node routing, concurrency, owner loss, or fault injection -was measured. Provider and release qualification remain separate gates in -`crates/crab-cell-runtime/docs/delivery.md`. diff --git a/crates/crab-cell-app/performance/2026-09-21-three-node-local.md b/crates/crab-cell-app/performance/2026-09-21-three-node-local.md deleted file mode 100644 index 5e84df593..000000000 --- a/crates/crab-cell-app/performance/2026-09-21-three-node-local.md +++ /dev/null @@ -1,65 +0,0 @@ -# Cell three-runtime fleet performance — 2026-09-21 - -Measured code: `a87ac6c2b20`. - -This is a local three-runtime topology on one Apple M2 Max host (macOS Darwin -25.5.0 arm64, Rust 1.97.0). Each runtime has a distinct node session, SQLite -worker pool, and local database files. All three share one in-memory object -store for Cell authority, publication, and Blob artifacts. A load generator -uses the compiled typed application and signed private peer requests. Static -Cell-ID routing sends every Cell request over loopback TCP to the owning runtime; -the receiver verifies the signature and dispatches through its local Cell actor. -Every peer request opens a new TCP connection. Cron effect delivery crosses from -the Cron runtime to the SQL runtime through the same peer path. - -| Runtime | Owned Cells | -| --- | --- | -| 0 | SQL, Queue, Workflow | -| 1 | KV, Queue dead letter | -| 2 | Blob, Cron | - -The benchmark runs six concurrent lanes, one per primitive. Each lane performs -100 serial, fully verified user actions from `PERFORMANCE.md`. A complete action -includes its write, follow-up read or terminal/delivery check, and all peer -round trips. The timer excludes seven-Cell bootstrap and shutdown. Cron actions -include the intentional 6 ms due-time wait. Fleet throughput is 600 completed -actions divided by the wall time from starting all lanes until the last lane -finishes. The fleet latency percentiles pool all 600 action samples. The lanes -finish at different times, so this is a finite equal-mix workload at at most -six concurrent actions, not a fleet saturation limit. - -Run from the repository root with an external target directory unique to this -checkout: - -```bash -CRAB_CELL_PERF_ITERATIONS=100 \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-89be5c6d \ - cargo test -p crab-cell-app --test reference_application \ - performance::reference_three_node_fleet_end_to_end_performance \ - --release --locked -- --ignored --nocapture -``` - -| Verified action | Run | actions/s | p50 ms | p95 ms | p99 ms | max ms | -| --- | ---: | ---: | ---: | ---: | ---: | ---: | -| SQL order insert and read | 1 | 172.15 | 4.314 | 13.881 | 22.203 | 30.736 | -| SQL order insert and read | 2 | 158.07 | 4.886 | 14.215 | 27.095 | 31.987 | -| KV cart put and get | 1 | 231.45 | 3.024 | 7.788 | 28.672 | 35.787 | -| KV cart put and get | 2 | 207.66 | 3.325 | 10.209 | 22.365 | 33.185 | -| Blob attachment upload and read, 32 KiB | 1 | 77.31 | 9.293 | 33.081 | 46.598 | 46.978 | -| Blob attachment upload and read, 32 KiB | 2 | 71.87 | 9.984 | 32.401 | 44.580 | 46.666 | -| Queue notification send, claim, acknowledge | 1 | 82.20 | 8.671 | 30.963 | 35.811 | 38.561 | -| Queue notification send, claim, acknowledge | 2 | 75.22 | 9.510 | 28.723 | 39.637 | 43.682 | -| Workflow start, native activity, terminal read | 1 | 68.71 | 10.871 | 35.750 | 48.053 | 49.629 | -| Workflow start, native activity, terminal read | 2 | 63.53 | 11.962 | 32.491 | 50.760 | 52.046 | -| Cron schedule, tick, effect delivery, SQL read | 1 | 37.41 | 23.270 | 45.311 | 59.227 | 67.312 | -| Cron schedule, tick, effect delivery, SQL read | 2 | 35.02 | 25.811 | 49.500 | 64.387 | 66.511 | -| **Fleet, all six actions** | **1** | **224.41** | **8.763** | **35.718** | **46.978** | **67.312** | -| **Fleet, all six actions** | **2** | **210.10** | **9.394** | **36.952** | **50.760** | **66.511** | - -These measurements verify three independent runtime owners and the signed peer -protocol over the local TCP stack. They do not measure three processes or -machines, mTLS, product ingress and dynamic routing, real network delay, cloud -object storage, follower durability, ownership movement, or failure recovery. -The Workflow activity handler returns its input in-process. Production fleet -capacity and latency require a deployed three-node qualification with a real -provider and representative client concurrency. diff --git a/crates/crab-cell-app/performance/2026-09-21-three-process-local.md b/crates/crab-cell-app/performance/2026-09-21-three-process-local.md deleted file mode 100644 index 4eb636920..000000000 --- a/crates/crab-cell-app/performance/2026-09-21-three-process-local.md +++ /dev/null @@ -1,67 +0,0 @@ -# Cell three-process fleet performance — 2026-09-21 - -Measured code: `a87ac6c2b20`. - -The load generator and three Cell owners ran as four OS processes on one Apple -M2 Max host (macOS Darwin 25.5.0 arm64, Rust 1.97.0). Each owner had a distinct -node session, SQLite worker pool, and database directory. The seven Cells were -placed across owners as follows: - -| Owner process | Cells | -| --- | --- | -| 0 | SQL, Queue, Workflow | -| 1 | KV, Queue dead letter | -| 2 | Blob, Cron | - -The processes shared a test-only filesystem object store. Its conditional -control-record update uses a cross-process file lock; object writes use the -local filesystem. The load generator routed each typed Cell request by Cell ID -over loopback TCP. Each owner verified the signed peer request, resolved only -its own Cells, and dispatched through the Cell actor. Every request opened a -new TCP connection. Cron effect delivery crossed from process 2 to SQL on -process 0. Blob part bytes used the shared object store. The load generator -started the six verified actions in `PERFORMANCE.md` as concurrent lanes, with -100 serial actions per lane. The timer excluded bootstrapping and graceful -shutdown; each action included its follow-up read or durable outcome check. -Cron included the intentional 6 ms due-time wait. - -Fleet throughput is 600 verified actions divided by the wall time until all -lanes finish. Fleet latency percentiles pool all 600 end-to-end action samples. -The lanes finish at different times. This is a finite equal-mix workload with -at most six actions in flight, not a saturation or production capacity limit. - -Run from the repository root with a target directory unique to this checkout: - -```bash -CRAB_CELL_PERF_ITERATIONS=100 \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-89be5c6d \ - cargo test -p crab-cell-app --test reference_application \ - process_performance::reference_three_process_fleet_end_to_end_performance \ - --release --locked -- --ignored --nocapture -``` - -| Verified action | Run | actions/s | p50 ms | p95 ms | p99 ms | max ms | -| --- | ---: | ---: | ---: | ---: | ---: | ---: | -| SQL order insert and read | 1 | 66.27 | 11.825 | 30.225 | 53.176 | 78.881 | -| SQL order insert and read | 2 | 69.41 | 12.003 | 24.490 | 60.409 | 61.126 | -| KV cart put and get | 1 | 88.17 | 9.450 | 18.191 | 55.173 | 59.934 | -| KV cart put and get | 2 | 89.56 | 9.250 | 17.136 | 52.264 | 53.874 | -| Blob attachment upload and read, 32 KiB | 1 | 29.50 | 29.277 | 65.929 | 78.255 | 79.361 | -| Blob attachment upload and read, 32 KiB | 2 | 31.37 | 27.980 | 61.950 | 76.333 | 78.234 | -| Queue notification send, claim, acknowledge | 1 | 31.52 | 26.549 | 72.309 | 89.642 | 100.055 | -| Queue notification send, claim, acknowledge | 2 | 33.16 | 26.395 | 62.790 | 75.801 | 80.204 | -| Workflow start, native activity, terminal read | 1 | 28.88 | 30.384 | 64.314 | 96.724 | 97.024 | -| Workflow start, native activity, terminal read | 2 | 30.64 | 30.138 | 61.203 | 73.995 | 78.475 | -| Cron schedule, tick, effect delivery, SQL read | 1 | 19.65 | 48.196 | 102.352 | 115.194 | 130.834 | -| Cron schedule, tick, effect delivery, SQL read | 2 | 20.69 | 45.695 | 92.352 | 100.961 | 101.188 | -| **Fleet, all six actions** | **1** | **117.86** | **24.769** | **73.771** | **102.293** | **130.834** | -| **Fleet, all six actions** | **2** | **124.16** | **24.549** | **65.847** | **89.620** | **101.188** | - -The two runs vary under uncontrolled desktop load. The process test proves -separate owner processes, signed peer dispatch, shared authority, and -fleet-level result verification. It does not measure separate machines, -inter-host network latency, mTLS, product ingress or dynamic placement, a cloud -object provider, follower durability, node loss, or a representative sustained -traffic mix. The in-process Workflow activity handler returns its input. The -three-runtime report uses in-memory storage, so its numbers do not isolate the -cost of process separation. diff --git a/crates/crab-cell-app/performance/2026-09-25-public-host-action.md b/crates/crab-cell-app/performance/2026-09-25-public-host-action.md deleted file mode 100644 index b25ca2215..000000000 --- a/crates/crab-cell-app/performance/2026-09-25-public-host-action.md +++ /dev/null @@ -1,49 +0,0 @@ -# Reference application public-host action measurements — 2026-09-25 - -Two release-profile runs used the same three-`CellNode` reference fixture with -100 serial actions in each lane. The source was commit `3b8d3b3614c` plus the -uncommitted application-boundary changes in this worktree. The host was an -Apple M2 Max, Darwin arm64, with Rust 1.97.0. SQLite used temporary local -files; the object store was in memory. Signed peer requests crossed loopback -TCP. The forwarded lane went through a second gateway peer before reaching the -owner. The two lanes ran in sequence against the same SQL Cell. - -```bash -CRAB_CELL_PERF_ITERATIONS=100 \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-8bc8 \ - cargo test -p crab-cell-app --test reference_application \ - public_host::reference_public_host_action_performance \ - --release --locked -- --ignored --nocapture -``` - -Each action prepared the generated typed command, waited for its published -receipt, and verified the resulting count with a generated typed query at that -receipt. `execute → durable ack` starts after preparation and ends when the -command returns its receipt. It includes command execution and, on the -forwarded lane, network transport. `Object durability proof wait` is the -runtime telemetry from post-commit submission until the object proof; it -includes publication queue time and is separate from transport and query time. -Throughput below is verified full actions per second for the whole lane. - -| Measure | Run | Throughput | p50 ms | p95 ms | p99 ms | -| --- | ---: | ---: | ---: | ---: | ---: | -| Local verified action | 1 | 811.85 | 0.736 | 1.030 | 16.286 | -| Local verified action | 2 | 827.39 | 0.720 | 0.885 | 16.492 | -| Local execute → durable ack | 1 | — | 0.700 | 0.997 | 16.251 | -| Local execute → durable ack | 2 | — | 0.688 | 0.855 | 16.453 | -| Local object durability proof wait | 1 | — | 0.195 | 0.317 | 15.779 | -| Local object durability proof wait | 2 | — | 0.194 | 0.267 | 15.935 | -| Forwarded verified action | 1 | 347.99 | 2.369 | 3.049 | 15.065 | -| Forwarded verified action | 2 | 350.52 | 2.348 | 3.137 | 15.665 | -| Forwarded execute → durable ack | 1 | — | 1.165 | 1.491 | 13.815 | -| Forwarded execute → durable ack | 2 | — | 1.147 | 1.410 | 14.445 | -| Forwarded object durability proof wait | 1 | — | 0.234 | 0.338 | 12.827 | -| Forwarded object durability proof wait | 2 | — | 0.227 | 0.331 | 13.489 | - -Owner fencing, authority takeover, exact-root restore, and the first verified -typed query took **9.682 ms** in run 1 and **8.695 ms** in run 2. Those are one -recovery sample per run, so they cannot define a recovery percentile or SLO. - -The runs did not include cloud storage, a real node-log durability provider, -dynamic placement, process failure, concurrent load, or large Cell counts. -They do not set a supported capacity, throughput, or recovery limit. diff --git a/crates/crab-cell-app/performance/2026-09-25-public-host-rustfs.md b/crates/crab-cell-app/performance/2026-09-25-public-host-rustfs.md deleted file mode 100644 index b147af993..000000000 --- a/crates/crab-cell-app/performance/2026-09-25-public-host-rustfs.md +++ /dev/null @@ -1,49 +0,0 @@ -# Reference application public-host actions with RustFS — 2026-09-25 - -The reference application's three-`CellNode` public-host fixture used the -local Docker Compose RustFS bucket instead of `InMemory`. The measured -implementation is commit `8eb19762d65`. The host was an Apple M2 Max with -32 GiB RAM; Colima had 8 CPUs, 16 GiB RAM, and an 80 GiB virtual disk. The -20-node issue fleet was healthy but idle during these tests. The three -application hosts ran in the test process, signed peer requests used loopback -TCP, SQLite used local temporary files, and RustFS used the host's loopback -port `19010`. Each run wrote to a distinct prefix in the Compose bucket. - -Each release-profile run measured 100 serial local actions followed by 100 -serial forwarded actions against one SQL Cell. An action prepared the generated -typed command, waited for its published receipt, then verified the resulting -count with a generated typed query at that receipt. `Execute → durable ack` -starts after preparation and ends when the command returns its receipt. -`Object proof wait` starts at post-commit submission and includes publication -queue time and the object-store write. Throughput is verified full actions per -second within that serial lane. - -| Measure | Run 1 actions/s | Run 1 p50 / p95 / p99 ms | Run 2 actions/s | Run 2 p50 / p95 / p99 ms | -| --- | ---: | ---: | ---: | ---: | -| Local verified action | 117.76 | 6.564 / 17.110 / 38.721 | 98.74 | 7.574 / 33.333 / 43.877 | -| Local execute → durable ack | — | 6.508 / 16.809 / 38.631 | — | 7.476 / 33.129 / 43.765 | -| Local object proof wait | — | 5.792 / 14.448 / 37.720 | — | 6.566 / 31.858 / 42.705 | -| Forwarded verified action | 93.49 | 8.482 / 21.596 / 40.166 | 94.74 | 8.469 / 16.982 / 42.362 | -| Forwarded execute → durable ack | — | 7.170 / 20.046 / 38.657 | — | 7.077 / 15.181 / 40.907 | -| Forwarded object proof wait | — | 5.989 / 18.592 / 37.441 | — | 5.895 / 13.721 / 39.680 | - -The one owner-loss takeover and first verified typed read took 28.734 ms and -28.951 ms. The second run left 1,096 objects under -`reference-performance/public-host-89829-1790367055486167000/` in the -RustFS bucket, verified through the S3 API. The application recovered its -acknowledged count from that bucket. - -For comparison, a same-branch, 100-iteration `InMemory` run measured 827.84 -local verified actions/s at 1.022 ms p95 and 307.60 forwarded verified -actions/s at 4.200 ms p95. Its local and forwarded object proof waits were -0.389 ms and 0.612 ms p95. The RustFS measurements include an S3-compatible -HTTP service and its local object-store write path; the difference is not a -cloud-network estimate or an isolated RustFS service-time measurement. - -The raw outputs are retained outside the checkout in -`$HOME/.codex/cell-issue-fleet/main-20260925/rustfs-action-main-run1.log`, -`rustfs-action-main-run2.log`, and `memory-action-main-run.log`. The runnable command is -in the [Compose example](../../crab-http-server/deploy/cell-issue-fleet/README.md). -These serial, single-host observations do not establish saturation throughput, -20-node action latency, peak resource use, a recovery percentile, or a -production SLO. diff --git a/crates/crab-cell-app/performance/2026-09-27-host-readers-compose.md b/crates/crab-cell-app/performance/2026-09-27-host-readers-compose.md deleted file mode 100644 index 3d7fbfa27..000000000 --- a/crates/crab-cell-app/performance/2026-09-27-host-readers-compose.md +++ /dev/null @@ -1,82 +0,0 @@ -# Host-owned reader refresh and drain on GA RustFS - -Three constrained public hosts passed the reference application smoke on -2026-09-27. Initial reader admission uses an explicit fixture owner hint; -subsequent refresh, placement eviction, and shutdown use the production -`crab-cell-host` manager shared with the issue service. - -## Source and environment - -- Source: `6a04c638d54fb4bf26eea27e28c6a5be46635240`. -- Release binary SHA-256: - `6b01f660ba308a3a5d07cc8dad9df14018a63384e613eb6d1c3ce5017e17220f`. -- Rust image: `rust:1.97-bookworm`, digest - `sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97`. -- RustFS: `1.0.0-glibc`, digest - `sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. -- Dedicated Colima VM: Linux ARM64, four CPUs and 8,307,101,696 memory bytes. - Another idle RustFS fixture remained in the same VM. -- Each node and driver: one CPU, 1 GiB, zero swap, 256-process limit, - dropped capabilities, read-only root/source/binary, distinct disk-backed - scratch volume. The driver hosts the TCP balancer. This is one physical host. -- Build used a fresh target and exact cached image digests. Only downloaded - Cargo dependencies were shared with the earlier qualification project. - -Reproduce with the [Compose procedure](../PERFORMANCE.md#three-constrained-compose-nodes), -30 iterations per primitive lane, and a fresh state directory/project. - -## Verified behavior - -The release build passed in 3 minutes 48 seconds. Three nodes and driver each -executed exactly one selected test and exited zero. All node sessions renewed -and withdrew at generation 6. - -- Missing readers returned unavailability; wrong-origin activation was fenced. -- Each selected reader served six generated queries; writer served zero. -- Both readers discovered a new exact root without a refresh hint. A duplicate - owner command retained its receipt and produced one effect. -- Setting desired readers to zero automatically removed both admitted views. - Fixture markers only observed receipt progress and eviction. -- Retained managers rejected activation and resolution after host shutdown. - -Node whole-cgroup peaks were 23.852 / 13.629 / 21.246 MiB; driver peak was -20.613 MiB. Kernel counters and Docker inspection confirmed the configured -limits, zero OOM/OOM kills, and zero CPU throttling. These are short process -lifetime peaks, including filesystem cache, not sustained RSS estimates. - -The preceding six primitive lanes completed 180 actions in 10.702 seconds -(16.82 actions/s), with combined p50/p95/p99 129.558 / 704.751 / 1033.552 ms. -This was slower than the earlier smoke. The runs are not a controlled causal -comparison, and no performance improvement or supported limit is claimed. -Replica lifecycle checks occur after the action timer. Balancer traffic was -524/524/524; the seven writer Cells were assigned 3/2/2. - -## Other verification and evidence - -The stalled-store shutdown regression first failed its one-second drain -deadline, then passed after terminal cancellation was moved ahead of the -activation join. Application correctness: 15 passed; host suite: 35 passed. -Direct and balanced native GA RustFS process tests passed. The product public -HTTP/private-mTLS/GA-RustFS test executed one exact case and passed in 21.23 s. -Strict host/app all-target Clippy, server library Clippy, minimal host features, -frontend build, format, layout, policy, and documentation checks passed. -The existing main catalog-path and BeyondDB inventory guard failures remain. - -Raw evidence is retained in local qualification directory -`host-readers-6a04c63-20260927/evidence`, including source/binary identity, -resolved Compose, logs, container inspections, kernel counters and independent -`verification.json`. This is not a protected provider or release receipt. - -| File | SHA-256 | -| --- | --- | -| `driver.log` | `e262199d8e95cadffe85ee58560014dbd06338531c5b747a03ad33e9e4983f07` | -| `node-0.log` | `2d560ac6668fc489fd9d84bf551744ece679bd6752d2a5f72c8476ce341bc1de` | -| `node-1.log` | `18195bd020a104bbfa5ad555356debd2fba14ac21306a72e54645775bc07d4c2` | -| `node-2.log` | `99e2b5ce9e2d905146825e04dd0b4554d72405d464ad862ebb523a27ebf4054d` | -| `containers.json` | `e43cdb84ad1a587982153732523f3f464d2173647c8a4e32325ffa43c2ab4938` | - -Open work: general application-host recruitment/replacement after node loss, -sustained read freshness/throughput under writes, 5/10/20-node application -capacity, owner loss during arrivals, independent-host/provider qualification, -and protected release gates. Durability-log followers remain separate from -these SQL read snapshots. diff --git a/crates/crab-cell-app/performance/2026-09-27-mixed-readers.md b/crates/crab-cell-app/performance/2026-09-27-mixed-readers.md deleted file mode 100644 index e7ed0f385..000000000 --- a/crates/crab-cell-app/performance/2026-09-27-mixed-readers.md +++ /dev/null @@ -1,241 +0,0 @@ -# Generated replica reads during sustained writes - -The initial run completed sixty-second mixed windows at 3, 5, 10 -and 20 nodes against GA RustFS. All **691,427 successful replica queries** -matched their snapshot receipts. The run acknowledged **1,056 mutations**, -but missed **144 of 1,200 scheduled write arrivals**. Correctness and lifecycle -checks passed; none of the four windows fully served the offered write load. - -The [current-main integration rerun](#current-main-integration-rerun) below -verified another 703,981 reads. Its twenty-node window served all scheduled -writes, while the smaller stages missed 28 arrivals in total. Neither run -establishes a supported mixed-workload capacity. - -## Exact source and deployment - -- Source: `cb6445cd3e6b7602bb1ce3b0e92e4df48f9b8e9d`. -- Release test binary SHA-256: - `d8bcd662c5fd6556d34cb91751b1461b97613f62f2440902859e0da4c3a563a3`. -- Same pinned Rust 1.97-bookworm and RustFS 1.0.0-glibc images as the - [preceding publication run](2026-09-27-reader-publication.md). -- Colima ARM64: four CPUs, 8,307,101,696 memory bytes. Another idle RustFS - fixture shared the VM. Application nodes and driver each had one CPU cap, - 1 GiB memory, zero swap, 256 processes, dropped capabilities, read-only - source/binary/root and a distinct scratch volume. CPU caps are not reserved - physical cores. The object store used the existing Compose service profile. -- Fresh immutable source archive and Linux build target. Only downloaded - Cargo dependencies were shared. Release build completed in 3m46s; the - driver test completed in 305.36s. The controller exited zero. - -An initial build outside Colima's configured mount failed before compilation. -Its container and log remain separate. The measured run used a fresh directory -inside the existing mount; no VM restart or mount-policy change was required. - -## Workload and correctness - -The default thirty iterations of all six primitive lanes ran first. The -existing seven writer Cells remained on their original three owners. Mixed -traffic used one SQL Cell, with one generated writer scheduling five unique -mutations per second and eight concurrent generated replica-query clients. -Four clients required the latest acknowledged receipt; four accepted an older -snapshot. All requests used the public application handles and signed peer -transport. Writer traffic crossed the TCP balancer evenly; replica queries -went directly to selected nodes. This does not measure product HTTP latency. - -The driver joined each returned count to the complete command history at its -actual receipt, including reads overlapping a not-yet-returned command. The -controller independently verified the raw TSV files. It rejected false values -at covering receipts, results below the requested minimum, duplicate effects, -and missing arrival records. All selected readers served traffic; the writer -served zero replica queries. The final owner read and every selected reader -covered the last acknowledged mutation before proceeding. - -Late write arrivals were recorded and skipped without a catch-up burst. -`ReplicaBehind` is an explicit outcome, with no owner fallback. Other query -errors fail the workload. Successful-query percentiles below exclude behind -responses; their counts remain visible. - -## Measured windows - -| Nodes / readers | Successful reads | Reads/s | Read p50 / p99 ms | Writes / offered | Write p50 / p99 ms | Behind responses | -| --- | --- | --- | --- | --- | --- | --- | -| 3 / 2 | 206,661 | 3,443.82 | 2.064 / 5.333 | 273 / 300 | 25.010 / 868.645 | 3,766 | -| 5 / 3 | 199,797 | 3,329.34 | 2.165 / 5.698 | 286 / 300 | 26.275 / 563.034 | 3,046 | -| 10 / 9 | 156,354 | 2,603.54 | 2.391 / 11.191 | 231 / 300 | 58.677 / 1,047.752 | 1,073 | -| 20 / 19 | 128,615 | 2,143.39 | 2.723 / 24.030 | 266 / 300 | 36.797 / 694.445 | 486 | - -Maximum successful-read latency reached 1,190.191 ms at ten nodes; low p99 -does not remove these outliers. The strict-minimum lane's twenty-node p99 -was 30.497 ms; the lane permitting older snapshots measured 17.162 ms. - -Older snapshots lagged the latest acknowledgement known at query dispatch -by at most 1 / 1 / 2 / 2 mutations. Joining write acknowledgements and query -completion times on the same driver clock gives first-observed covering-read -p99 of 39.790 / 26.657 / 114.673 / 139.095 ms. These are sampled visibility -bounds on any reader, not every-reader refresh times or a freshness SLA. One -ten-node write had no covering query before the window ended; the separate -final convergence gate covered it. No covering read preceded its write's -acknowledgement in this run. - -The owner process's object-proof wait, aggregated across the initial primitive -workload and all four stages, had p50 12.469 ms and p99 479.661 ms. This does -not identify the cause of each write outlier or isolate provider service time. -No saturated-resource or linear-scaling claim follows from a shared four-core -VM. The database also grew between stages. - -## Lifecycle and resources - -After the five-node window, the controller killed selected reader-only node 3 -and verified exit 137 without OOM. Automatic membership expiry and recruitment -replaced it in 14.771s, including a 1.035s fault-command round trip. Twelve -generated queries recovered the exact final value/receipt. This fault occurred -after the mixed window, so it does not prove continuous traffic availability. - -The fleet then grew to ten and twenty live nodes using fresh identities. -Writer/session, epoch and incarnation stayed unchanged. Target zero evicted -all nineteen final readers. Twenty surviving node tests and the driver exited -zero, withdrew sessions and proved admission remained closed after drain. - -Surviving node whole-cgroup memory peaks ranged from 23,117,824 to 48,472,064 -bytes; driver peak was 50,810,880 bytes. The killed reader's last peak sample -was 30,466,048 bytes. No OOM event or CPU throttling was recorded in these -node/driver counters. These are whole-run measurements, not per-reader slopes -or complete VM/object-store resource measurements. All new containers are -stopped; their volumes and evidence remain available. - -## Reproduction and retained evidence - -Use the [scaling procedure](../PERFORMANCE.md#constrained-reader-scaling-and-loss) -with the default thirty primitive iterations. The existing Compose CI runs -the same mixed windows and the evidence-verifier tests. - -Raw state is retained under the per-worktree Workspace target in -`activation-profile-f9d54fc-20260927/mixed-readers-cb6445c-20260927/evidence/scaling`. -`verification.json` contains the hash of every command/query TSV. A separate -analysis joins the same-driver timing samples for the visibility figures above. - -| Artifact | SHA-256 | -| --- | --- | -| `verification.json` | `aa24c53e5d6c8eef69021f571b70fe732a3264ebcd54c1f1f08d99b9a06cfbfb` | -| `driver.log` | `09fc0f490620258da36e1047cfa57595f2505aed67d71386ba7d02793293f75c` | -| `events.json` | `b6963da7aeda8bfe309a6c9101da5f9bbd620b68095b025417235d6fa78f8bb7` | -| `containers.json` | `95f52a18e7f9e477b2213aeb114337ed8a582fa4215cb03cb04cf128483d8213` | -| `mixed-load-analysis.json` | `0ad0a9b3939d157a6b90134b52fa03b04f1a124b3bd2f33f007c0dfdbab8c622` | - -Native test-binary compilation, strict scoped Clippy, six verifier tests, -format, Cell/LTX layout and workflow YAML parsing passed. No production Rust, -API, dependency, lockfile, runtime option or existing qualification threshold -changed. Remaining: controlling write tail latency under reader load, -saturation curves, many-Cell resource/storage costs, faults and rollout during -traffic, and independent-host/provider production qualification. - -## Current-main integration rerun - -The reader stack was replayed onto main -`de215cd0c49a1bcd52b06a62127be1bcc7a80533`, retaining its newer catalog and -BeyondDB changes. Earlier stacked reader PRs had been merged into their former -parent branches after those parents landed, leaving this stack absent from -main. This rerun verifies the resulting integration, rather than assuming -the earlier binary's evidence covers it. - -- Source: `66a893523c13e071f4da1e6f280afd30cd980f31`. -- Release binary SHA-256: - `cb803624c8c1a0decd4a0932d26b9c80268f8a2e732b34d190090857b750e302`. -- Fresh source archive, target, storage volume and Compose project; same pinned - images, workload and Colima resource profile. Native verification also ran - on the macOS host during part of this run; this is not isolated capacity. -- Linux release build: 4m45s. Driver: 301.57s. Controller exited zero. - -| Nodes / readers | Successful reads | Reads/s | Read p50 / p99 ms | Writes / offered | Write p50 / p99 ms | Behind responses | -| --- | --- | --- | --- | --- | --- | --- | -| 3 / 2 | 210,886 | 3,503.80 | 2.077 / 4.794 | 291 / 300 | 18.651 / 358.888 | 3,885 | -| 5 / 3 | 186,006 | 3,099.45 | 2.304 / 6.424 | 289 / 300 | 20.718 / 390.874 | 2,850 | -| 10 / 9 | 171,543 | 2,858.65 | 2.397 / 9.835 | 292 / 300 | 20.857 / 420.350 | 1,057 | -| 20 / 19 | 135,546 | 2,258.83 | 2.647 / 21.439 | 300 / 300 | 19.645 / 222.091 | 460 | - -All **703,981 successful reads** matched their exact receipts. The independent -verifier retained **1,172 acknowledgements**, **28 missed arrivals** and -**8,252 behind responses**. Only the twenty-node stage set -`fully_served_writes=true`. Variation between these two short runs prevents a -supported-rate or performance-improvement claim. Percentiles use raw -microsecond records and exclude behind responses. - -Maximum known acknowledgement lag was 1 / 1 / 1 / 2 mutations. First observed -covering-read p99 after acknowledgement was 42.946 / 42.043 / 40.030 / 61.926 ms. -One three-node write was not observed during the query window; the separate -final convergence check covered it. These remain sampled observations of any -reader, not a freshness guarantee. - -Reader 3 was killed after the five-node window. Replacement completed in -14.643s, including a 0.651s fault-command round trip; twelve exact queries -then passed. Owner, -epoch and incarnation stayed unchanged. All selected readers served traffic; -the owner served no replica query. Target-zero eviction, terminal close and -all twenty surviving node tests passed. Surviving node memory peaks ranged -from 24,768,512 to 49,852,416 bytes; driver peak was 54,063,104 bytes. No recorded -node/driver cgroup CPU throttling or OOM occurred. All new containers stopped; -volumes and raw evidence remain retained. - -Evidence is under the same mounted parent as the initial run, in -`mixed-readers-66a8935-20260927/evidence/scaling`. Every raw TSV hash is checked -again by the independent analysis. - -| Artifact | SHA-256 | -| --- | --- | -| `verification.json` | `8d442d7d01c69b6e4b8feb59448aad92543a043c93e2fd75268da00f7ac8afef` | -| `driver.log` | `478ecfe6e765d135914d6a87025204842f8f216646be0e6db0302966320fa859` | -| `events.json` | `330a92cbe22e5a9abf3a95b8c9d8bbe6ccbbeeb864ddc9861d1dca58c0159db6` | -| `containers.json` | `026553c8679a052b66ddfc75009ff5a7734779b02b51aee0476c40b237acb6e5` | -| `mixed-load-analysis.json` | `024a5a2cfe60b5403644d4081cc464a5c87360953743c9d0ffdcd66e77d55f13` | - -Native integration proof: six snapshot lifecycle tests, six public-host tests -and seven product recruitment tests passed. Strict runtime/app/host all-target -Clippy, HTTP library/binary Clippy, the native server build, six verifier tests, -format, layout and workflow parsing passed. Manual cases stayed ignored in -native suites; the separate container run above supplies the real-store proof. -Existing BeyondDB dependency-inventory and frozen product-descriptor baseline -failures remain unchanged from current main. Their baselines were not edited. - - -## Linux CI integration evidence - -[Compose run 36333012095](https://github.com/crabbuild/crab/actions/runs/36333012095) -passed on Ubuntu 24.04.5, x86_64, one shared 4-CPU / 16,766,414,848-byte host. -The tested PR merge commit was `d8ceaf89632f7cc1a38294606585633bb620a8b7`; -its tree equals PR490 head `05ed72527cbb7a331f1ccc149de7c5593205fd4f`. -Release binary SHA-256 was -`ba51f66c24400e30d136bab37a1cf7fb756fd172ea2e7d21dc5e5651a6b2713d`. -The workload and pinned RustFS GA image were unchanged. - -| Nodes | Exact reads | Writes / offered | Read p99 ms | Write p99 ms | Behind responses | -| --- | --- | --- | --- | --- | --- | -| 3 | 86,364 | 300 / 300 | 12.005 | 269.974 | 4,341 | -| 5 | 86,753 | 300 / 300 | 12.758 | 249.793 | 3,118 | -| 10 | 66,199 | 300 / 300 | 32.195 | 210.780 | 1,513 | -| 20 | 35,344 | 286 / 300 | 94.594 | 578.271 | 726 | - -Independent replay of all raw TSVs confirmed 274,660 exact reads, 1,186 -acknowledged writes, 14 missed arrivals and 9,698 typed behind responses. -The twenty-node window did not fully serve the offered writes. Reader -replacement between windows took 14.869s including the 0.408s fault command; -twelve exact queries followed. All twenty surviving nodes and the driver -exited zero with verified one-CPU / 1-GiB / zero-swap limits. Their memory -peaks ranged from 23,945,216 to 48,209,920 bytes; one throttled CPU period -and no OOM events were recorded. These are shared-host observations, not -supported distributed capacity or availability under continuous faults. - -The same job passed the three-node application smoke (180 actions, ingress -524/524/524, twelve generated replica reads) in 13.30s and the additive code -rollout in 0.28s (four exact receipts, two duplicate replays, fresh-host -recovery). The rollout still uses three public hosts inside one driver -container and pauses at cutover; it does not prove continuous rolling updates. - -Artifact `cell-reference-compose-36333012095-1` retains raw samples, logs, -source identity and kernel/container evidence. Rechecked SHA-256 values: - -| Artifact | SHA-256 | -| --- | --- | -| `scaling/verification.json` | `069df70b59831bd987700234d5d03b1b09a06f0794c6e12b25d1235915c0c41f` | -| `scaling/driver.log` | `51e6ad6f5ccffab2b629fb386443d7330172dc04e959eba034d15679ee38f688` | -| `scaling/containers.json` | `306293e54ceb39c95b2602f76846047e641e5387d7e947447dea7fdcc5f72361` | -| `rollout.log` | `f7ad473004479c7b0727758d2097a781801060ee1919459bf6f35191841cf5bd` | diff --git a/crates/crab-cell-app/performance/2026-09-27-reader-expiry.md b/crates/crab-cell-app/performance/2026-09-27-reader-expiry.md deleted file mode 100644 index 91b46ef08..000000000 --- a/crates/crab-cell-app/performance/2026-09-27-reader-expiry.md +++ /dev/null @@ -1,110 +0,0 @@ -# Reader replacement after signed-session expiry - -The constrained 3/5/10/20-node reference application passed against GA RustFS -on 2026-09-27 after pending activation hints began rechecking their selected -boot at its observed lease expiry. Reader replacement took **10.983 s** in this -run, compared with **34.830 s** in the preceding -[same-profile local run](2026-09-27-reader-scaling.md). This is one comparison, -not a recovery SLO or sustained-capacity qualification. - -## Cause and change - -Recruitment joins its selected activation attempts before selecting again. -Previously an unresponsive peer could hold that pass for the thirty-second -transport budget after its signed session expired. The earlier spare's own -readiness marker appeared 34.191 seconds after the kill acknowledgement; -driver polling did not account for the delay. Independent baseline Compose CI -also observed 34.878 seconds. - -`ReplicaPeerClient::activate` now races the existing request against a fresh -exact-session lookup at the observed expiry. Expired, withdrawn or invalid -sessions release the hint; a renewed session preserves the original request -and transport deadline. A ready response can interrupt a stalled directory -lookup. Existing receiver admission, ownership and receipt checks still apply. -The change adds no task, timeout setting, wire shape or dependency. - -The reduced expiry test failed before the change and passes afterward. Its -renewal companion proves a slow valid request is neither canceled nor resent. -Both the reference TCP transport and product HTTP transport honor the caller's -request budget. Product reqwest 0.12.28 has no separate connect timeout by -default, and the product client does not set one. No packet-level cause for -the original connection delay is claimed. - -## Source and environment - -- Source: `72cb8b83adf6300e71b42486cc1c801af3982b77`. -- Release binary SHA-256: - `a6db7c6fdbd31ec338fcda22aad29fdcb98128d2e6d6826ce51c400f029b84e5`. -- Fresh immutable source archive and separate Linux target; only downloaded - Cargo dependencies were shared. Release build: 3 minutes 11 seconds. -- Same pinned Rust 1.97-bookworm and RustFS 1.0.0-glibc image digests and - dedicated four-CPU / 8,307,101,696-byte ARM64 Colima VM as the baseline. -- Each node and driver: one CPU cap, 1 GiB memory, zero swap, 256 processes, - private disk volume, read-only root/source/binary and dropped capabilities. - The VM supplies four physical CPUs to the whole fleet; per-node caps do not - reserve twenty physical cores. -- Unchanged `qualification/scale.py` profile, five initial iterations per - primitive lane. Only activation lifetime behavior changed in runtime code. - -## Results - -| Live nodes | Readers | Exact reads | p50 ms | p95 ms | p99 ms | Serial window | Second-write readiness | -| --- | --- | --- | --- | --- | --- | --- | --- | -| 3 | 2 | 60 | 1.030 | 1.473 | 2.914 | 0.068 s | 5.050 s | -| 5 | 3 | 90 | 0.969 | 1.315 | 2.572 | 0.094 s | 5.016 s | -| 10 | 9 | 270 | 1.046 | 1.404 | 1.872 | 0.290 s | 4.774 s | -| 20 | 19 | 570 | 1.040 | 1.512 | 1.994 | 0.635 s | 4.865 s | - -Every selected reader served exactly thirty measured queries; the owner served -zero replica queries. All values and receipts matched. Duplicate commands -retained one effect and receipt. Separate owner queries traversed all live -ingresses with counts differing by at most one; replica queries used selected -peer routing. Seven writer Cells stayed on the original three owners. - -At five nodes the controller killed selected reader-only node 3, verified -exit 137 without OOM, and retained its pre-kill counters. Normal signed -membership expiry and recruitment admitted the spare without a fixture hint. -Twelve queries then verified the acknowledged value and receipt on the three -current readers. Driver fault-request-to-ready time was 10.983 seconds, -including 0.723 seconds for command acknowledgement. Owner/session/epoch and -incarnation stayed unchanged. The remaining delay includes lease expiry and -reconciliation; it has not been characterized across fault timing phases. - -Target zero evicted all nineteen final reader views. Twenty surviving nodes -and the driver each passed exactly one selected test and exited zero. Nodes -withdrew renewed sessions and proved their retained managers could not reopen -after drain. Driver duration: 76.34 seconds. - -Independent inspection verified all twenty-two node/driver records, kernel -limits, private volumes and identical binary hashes. No OOM or CPU throttling -was recorded. Surviving-node whole-cgroup peaks were 7.609–17.195 MiB; driver -peak was 23.656 MiB. The killed node's last sample was 20.023 MiB. These short -measurements do not establish many-Cell resource slopes. - -Seven server recruitment tests, seventeen protocol tests, four public-host -correctness tests and four snapshot-lifecycle tests passed. Three manual host -performance cases and one manual single-process RustFS case were ignored in -those focused suites; the dedicated Compose run supplies this change's live -RustFS proof. Minimal runtime/host builds, strict Clippy for runtime/host/app -and HTTP library/tests, formatting, layout, policy entry points and runtime -documentation validation passed. - -## Evidence and remaining gates - -Raw evidence is retained under -`reader-expiry-72cb8b8-20260927/evidence/scaling`; the project was stopped after -capture, with containers and volumes retained. The existing Compose CI runs -this same profile. This local result is not a protected release receipt. - -| Artifact | SHA-256 | -| --- | --- | -| `driver.log` | `e191654c556331df0bc850dde766cecd81655524ece1812b66158c2ca6b5ab5e` | -| `events.json` | `ec217d78cb3e20a8b4b5d8b61809f87213da00c404ad40435b39ae8cc2d814c8` | -| `containers.json` | `e6bb17c0f5549c955dd757e569aeeb281e2fbb6a846c10176a3299b181c2b642` | -| `verification.json` | `092d25ebd960881ecc8b8c53bf973280e4c13f29fad7ed6cef7d8edcf532f14d` | - -The five-second refresh interval remains visible. Query samples exclude -freshness waiting and are serial, brief and limited to one replicated SQL -Cell. Sustained concurrent reads/writes, many-Cell admission, traffic during -owner loss and rollout, writer redistribution, independent hosts/providers -and protected release qualification remain open. diff --git a/crates/crab-cell-app/performance/2026-09-27-reader-loss-during-load.md b/crates/crab-cell-app/performance/2026-09-27-reader-loss-during-load.md deleted file mode 100644 index eea6092f5..000000000 --- a/crates/crab-cell-app/performance/2026-09-27-reader-loss-during-load.md +++ /dev/null @@ -1,116 +0,0 @@ -# Reader loss during scheduled traffic - -## Profile and acceptance - -The canonical `qualification/scale.py` runner grows a fleet through 3, 5, 10, -and 20 independent public `CellNode` processes using the digest-pinned RustFS -1.0.0 image in `qualification/compose.yaml`. Each node has one CPU, 1 GiB of -memory, no swap, and a private disk-backed SQLite/WAL/cache volume. The local -Colima VM has four CPUs and 8,307,101,696 bytes of memory; RustFS shares it. -Container ceilings do not provide independent physical CPU or failure domains. - -Each size retains the existing 60-second window: five scheduled writes per -second to one SQL Cell and eight closed-loop replica query lanes. Even lanes -request at least the latest acknowledged receipt; odd lanes allow an older -snapshot. A separate 60-second five-node window kills a selected reader after -ten seconds. Writer ingress remains on the original three gateways, isolating -reader loss from gateway failure. The host must recruit a replacement without -manual activation, and that replacement must serve workload queries before -second 50. Every successful result must match acknowledged receipt history. - -The independent verifier requires successful writes and successful queries -from **every lane** wholly before, during, and after replacement. A request -spanning recovery does not count as progress during the outage. Typed -`ReplicaBehind` responses remain visible but do not count as successful reads. -Missed scheduled writes remain failed offered work, even when correctness and -reader replacement pass. - -## Failed runs retained - -Source `9a46fa03f914b6e89227c0e430ad494352ae9513` failed after reader loss with -`ReplicaUnavailable`: one stalled selected peer consumed the entire five-second -query deadline while other readers remained usable. Both a stalled peer and a -stalled local resolver reproduced the failure in public runtime tests. Shared -routing now allocates remaining query time across remaining candidates, bounds -both paths, and sends the reduced budget to remote readers. - -Source `ba6f1f984baf0155416f6b589b55c3a2763bcd58` completed the Rust driver in -338.17 seconds but **failed the independent availability gate**. Replacement -first served 14.726 seconds after the fault request. All four minimum-receipt -lanes had zero successful queries during replacement. The fault window recorded -261 acknowledged writes, 39 missed writes, 156,178 correct reads, 2,095 typed -behind responses, and a maximum observed lag of 20 acknowledged writes. - -The owner recruiter awaited the complete activation fanout before processing -another publication. A pending activation therefore blocked fresh hints to -healthy readers. A public-host regression reproduced this independently of -load. The supervisor now keeps bounded activation work in flight while handling -later publications. Pending Cell/session pairs coalesce; healthy completed -pairs can receive another hint. Existing lease renewal, receiver admission, -receipt verification, and shutdown cancellation remain in the shared path. - -These failed runs and their raw timelines remain retained in the external -qualification state directory. A passing Rust test alone does not supersede -their verifier failures. - -## Current-source verification - -Source `00e6214d9cd3ecdb12ebbb70e41cef0be0d559ff` includes both fixes on main -`322ba3ed4f06b73186a0e2543045a7f2fe2502b0`. The fresh Linux release build passed -in 1m57s; the Rust driver finished in 371.17 seconds. The driver, canonical -verifier, and a separate raw-data audit passed. Binary SHA-256: -`3c7f8b637de27b264bd410cd4d76cd8c11a018ea852ae2921cf0b9db5e388e74`. - -All 534,953 successful queries matched acknowledged receipt history. Every -lane made progress wholly before, during, and after reader replacement. During -replacement, minimum-receipt lanes 0/2/4/6 completed 380/95/88/90 successful -queries, and 30 writes completed. The replacement first served 14.743 seconds -after the fault request, leaving more than 35 seconds of subsequent traffic. -The killed reader exited 137 without OOM. All surviving nodes and the driver -exited zero; node sessions withdrew and reader facilities drained. - -| Window | Acknowledged / planned writes | Missed writes | Correct reads | Typed behind | Write p99 ms | Read p99 ms | -| --- | --- | --- | --- | --- | --- | --- | -| 3 nodes | 241 / 300 | 59 | 162,388 | 3,499 | 1,092.458 | 10.450 | -| 5 nodes | 240 / 300 | 60 | 138,277 | 2,444 | 718.331 | 14.545 | -| 10 nodes | 264 / 300 | 36 | 78,784 | 466 | 674.999 | 48.391 | -| 20 nodes | 218 / 300 | 82 | 64,485 | 457 | 1,403.977 | 75.525 | -| 5 nodes, reader loss | 221 / 300 | 79 | 91,019 | 1,520 | 1,350.812 | 32.068 | - -Latency percentiles above cover successful calls, excluding missed write -slots and typed behind responses. The run missed 316 of 1,500 scheduled writes; -**it does not qualify the offered write rate**. A host sample during the fault -window recorded load averages 14.76/10.54/10.04, about 10.7 GiB of used swap, -and unrelated native Rust compilers consuming substantial CPU. This task ran -no concurrent native compilation during measurement. The shared-host run -proves the fault behavior; it cannot isolate a performance regression or -improvement from these code changes. - -The 21 surviving node/driver roles enforced their kernel CPU, memory and -no-swap limits. Peak memory ranged from 23,343,104 to 49,549,312 bytes; total -CPU throttling was 17 periods, with no OOM. The controller stopped its RustFS -and retained its containers, volumes and evidence after verification. - -Raw evidence is retained under -`reader-loss-00e6214-20260927/evidence/scaling/` in the external qualification -state directory. `verification.json` binds every workload TSV by SHA-256; -`independent-audit.json` rechecks counts, receipt/value joins, per-lane phase -progress and kernel limits. Artifact hashes: - -| Artifact | SHA-256 | -| --- | --- | -| `verification.json` | `9cdcac4cfbe589939b399ec3c44ccda23f350125f5224bcf664f40e134dc5ae5` | -| `driver.log` | `b643b3f8c0f18db310a24977ded69101daa24ccd2b8a9938c88400e741f6c251` | -| `containers.json` | `af83b2fd6b19297e0de350f055e47b2a4096f2821ff66e29a20b4e499ceabbb2` | -| `events.json` | `2ba5c51c1c258991daaaa35821053e1b88fabdcd1eeab07182f02aeb6ccef132` | -| `independent-audit.json` | `6019c6d7638bed455c128c3b11e0b5e615233a33af4a61ba86322b259730567e` | - -Focused native proof before the rebase: 19 reference application tests, -35 host lifecycle tests, seven HTTP recruitment tests, and ten independent -evidence-verifier tests passed. Strict runtime/host all-target and reference -test Clippy, format, layout, policy, and documentation checks passed. The -intervening main commits did not change these routing or recruitment paths. - -This profile cannot establish many-Cell capacity, owner or gateway failure, -continuous container/code/schema rollout, multi-host durability, or supported -latency and throughput limits. Those remain separate plan gates. diff --git a/crates/crab-cell-app/performance/2026-09-27-reader-publication.md b/crates/crab-cell-app/performance/2026-09-27-reader-publication.md deleted file mode 100644 index 8b9e27c65..000000000 --- a/crates/crab-cell-app/performance/2026-09-27-reader-publication.md +++ /dev/null @@ -1,116 +0,0 @@ -# Reader refresh after object publication - -The constrained 3/5/10/20-node reference application passed against GA RustFS -on 2026-09-27. Second-write readiness fell from about five seconds in the -[preceding run](2026-09-27-reader-expiry.md) to **110–152 ms** in this run. -These short observations establish functional progress, not a supported -freshness or throughput limit. - -## Changes and failure diagnosis - -Successful object publication, activation and migration now send bounded -advisory notifications from `CellRuntime`. The existing public-host recruiter -coalesces queued Cells and uses its normal authenticated activation path. -Writer acknowledgement never waits for readers. Fleet-only proof sends no -notification until object publication completes. Periodic reconciliation -still observes membership/policy changes and repairs missed notifications. - -The first attempt, source `9e6b8d292f91621b4110f0049ac5b0b749c9aed3`, failed -before any reader measurements. Retained SQLite showed a Workflow start at -logical time `1790514623093`, followed by an empty activity claim whose request -was created at `1790514623029`. The new activity remained unclaimed. The -sequential caller and persisted records identify a backward clock sample; -they do not identify why the VM clock moved. - -The executor already kept persisted logical time nondecreasing, but typed -handlers received the earlier raw sample. A deterministic public-host test -reproduced `Idle` after a clock rollback. The fix constructs command, owner -query, effect-delivery and replica-query contexts using at least the committed -snapshot time, reusing their existing SQLite metadata lookup. Request identity -timestamps and authority/session checks remain separate. The regression now -completes through local and three-node application handles. Owner and replica -query tests verify the same committed-time floor. Current main has the raw -handler timestamp behavior; the defect predates publication notifications. - -## Exact source and profile - -- Source: `f42369386a2e48ae88f337538185d8c972817590`. -- Release binary SHA-256: - `7a93297a6b5424d5f9e81504445fca07389fa529fdf39228b3ebce1cc2bf7486`. -- Fresh immutable archive and separate Linux build target; only downloaded - Cargo dependencies were shared. Release build: 3 minutes 11 seconds. -- Same pinned Rust 1.97-bookworm and RustFS 1.0.0-glibc images as the preceding - run, on the same dedicated four-CPU / 8,307,101,696-byte ARM64 Colima VM. -- Each node and driver: one CPU cap, 1 GiB memory, zero swap, 256 processes, - private disk volume, read-only root/source/binary, dropped capabilities. - These caps do not reserve twenty physical cores on the four-CPU VM. -- Unchanged `qualification/scale.py`, five initial iterations per primitive - lane. The mixed workload exercises SQL, KV, Blob, Queue, Workflow and Cron. - -## Results - -| Live nodes | Readers | Exact reads | p50 ms | p95 ms | p99 ms | Serial window | Second-write readiness | -| --- | --- | --- | --- | --- | --- | --- | --- | -| 3 | 2 | 60 | 0.890 | 1.216 | 1.694 | 0.057 s | 0.110 s | -| 5 | 3 | 90 | 0.895 | 1.259 | 1.792 | 0.086 s | 0.111 s | -| 10 | 9 | 270 | 1.014 | 1.308 | 1.816 | 0.284 s | 0.126 s | -| 20 | 19 | 570 | 1.040 | 1.574 | 1.947 | 0.619 s | 0.152 s | - -Readiness starts after the write acknowledgement and includes its duplicate -check, an owner read, and polling all selected readers at 100 ms intervals. -It is not isolated refresh execution time. Query percentiles exclude this wait. -Every selected reader served exactly thirty queries; the owner served zero -replica queries. All values and receipts matched. Separate owner calls used -all live ingresses with counts differing by at most one. Seven writer Cells -remained on the original three owners. - -At five nodes, selected reader-only node 3 was killed with verified exit 137 -and no OOM. Signed-session expiry and normal recruitment admitted the spare -without a fixture-issued activation. Replacement took **14.634 s**, including -0.669 s for fault-command acknowledgement. Twelve subsequent queries matched -the acknowledged value and receipt. Writer owner/session, epoch and incarnation -stayed unchanged. This sample is slower than the preceding 10.983 s sample; -publication notifications do not remove membership-expiry or polling delay. - -Target zero evicted the final nineteen reader views. Twenty surviving nodes -and the driver each passed exactly one selected test and exited zero; nodes -withdrew renewed sessions and proved readers could not reopen after drain. -Driver duration: 46.09 seconds. - -Independent inspection verified all twenty-two node/driver records, identical -binary hashes, distinct scratch volumes and actual kernel limits. No OOM or -CPU throttling was recorded. Surviving node whole-cgroup peaks ranged from -6.758 to 16.594 MiB; driver peak was 24.211 MiB. The killed node's last sample -was 20.055 MiB. These short samples do not establish resource slopes. - -## Verification and retained evidence - -Focused proof: six public-host tests, five snapshot-lifecycle tests, seventeen -client protocol tests, ten durability-proof tests and seven server recruitment -tests passed. Three manual host tests and one manual single-process RustFS -test were ignored; the dedicated Compose run supplies live RustFS proof here. -Strict Clippy, minimal runtime/host builds, format, layout, policy entry points -and documented Rust syntax checks passed. Compose CI now also triggers on the -runtime and its relevant host/app/storage/build inputs. - -Raw evidence is retained under -`reader-publication-f423693-20260927/evidence/scaling`. The stopped project and -its volumes remain available. The failed earlier attempt and its copied -SQLite/WAL evidence remain separately under -`reader-publication-9e6b8d2-20260927/evidence/diagnosis`. - -| Artifact | SHA-256 | -| --- | --- | -| `driver.log` | `abb4fbd821a851834cc0baf60b75b88f39325cf9cb66ee20702ce651b5acf982` | -| `events.json` | `468c86dbf9baabda20d5978ad0dad1e191fc6bd70cab5f96001611597fa034cf` | -| `containers.json` | `787031ef396f81acd11de55c332cb3180e317330ca59c45f7e75a758bed711f0` | -| `verification.json` | `604d3d8998c3a340459d4659f128ba35e139eea95b6fb89509a61c395d8a55af` | - -Remaining: sustained concurrent reads/writes, object-store request amplification, -many-Cell admission/resource slopes, owner loss and rollout during arrivals, -writer redistribution, independent hosts/providers and protected release gates. -This local smoke establishes none of their supported limits. Separately, -[product fleet CI run 36311817281](https://github.com/crabbuild/crab/actions/runs/36311817281) -failed its 600-second 20-node placement gate: its final balanced observation -finished at 633.299 seconds. No 20-node offered-rate stage or subsequent owner -loss during arrivals ran. That source/image is distinct from this reader proof. diff --git a/crates/crab-cell-app/performance/2026-09-27-reader-recruitment.md b/crates/crab-cell-app/performance/2026-09-27-reader-recruitment.md deleted file mode 100644 index c892dccc3..000000000 --- a/crates/crab-cell-app/performance/2026-09-27-reader-recruitment.md +++ /dev/null @@ -1,127 +0,0 @@ -# Public-host reader recruitment and replacement on GA RustFS - -The reference application now recruits readers through the same public host -loop as the issue service. Signed activation/status operations use the shared -runtime peer dispatcher. Fixture markers observe readiness; they do not -activate readers. Two complementary runs passed on 2026-09-27: constrained -three-node Compose and native three-to-five-process reader replacement. - -## Source and environment - -- Source: `11947641f56a09addf38565ad481524ad6aa1120`. -- Compose release binary SHA-256: - `1360f2bcbcd5383535c7ec5bc8c9583d56d6a7817096a270976771da9c278f97`. -- Rust image: `rust:1.97-bookworm`, digest - `sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97`. -- RustFS: `1.0.0-glibc`, digest - `sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. -- Dedicated Colima VM: Linux ARM64, four CPUs and 8,307,101,696 memory bytes. - Another idle RustFS fixture remained in the same VM. -- Each Compose node and driver: one CPU, 1 GiB, zero swap, 256-process limit, - dropped capabilities, read-only root/source/binary, distinct disk-backed - scratch volume. The driver contains the TCP balancer. All share one host. -- Fresh source archive and build target; only Cargo downloads were shared. - The separate native process run used Rust 1.98 and GA RustFS, without - per-process CPU or memory limits. - -Reproduce using the [Compose procedure](../PERFORMANCE.md#three-constrained-compose-nodes) -with 30 iterations per primitive lane and the -[reader-loss command](../PERFORMANCE.md#reader-recruitment-and-process-loss) -with five iterations per lane. Use isolated object prefixes and fresh state. - -## Constrained Compose result - -The release build passed in 3 minutes 54 seconds. All three nodes and the -driver executed exactly one selected test and exited zero. Node sessions -renewed and withdrew at generation 6. Driver duration was 21.24 seconds. - -- Missing readers returned unavailability before policy enablement. -- Both non-owner readers opened through automatic signed owner recruitment. - Each served six generated queries; the writer served zero replica queries. -- Both readers automatically refreshed to the newly acknowledged root. - Repeating the owner mutation retained one receipt and one effect. -- Target zero automatically evicted both admitted views. After host shutdown, - retained managers rejected activation and resolution. -- No manual `readers.activate` control marker was present. - -Kernel counters and Docker inspection confirmed the declared limits, zero -OOM/OOM kills, and zero CPU throttling. Whole-cgroup peak memory was -29.477 / 12.613 / 16.484 MiB for nodes 0/1/2 and 21.465 MiB for the driver. -These short process-lifetime peaks include filesystem cache; they are not -sustained RSS or capacity estimates. - -The preceding six primitive lanes completed 180 verified actions in 6.245 -seconds, or 28.82 actions/s. Combined p50/p95/p99 was -73.467 / 261.785 / 357.842 ms; maximum was 662.172 ms. This is a short -closed-loop workload. It is not a controlled performance comparison, and -the reader lifecycle checks occur outside its timer. No supported throughput -or replica-query latency is established. Balancer ingress was 524/524/524; -the seven writer Cells were assigned 3/2/2. - -## Actual reader-process loss - -`owner_replaces_killed_reader_through_public_hosts` passed on the same source -in 46.17 seconds. It exercised all six primitive lanes on three public hosts, -published a generated SQL command, and observed two serving readers. A -principal with read/status privileges was rejected when attempting activation. - -The test then started two more processes and verified five live directory -members. It killed a selected reader process (node 2), confirmed its abnormal -exit, and waited for two selected, ready readers excluding the dead session. -At least one was a new reader. Twelve generated queries returned the exact -acknowledged value and receipt across the two readers. Writer session, epoch, -and incarnation stayed unchanged; all four survivors drained and exited zero. - -Observed time from confirmed process exit to two ready readers was 14.777 -seconds. This includes membership expiry and recruitment. It is one recovery -sample, not an SLO. This native run does not establish five-node container -capacity, uninterrupted reads during the fault, or owner-loss behavior. - -## Other verification - -- Application: nine unit tests, three contracts, fifteen reference correctness - cases, one ordinary doctest and three compile-fail doctests passed. -- Host suite: 35 passed. The stalled-provider regression covers cancellation - of both activation and recruitment before the host's one-second drain bound. -- Product recruitment/router regressions: five passed, including refreshed - advertisements, wrong-incarnation replies, interrupted cursor progress, - corrupt policy isolation and stalled-reader fanout. -- Peer protocol: 17 passed. Existing signed-request time validation exposed a - transit-budget bug during extraction; the client now uses the canonical - authorization-budget helper, and the test includes transit time. -- Public HTTP/private-mTLS/GA-RustFS product E2E: one exact case passed in - 19.68 seconds. -- Strict all-target runtime/app/host Clippy, strict server library Clippy, - minimal host features, format, layout, policy and documentation checks passed. - The two existing main architecture-guard failures remain: retired-LTX path - detection and the BeyondDB development-dependency inventory mismatch. - -The final Compose and reader-loss runs use the exact source above. The broad -native application/host/router and product tests preceded a return-type -spelling-only change to the existing `BoxFuture` alias; final Clippy and peer -protocol checks include that change. No dependency versions changed. - -## Retained evidence and limits - -Raw Compose evidence is retained under local qualification directory -`reader-recruitment-1194764-20260927/evidence`: source/binary identities, -resolved Compose, build and test logs, container inspections, kernel counters, -and independent `verification.json`. Native reader-loss evidence is -`reader-replacement-1194764.log`. These are local qualification artifacts, -not protected provider or release receipts. - -| File | SHA-256 | -| --- | --- | -| `driver.log` | `66f02fa8cab6065ec2ab9bea9434d0eae1e5b8cd8c3f4ede6810b7c9d8e0d615` | -| `node-0.log` | `a857fda0b4102fabe7e1fe3ff4e0dc173d1b8f9b4d882c45f681aa8d8b98e150` | -| `node-1.log` | `2853f6ea36ff7b0db9ef527396ae74d22588b926ec167a86cfbbe8c16eb8f9d8` | -| `node-2.log` | `2b0be0e572e0a41e368aa11481238296a164fabcb7ac152667aeefd3049444a9` | -| `containers.json` | `7d154c231b91687ca837e903274c917443db99a8eaa4d2e4eb86eafa0e9ffd71` | - -Open gates: constrained 5/10/20-node application capacity and fault runs, -sustained read freshness/throughput under writes, owner loss during arrivals, -large-Cell resource slopes, independent-host/provider qualification and -protected release evidence. SQL read replicas remain explicit published -snapshots; a newer minimum receipt may return `ReplicaBehind`. Durability-log -followers do not execute SQL. General product reads remain owner-ordered -unless the caller selects an implemented replica route. diff --git a/crates/crab-cell-app/performance/2026-09-27-reader-scaling.md b/crates/crab-cell-app/performance/2026-09-27-reader-scaling.md deleted file mode 100644 index 115be121e..000000000 --- a/crates/crab-cell-app/performance/2026-09-27-reader-scaling.md +++ /dev/null @@ -1,125 +0,0 @@ -# Constrained public-host reader scaling on GA RustFS - -One reference-application fleet grew through 3, 5, 10 and 20 live Compose -nodes on 2026-09-27. Generated replica reads, actual container loss and -replacement, target-zero eviction and survivor shutdown all passed. This is -a short integration smoke; sustained capacity remains unqualified. - -## Source and deployment - -- Source: `a220f05496220a501ad4f644ffb4ab0fd4e8e18a`. -- Release binary SHA-256: - `9b25aadc0d1e58540438d9346c5e338026a55ce85170437e9db7520c7a40a39a`. -- Rust: `1.97-bookworm`, digest - `sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97`. -- RustFS: `1.0.0-glibc`, digest - `sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. -- Dedicated Colima ARM64 VM: four CPUs, 8,307,101,696 memory bytes. Another - idle RustFS fixture shared the VM. Each node and driver had a one-CPU cap, - 1 GiB memory, zero swap, 256-process limit, read-only root/source/binary, - dropped capabilities and its own disk-backed scratch volume. -- Fresh source archive and build target; downloaded Cargo dependencies were - shared. Release build completed in 3 minutes 24 seconds. No host ports were - published. A CPU cap is not a dedicated physical core: this VM had four - cores for the complete fleet, driver and object store. - -Run the [scaling procedure](../PERFORMANCE.md#constrained-reader-scaling-and-loss) -with five iterations per initial primitive lane. The reader profile always -checks thirty queries per selected reader at each stage. - -## Observed reads and freshness - -The initial three owners exercised all six primitive lanes. The seven writer -Cells stayed assigned 3/2/2 to these owners. Added nodes became gateways and -readers for one SQL Cell; this run does not redistribute writer Cells. - -At every stage, two new generated commands were published and duplicate -delivery preserved each receipt and one effect. Readers opened and refreshed -through the host's signed recruitment/supervision path. After all selected -readers reached the acknowledged receipt, every generated read returned its -exact expected value and receipt. The owner served zero replica queries. - -| Live nodes | Selected readers | Measured reads | p50 ms | p95 ms | p99 ms | Serial measurement duration | -| --- | --- | --- | --- | --- | --- | --- | -| 3 | 2 | 60 | 0.887 | 1.825 | 2.233 | 0.062 s | -| 5 | 3 | 90 | 1.344 | 2.227 | 4.781 | 0.127 s | -| 10 | 9 | 270 | 0.932 | 1.241 | 2.230 | 0.269 s | -| 20 | 19 | 570 | 0.975 | 1.238 | 1.802 | 0.584 s | - -These 990 serial samples measure small queries against ready snapshots on -the local Docker network. They exclude recruitment/freshness waiting and are -too short to establish tail-latency or throughput support. Each selected -reader served exactly thirty samples. Separately, owner queries traversed all -live ingress nodes through the TCP balancer; entry counts differed by at most -one at each stage. Replica queries used direct selected-node peer routing. - -The second acknowledged write at each stage took 4.951 / 5.017 / 4.912 / -4.992 seconds to become observable on every selected reader. Timing includes -the duplicate check, owner read-back and status polling. Low query latency -does not imply equally fresh snapshots: the host's five-second reconciliation -interval remains visible. No freshness SLO is established. - -## Reader-container loss - -At five nodes, three readers were selected and one non-owner node was a -spare. The controller killed selected reader-only node 3, verified exit 137 -without an OOM event, and retained its kernel counters before the kill. No -fixture activation was issued. Normal membership expiry and owner recruitment -selected a replacement, and twelve generated queries verified the exact -acknowledged value and receipt across all three current readers. - -Time from the driver's fault request to three ready readers was **34.830 s**; -the controller's kill acknowledgement took 0.669 s of that interval. This is -one sample. The native regression observed 14.817 s after process exit, so it -cannot stand in for container-network failure behavior. Follow-up should -determine whether waiting on a thirty-second peer attempt delays the next -placement pass. Continuous reads during the fault and owner loss were not -measured. - -The fleet temporarily had four survivors, then grew to ten and twenty. New -nodes used fresh fixture identities; the killed boot was not restarted. -Twenty-one node containers were created in total. Writer session, epoch and -incarnation remained unchanged throughout. - -## Shutdown, resources and proof - -Target zero evicted all nineteen final reader views. All twenty surviving -nodes and the driver ran exactly one selected test and exited zero. Every -surviving node withdrew its renewed session and proved its retained manager -rejected activation/resolution after drain. Driver duration was 101.64 s. - -Docker inspection and kernel counters independently confirmed resource limits, -private scratch volumes, identical binary hashes, no OOM events and no CPU -throttling. Node whole-cgroup peaks ranged from 7.438 to 18.211 MiB; driver -peak was 23.762 MiB. The killed node's 13.137 MiB is its last pre-kill sample. -These short lifetime measurements include filesystem cache and do not -establish many-Cell resource slopes or sustained RSS. - -App checks passed: nine unit tests, three contracts, fifteen reference -correctness tests, one ordinary and three compile-fail doctests. The shared -native reader-loss regression passed in 50.72 s. Strict reference-suite -Clippy, format, layout, shell syntax and 66 documented Rust snippets passed. -The final controller-only correction does not change the tested Rust source. - -The first controller attempt stopped before workload because Python's -`Path.with_suffix` requires a leading dot. Its source and failed logs remain -retained; it supplies no scaling proof. The corrected run above used a fresh -archive, target, project and object-store volume. - -Raw evidence is retained under local qualification directory -`reader-scale-a220f05-20260927/evidence/scaling`: resolved Compose, controller -events, logs, kernel counters, container inspections and `verification.json`. -All project services were stopped after capture; containers and volumes remain -available. This is not a protected release receipt. - -| Artifact | SHA-256 | -| --- | --- | -| `driver.log` | `aafdb1157df12fda31d27ade86ac33f1c795d8d8d74630c65567ac12081153cf` | -| `events.json` | `f609d6a2403e9b3ebe77a647a484ded6d9350ae672fa1a07c9fe14529e003117` | -| `containers.json` | `4eaadaf5b98827cef3b7a52d867b0a557db395881368ac543a73aab329506b92` | -| `verification.json` | `c9e6e5ef807a13ef4034c24361a4c2b95c1d73e841fe961edb489bc770ac928f` | - -Remaining gates: replacement latency investigation, sustained/concurrent reads -and writes, many-Cell admission and resource slopes, arrivals during faults, -owner-loss recovery, independent-host/provider qualification and protected -release evidence. diff --git a/crates/crab-cell-app/performance/2026-09-27-replica-compose.md b/crates/crab-cell-app/performance/2026-09-27-replica-compose.md deleted file mode 100644 index ec6096aa1..000000000 --- a/crates/crab-cell-app/performance/2026-09-27-replica-compose.md +++ /dev/null @@ -1,95 +0,0 @@ -# Generated replica reads across three constrained hosts - -Integration smoke passed on 2026-09-27. Generated application queries executed -on two non-owner `CellNode` processes using GA RustFS. This records correctness -and resource admission for the public application path; sustained capacity and -automatic reader replacement require separate qualification. - -## Reproduce the measured source - -Use the [Compose procedure](../PERFORMANCE.md#three-constrained-compose-nodes) -with 30 iterations per primitive lane and a fresh source archive/state directory. - -- Source: `a815f9ad46bf700b1f603da2ea7cf15d07fa2713`. -- Release binary SHA-256: - `8da34e33a8eb724650443884af5daff35cf0babbd50267325b9da7f14cddac5a`. -- Build/node image: `rust:1.97-bookworm`, digest - `sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97`. -- RustFS: `1.0.0-glibc`, digest - `sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. -- Dedicated Colima VM: Linux ARM64, four CPUs, 8,307,101,696 memory bytes. - Another idle RustFS fixture remained on the same VM. -- Each node and the driver had one CPU, 1 GiB memory, zero swap, 256-process - limit, dropped capabilities, read-only root/source/binary mounts, and a - distinct disk-backed scratch volume. The driver hosted the TCP balancer. -- Cached exact image digests were selected with `pull_policy: never`. - The build used a fresh compiled target; only downloaded Cargo dependencies - were shared with the earlier qualification project. - -## Replica behavior proved - -The driver constructs `ApplicationHandle` and the generated `ReferenceClient` -with explicit `ReadPolicy::Replica`. Signed peer queries reach each selected -node's admitted immutable view, including through its gateway receiver. - -| Check | Observed result | -| --- | --- | -| Readers selected but not opened | `ReplicaUnavailable`; no owner fallback | -| First snapshot | Six generated reads return identical data and receipt | -| New owner mutation delivered twice | Same committed receipt; one visible effect | -| Unqualified read before refresh | Returns the previous data and receipt | -| New minimum receipt before refresh | `ReplicaBehind` with exact observed/minimum sequences | -| Controlled refresh on both readers | Six reads return the new data and exact commit receipt | -| Successful direct peer replies | Writer 0; reader 1: 7; reader 2: 6 | - -Refresh is driven by fixture control markers. The successful-reader counters -count decoded query outputs from selected physical nodes; they are independent -of balancer ingress counts. These steps run outside the action timer. - -The first native attempt failed to admit readers: a 32-writer pool reserves -only 2 MiB by default, below one 12 MiB read view. The measured host explicitly -sets native admission to 32 MiB through `SqlWorkerPool::with_native_memory_limit`. -It keeps the writer ceiling at 32 and charges old and replacement snapshots -concurrently. Retained cuts have a separate 64 MiB limit. Neither admission -budget claims to cap actual RSS. - -## Process and resource evidence - -The release build completed in 4 minutes 32 seconds. Each of the three nodes -and driver ran exactly one selected test and exited zero. All nodes withdrew -their renewed signed sessions at generation 3. Node cgroups recorded zero CPU -throttling, OOM, and OOM kills. Peak whole-cgroup memory was 29.215 / 13.715 / -16.398 MiB; driver peak was 19.809 MiB. These short process-lifetime peaks -include filesystem cache and do not establish sustained memory headroom. - -The six preceding primitive lanes verified 180 business actions in 5.938 s -(30.31 actions/s), with combined p50/p95/p99 of 108.634 / 225.915 / 342.918 ms. -Balancer requests were 524/524/524. These action timings exclude the replica -proof and do not measure replica throughput. Seven writer Cells were assigned -3/2/2; action mix determines owner load. - -Native sibling evidence also passed: direct and balanced three-process GA -RustFS tests; four runtime replica cases; the ignored real-store exact-root -and policy-CAS case; native-budget/writer-limit regression; 14 application -correctness cases; strict Clippy and minimal-feature checks. - -Raw evidence is retained in local qualification directory -`reference-replicas-a815f9a-20260927/evidence`: source identity, logs, binary -hashes, resolved Compose, container inspection, kernel counters, and -`verification.json`. The Compose CI workflow reproduces this fixture; -this local run is not a protected release receipt. - -| File | SHA-256 | -| --- | --- | -| `driver.log` | `09d3299fe28c0f7c15261ea3b50a907ca14cc44a2e50fc28cf31fc158e8df3be` | -| `node-0.log` | `7726cb243dd8b78780be07a821d7722dfaa53389f310b7a43aa25089af1745f0` | -| `node-1.log` | `0abae88301f902c0f41ad5345a85d98ddff319326f9eee3f596b87a4cd8cd904` | -| `node-2.log` | `354e73b82c5e21d34b1bf0dee08fa9d42da4908de85c0582ee04f08222efc9bc` | -| `containers.json` | `e6ada690153a06c3758698ae521af58de65cf2d94fba2eb779558b555da03189` | - -Open gates include automatic reader reconciliation/replacement in a general -application host, sustained replica throughput and freshness under writes, -5/10/20-node application capacity, owner loss during arrivals, independent -hosts and networks, and protected provider/release qualification. Durability -log followers do not execute SQL; this proof uses object-backed command -acknowledgements and separately admitted read snapshots. diff --git a/crates/crab-cell-app/performance/2026-09-27-three-node-compose.md b/crates/crab-cell-app/performance/2026-09-27-three-node-compose.md deleted file mode 100644 index 4822f5ed8..000000000 --- a/crates/crab-cell-app/performance/2026-09-27-three-node-compose.md +++ /dev/null @@ -1,95 +0,0 @@ -# Three public reference hosts on GA RustFS - -Integration smoke completed on 2026-09-27. This is a short closed-loop application -measurement, not a supported capacity or latency profile. - -## Reproduction and identity - -Use the [Compose procedure](../PERFORMANCE.md#three-constrained-compose-nodes) -with `CRAB_CELL_PERF_ITERATIONS=30` and a fresh project/state directory. - -- Source: `3de78e2edfb98055dafd66fd5ce348940064e58e`. -- Release binary SHA-256: - `8e1af19984ae2a8ecb22ab32b860a59fe322c7c62c7c27ef227b3107a0bb5447`. -- Build/node image: `rust:1.97-bookworm`, digest - `sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97`. -- RustFS: `1.0.0-glibc`, digest - `sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. -- Local images were imported into the isolated VM and selected by those exact - IDs with `pull_policy: never`; the resolved Compose file and image inspection - are retained with the evidence. -- Colima VM: Linux ARM64, 4 CPUs, 8,307,101,696 memory bytes, kernel - `6.8.0-117-generic`. Host: Mac ARM64, 12 logical CPUs, 32 GiB memory. -- Three node containers and one driver: each 1 CPU, 1 GiB memory, zero swap, - 256-process limit. SQLite/WAL/cache used four distinct Docker volumes. -- Another idle RustFS fixture remained in the same VM. This is a shared-host - measurement, not independent CPU, storage, or network failure domains. - -## Observed results - -All three node tests and the driver executed exactly one selected test and -exited zero. The build completed in 4 minutes 53 seconds. Node sessions renewed -before readiness and withdrew after runtime drain at generation 3. Every -primitive lane verified its visible result. The generated stable-ID command -returned the original receipt on duplicate delivery and added exactly one -visible effect. - -| Business action | Actions | p50 ms | p95 ms | Maximum ms | -| --- | ---: | ---: | ---: | ---: | -| SQL order insert/read | 30 | 34.711 | 66.182 | 156.539 | -| KV cart put/get | 30 | 30.285 | 42.106 | 43.764 | -| Blob attachment upload/read, 32 KiB | 30 | 94.012 | 179.871 | 324.373 | -| Queue send/claim/acknowledge | 30 | 85.254 | 198.233 | 230.030 | -| Workflow start/activity/completion | 30 | 133.988 | 260.200 | 262.065 | -| Cron schedule/delivery/readback | 30 | 175.463 | 291.495 | 415.869 | - -The six concurrent lanes completed 180 actions in 5.504 seconds: 32.70 verified -business actions/s. Combined p50/p95/p99: 89.411/208.440/324.373 ms. Individual -lanes have only 30 samples; their reported p99 equals the maximum. Actions have -different numbers of durable commands and reads, so this rate is not a write -requests/s limit. - -The balancer admitted 523/523/522 requests. Node local/forwarded dispatch counts -were 848/245, 120/481, and 600/310. Owner-local counts include requests forwarded -by another node. The seven Cells were assigned 3/2/2; the business-action mix -is not equal owner load. - -| Node | Object-proof samples | Proof wait p50/p95 ms | Whole-cgroup peak MiB | Local disk reservation bytes after actions | -| --- | ---: | ---: | ---: | ---: | -| 0 | 271 | 18.609 / 56.980 | 27.000 | 9,013,696 | -| 1 | 30 | 17.735 / 28.097 | 12.859 | 675,696 | -| 2 | 210 | 17.282 / 47.307 | 18.340 | 5,677,376 | - -All node cgroups reported zero CPU throttling, OOM, and OOM kills. The driver -peaked at 19.598 MiB. These peaks include the complete short process lifetime -and filesystem cache; they are not steady-state RSS or sustained headroom. -Object-proof samples are not an independent phase sum for the action latencies. - -## Evidence and scope - -Raw logs, binary hashes, container/image inspection, resolved Compose, -cgroup counters, and a verification summary are retained in the local -`reference-compose-3de78e2-20260927/evidence` qualification directory. The -`Cell reference Compose smoke` workflow retains the same evidence classes -for subsequent CI runs. - -| File | SHA-256 | -| --- | --- | -| `driver.log` | `707ab7f0ebedd012f50f437e45a31cd4baec1b40534f1f6ddaa75c69a0e48f37` | -| `node-0.log` | `0002e5a92712957d4dc29835a6fed1ed65de446475b16a722782600500847356` | -| `node-1.log` | `ef9878cb40131f71fe9a4416d6f9cfbc60d8c5cdab0cc7200e9c1fc72c84361b` | -| `node-2.log` | `f95d95cf25fe2f1caeeac47c6b44e2ebc37a24dedee4e8f6410ea675f7a52653` | -| `containers.json` | `650617fdea9562adfe224fea158ea9dab0520fd72709c68ddef5efc2fe748c8d` | - -The native sibling checks also passed: both process modes ran concurrently -against GA RustFS; the existing RustFS public-host action/recovery test passed; -14 application correctness cases passed; strict all-target Clippy, formatting, -layout, policy-entry, and documentation checks passed. The architecture guard -still reports the existing `main` catalog-path false positive and BeyondDB dev -dependency inventory mismatch. This change does not edit either frozen inventory. - -Open gates: 5/10/20-node application capacity, hot/skewed workloads, sustained -arrival rates, mTLS/product ingress, automatic placement, owner loss during -unpublished traffic, and independent-host fault qualification. This fixture -uses object-backed command proofs and owner reads; read replicas and follower -log durability have separate qualification paths. diff --git a/crates/crab-cell-app/qualification/compose.yaml b/crates/crab-cell-app/qualification/compose.yaml deleted file mode 100644 index d9ab83473..000000000 --- a/crates/crab-cell-app/qualification/compose.yaml +++ /dev/null @@ -1,108 +0,0 @@ -x-image: &image rust:1.97-bookworm@sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97 -x-storage: &storage - AWS_ACCESS_KEY_ID: crab - AWS_SECRET_ACCESS_KEY: crab - AWS_EC2_METADATA_DISABLED: "true" - AWS_DEFAULT_REGION: us-east-1 -x-environment: &environment - <<: *storage - CRAB_CELL_TEST_BUCKET: crab-reference-app - CRAB_CELL_TEST_ENDPOINT: http://rustfs:9000 - CRAB_CELL_PERF_PROCESS_ROOT: reference-compose - CRAB_CELL_PERF_PROCESS_SYNC: /evidence/control - CRAB_CELL_PERF_PROCESS_GATEWAY: "1" - CRAB_CELL_PERF_PROCESS_BIND: 0.0.0.0:8080 - CRAB_CELL_PERF_ITERATIONS: ${CRAB_CELL_PERF_ITERATIONS:-30} - TMPDIR: /scratch -x-node: &node - image: *image - entrypoint: [/bin/sh, /source/crates/crab-cell-app/qualification/run.sh] - command: [node] - volumes: - - ${CRAB_REFERENCE_STATE:?set an external state directory}/source:/source:ro - - ${CRAB_REFERENCE_STATE:?set an external state directory}/target-linux:/target:ro - - ${CRAB_REFERENCE_STATE:?set an external state directory}/evidence:/evidence - # Anonymous Docker volumes give each node its own disk-backed SQLite/WAL/cache. - - /scratch - cpus: 1.0 - mem_limit: 1g - memswap_limit: 1g - pids_limit: 256 - read_only: true - cap_drop: [ALL] - security_opt: [no-new-privileges:true] - depends_on: - bucket-init: - condition: service_completed_successfully - restart: "no" - -services: - build: - image: *image - working_dir: /source - environment: - CARGO_TARGET_DIR: /target - CARGO_HOME: /cargo-home - CARGO_BUILD_JOBS: "2" - CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER: cc - volumes: - - ${CRAB_REFERENCE_STATE:?set an external state directory}/source:/source:ro - - ${CRAB_REFERENCE_STATE:?set an external state directory}/target-linux:/target - - cargo-home:/cargo-home - command: [cargo, test, --release, --locked, -p, crab-cell-app, --test, reference_application, --no-run] - cpus: 2.0 - mem_limit: 4g - memswap_limit: 4g - restart: "no" - - rustfs: - image: ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858 - environment: - RUSTFS_ACCESS_KEY: crab - RUSTFS_SECRET_KEY: crab - RUSTFS_CONSOLE_ENABLE: "false" - RUSTFS_OBS_LOG_DIRECTORY: /data/logs - volumes: [rustfs-data:/data] - healthcheck: - test: [CMD, curl, --fail, --silent, http://127.0.0.1:9000/health/ready] - interval: 3s - timeout: 3s - retries: 30 - start_period: 5s - restart: "no" - - bucket-init: - image: public.ecr.aws/aws-cli/aws-cli:2.27.41@sha256:1c2d7a51b1ff4f460bc3f1e1ee46a6cef47c7429ad9089ef99012a95e472a1a5 - environment: *storage - command: [--endpoint-url, http://rustfs:9000, s3api, create-bucket, --bucket, crab-reference-app] - depends_on: - rustfs: - condition: service_healthy - restart: "no" - - node-0: - <<: *node - environment: - <<: *environment - CRAB_CELL_PERF_PROCESS_NODE: "0" - CRAB_CELL_PERF_PROCESS_ADVERTISE: node-0:8080 - node-1: - <<: *node - environment: - <<: *environment - CRAB_CELL_PERF_PROCESS_NODE: "1" - CRAB_CELL_PERF_PROCESS_ADVERTISE: node-1:8080 - node-2: - <<: *node - environment: - <<: *environment - CRAB_CELL_PERF_PROCESS_NODE: "2" - CRAB_CELL_PERF_PROCESS_ADVERTISE: node-2:8080 - driver: - <<: *node - environment: *environment - command: [driver] - -volumes: - cargo-home: - rustfs-data: diff --git a/crates/crab-cell-app/qualification/entities.py b/crates/crab-cell-app/qualification/entities.py deleted file mode 100644 index 38d63dda4..000000000 --- a/crates/crab-cell-app/qualification/entities.py +++ /dev/null @@ -1,191 +0,0 @@ -"""Audit scheduled entity traffic against receipts, ownership and resource samples.""" - -from __future__ import annotations - -import bisect -import csv -import hashlib -from pathlib import Path - -STAGES = (3, 5, 10, 20) -SHAPES = ("uniform", "hot", "skewed") -POINTS = ((1, 4), (4, 16), (16, 64)) -CELLS_PER_NODE = 4 -SECONDS = 10 - - -def rows(path: Path) -> list[dict]: - with path.open(newline="") as source: - return list(csv.DictReader(source, delimiter="\t")) - - -def destination(shape: str, arrival: int, cells: int) -> tuple[int, str]: - if shape == "uniform": - return arrival % cells, "write" - if shape == "hot": - return (1 + (arrival // 5) % (cells - 1) if arrival % 5 == 4 else 0), "write" - assert shape == "skewed" - return ((arrival // 5) % cells, "write") if arrival % 5 == 0 else (arrival % 4, "read") - - -def distribution(values: list[int]) -> dict: - ordered = sorted(values) - if not ordered: - return dict(count=0) - return dict(count=len(values), **{ - f"p{p}_ms": ordered[(len(ordered) * p + 99) // 100 - 1] / 1000 - for p in (50, 95, 99, 100) - }) - - -def verify_window(control: Path, nodes: int, shape: str, rate_per_node: int, - concurrency: int, window_id: int, positions: dict[int, list[int]]) -> dict: - label = f"entities-{nodes}-{shape}-{rate_per_node}" - metadata, = rows(control / f"{label}-window.tsv") - assert metadata["shape"] == shape - for key, expected in dict(window_id=window_id, nodes=nodes, rate_per_node=rate_per_node, - concurrency=concurrency, seconds=SECONDS).items(): - assert int(metadata[key]) == expected - elapsed_us = int(metadata["elapsed_us"]) - assert elapsed_us >= SECONDS * 1_000_000 - assert int(metadata["ended_boot_ms"]) >= int(metadata["started_boot_ms"]) + SECONDS * 1000 - rate = nodes * rate_per_node - planned = rate * SECONDS - samples = sorted(rows(control / f"{label}.tsv"), key=lambda row: int(row["arrival"])) - assert [int(row["arrival"]) for row in samples] == list(range(planned)), "missing or duplicate arrival" - successes, arrival_latencies, writes = [], [], [0] * nodes - outcomes, intervals = {}, [] - new_positions = {entity: [] for entity in range(nodes * CELLS_PER_NODE)} - for sample in samples: - arrival = int(sample["arrival"]) - scheduled = int(sample["scheduled_us"]) - started = int(sample["started_us"]) - elapsed = int(sample["elapsed_us"]) - entity = int(sample["entity"]) - sequence, read_sequence, count = [int(sample[key]) for key in ("sequence", "read_sequence", "count")] - outcome = sample["outcome"] - assert scheduled == arrival * 1_000_000 // rate - assert started >= scheduled and elapsed >= 0 - assert (entity, sample["kind"]) == destination(shape, arrival, nodes * CELLS_PER_NODE) - assert outcome in {"ok", "resolved", "write_only", "not_started", "absent", "client_full", "scheduler_late", "read_failed"} - outcomes[outcome] = outcomes.get(outcome, 0) + 1 - if outcome in ("client_full", "scheduler_late"): - assert elapsed == sequence == read_sequence == count == 0 - if outcome == "scheduler_late": - assert started >= (arrival + 1) * 1_000_000 // rate - continue - assert started < (arrival + 1) * 1_000_000 // rate - assert started + elapsed <= elapsed_us - intervals += [(started, 1), (started + elapsed, -1)] - if sequence: - assert sample["kind"] == "write" and outcome in ("ok", "resolved", "write_only") - new_positions[entity].append(sequence) - writes[entity // CELLS_PER_NODE] += 1 - if outcome in ("ok", "resolved"): - assert read_sequence >= max(sequence, 1) - assert sequence > 0 if sample["kind"] == "write" else sequence == 0 - successes.append(elapsed) - arrival_latencies.append(started + elapsed - scheduled) - else: - assert read_sequence == count == 0 - assert bool(sequence) == (outcome == "write_only") - for entity, values in new_positions.items(): - previous = positions.setdefault(entity, []) - assert len(values) == len(set(values)), "different requests reused one write receipt" - assert not values or min(values) > max(previous, default=0), "write receipt regressed" - previous.extend(sorted(values)) - for sample in samples: - if sample["outcome"] in ("ok", "resolved"): - expected = bisect.bisect_right(positions[int(sample["entity"])], int(sample["read_sequence"])) - assert int(sample["count"]) == expected, "read value disagrees with its receipt" - checks = rows(control / f"{label}-readback.tsv") - assert [int(row["entity"]) for row in checks] == list(range(nodes * CELLS_PER_NODE)) - for row in checks: - positions_for_cell = positions[int(row["entity"])] - assert int(row["actual"]) == int(row["expected"]) == len(positions_for_cell), "lost or duplicated write" - assert int(row["sequence"]) >= max(positions_for_cell, default=0) - inflight, peak = 0, 0 - for _, delta in sorted(intervals): - inflight += delta - assert inflight >= 0 - peak = max(peak, inflight) - assert inflight == 0 and peak <= concurrency - return dict(nodes=nodes, shape=shape, rate_per_node=rate_per_node, concurrency=concurrency, - planned=planned, outcomes=outcomes, fully_served_arrivals=len(successes) == planned, - acknowledged_writes_by_node=writes, completed_actions=len(successes), - completed_per_second=len(successes) * 1_000_000 / elapsed_us, - service_latency=distribution(successes), arrival_latency=distribution(arrival_latencies), - started_ms=int(metadata["started_ms"]), ended_ms=int(metadata["ended_ms"]), - started_boot_ms=int(metadata["started_boot_ms"]), ended_boot_ms=int(metadata["ended_boot_ms"]), - wall_clock_adjustment_ms=int(metadata["ended_ms"]) - int(metadata["started_ms"]) - elapsed_us / 1000, - elapsed_us=elapsed_us, peak_client_inflight=peak) - - -def verify_entities(control: Path) -> dict: - owners = rows(control / "entity-owners.tsv") - identity = {} - for stage in STAGES: - selected = [row for row in owners if int(row["stage"]) == stage] - assert [int(row["entity"]) for row in selected] == list(range(stage * CELLS_PER_NODE)) - for row in selected: - entity = int(row["entity"]) - assert int(row["owner"]) == entity // CELLS_PER_NODE - value = (row["cell"], row["owner"], row["epoch"], row["incarnation"]) - assert identity.setdefault(entity, value) == value, "existing Cell ownership changed" - ingress = list(map(int, (control / f"entity-ingress-{stage}.txt").read_text().split())) - assert len(ingress) == stage and min(ingress) > 0 and max(ingress) - min(ingress) <= 1 - assert len({value[0] for value in identity.values()}) == 80, "entity targets collapsed" - windows, positions = [], {} - for nodes in STAGES: - for shape in SHAPES: - for rate, concurrency in POINTS: - windows.append(verify_window(control, nodes, shape, rate, concurrency, len(windows), positions)) - roots = rows(control / f"entity-roots-{nodes}.tsv") - assert [int(row["entity"]) for row in roots] == list(range(nodes * CELLS_PER_NODE)) - for row in roots: - entity = int(row["entity"]) - assert (row["cell"], row["owner"], row["epoch"], row["incarnation"]) == identity[entity] - assert positions[entity], f"Cell {entity} received no acknowledged writes" - assert int(row["root_sequence"]) >= max(positions[entity]), "published root does not cover writes" - for node in range(nodes): - assert sum(window["acknowledged_writes_by_node"][node] for window in windows if window["nodes"] == nodes) > 0 - resources = {} - for node in range(20): - samples = rows(control / f"node-{node}-resources.tsv") - assert len(samples) >= 2 - assert all(int(row["active_cells"]) == CELLS_PER_NODE for row in samples) - for column in ("boot_ms", "cpu_usage_us", "throttled_us", "object_started", "object_finished", "bytes_read", "bytes_written"): - values = [int(row[column]) for row in samples] - assert values == sorted(values), f"node {node}: {column} regressed" - assert {int(row["stage"]) for row in samples} >= {stage for stage in STAGES if node < stage} - local, forwarded = map(int, (control / f"node-{node}.counts").read_text().split()) - assert local > 0 and forwarded > 0 - observations = rows(control / f"node-{node}-objects.tsv") - assert len(observations) == 99 and len({(row["operation"], row["outcome"]) for row in observations}) == 99 - assert all(int(row["count"]) >= 0 for row in observations) - assert any(row["operation"] == "put" and row["outcome"] == "success" and int(row["count"]) > 0 for row in observations) - waits = [int(row["object_wait_us"]) for row in rows(control / f"node-{node}-durability.tsv")] - assert waits and min(waits) >= 0 - resources[node] = dict(samples=len(samples), max_memory_current_bytes=max(int(row["memory_current_bytes"]) for row in samples), - max_disk_file_bytes=max(int(row["disk_bytes"]) for row in samples), - gateway_local=local, gateway_forwarded=forwarded, object_wait=distribution(waits), - logical_object_operations=sum(int(row["count"]) for row in observations)) - for window in windows: - if node >= window["nodes"]: - continue - observed = [row for row in samples if window["started_boot_ms"] <= int(row["boot_ms"]) <= window["ended_boot_ms"]] - assert len(observed) >= 2, "missing in-window node samples" - first, last = observed[0], observed[-1] - window.setdefault("node_samples", {})[node] = dict( - first_boot_ms=int(first["boot_ms"]), last_boot_ms=int(last["boot_ms"]), - cpu_usage_us=int(last["cpu_usage_us"]) - int(first["cpu_usage_us"]), - throttled_us=int(last["throttled_us"]) - int(first["throttled_us"]), - logical_object_started=int(last["object_started"]) - int(first["object_started"]), - logical_object_finished=int(last["object_finished"]) - int(first["object_finished"]), - max_memory_current_bytes=max(int(row["memory_current_bytes"]) for row in observed), - max_disk_file_bytes=max(int(row["disk_bytes"]) for row in observed), - max_worker_jobs=max(int(row["worker_jobs"]) for row in observed)) - return dict(integrity_verified=True, windows=windows, resources=resources, - verified_cells=len(positions), acknowledged_writes=sum(map(len, positions.values())), - raw_sha256={path.name: hashlib.sha256(path.read_bytes()).hexdigest() - for path in sorted(control.glob("*.tsv"))}) diff --git a/crates/crab-cell-app/qualification/run.sh b/crates/crab-cell-app/qualification/run.sh deleted file mode 100644 index d056c8b9c..000000000 --- a/crates/crab-cell-app/qualification/run.sh +++ /dev/null @@ -1,44 +0,0 @@ -#!/bin/sh -set -eu -case "${1:-}" in - node) role="node-${CRAB_CELL_PERF_PROCESS_NODE:?}"; selected=process_performance::fleet_process_role ;; - driver) role=driver; selected=process_performance::reference_compose_fleet_end_to_end_performance ;; - scale) role=driver; selected=process_scaling::reference_compose_reader_scaling ;; - rollout) role=rollout; selected=public_host::rollout::three_node_host_rustfs_additive_code_rollout ;; - entities) role=entities; selected=entities::hosts::entity_ledgers_are_isolated_across_three_rustfs_hosts ;; - entity-node) role="node-${CRAB_CELL_PERF_PROCESS_NODE:?}"; selected=entities::process::entity_process_node ;; - entity-scale) role=driver; selected=entities::process::driver::entity_process_scaling ;; - *) printf 'usage: run.sh node|driver|scale|rollout|entities|entity-node|entity-scale\n' >&2; exit 2 ;; -esac -binary= -for candidate in /target/release/deps/reference_application-*; do - if [ -f "$candidate" ] && [ -x "$candidate" ]; then - test -z "$binary" || { printf 'multiple test binaries; use a fresh target\n' >&2; exit 1; } - binary=$candidate - fi -done -test -n "$binary" -test ! -e "/evidence/$role.log" || { printf 'use a fresh evidence directory for each run\n' >&2; exit 1; } -mkdir -p /evidence/control -sha256sum "$binary" > "/evidence/$role-binary.sha256" -snapshot() { - for counter in cpu.max cpu.stat memory.max memory.swap.max memory.peak memory.events; do - printf '%s\n' "$counter" - cat "/sys/fs/cgroup/$counter" - done -} -snapshot > "/evidence/$role-kernel-before.txt" -read -r quota period < /sys/fs/cgroup/cpu.max -test "$quota" -eq "$period" -test "$(cat /sys/fs/cgroup/memory.max)" -eq 1073741824 -test "$(cat /sys/fs/cgroup/memory.swap.max)" -eq 0 -set +e -"$binary" --ignored --exact "reference_application::$selected" --nocapture \ - > "/evidence/$role.log" 2>&1 -result=$? -set -e -snapshot > "/evidence/$role-kernel-after.txt" -cat "/evidence/$role.log" -test "$result" -eq 0 -# An obsolete exact selector runs zero tests and still exits successfully. -grep -Eq 'test result: ok\. 1 passed; 0 failed;' "/evidence/$role.log" diff --git a/crates/crab-cell-app/qualification/scale.py b/crates/crab-cell-app/qualification/scale.py deleted file mode 100644 index 659c6f970..000000000 --- a/crates/crab-cell-app/qualification/scale.py +++ /dev/null @@ -1,351 +0,0 @@ -#!/usr/bin/env python3 -"""Run the built reference application on 3/5/10/20 constrained Compose nodes.""" - -from __future__ import annotations - -import argparse -import bisect -import copy -import csv -import hashlib -import json -import os -from pathlib import Path -import re -import subprocess -import time - -from entities import verify_entities - - -def verify_mixed_load(control: Path, nodes: int, label: str = "mixed") -> dict: - def rows(name): - path = control / name - with path.open(newline="") as source: - result = list(csv.DictReader(source, delimiter="\t")) - hashes[name] = hashlib.sha256(path.read_bytes()).hexdigest() - return result - - hashes = {} - writes = rows(f"{label}-{nodes}-writes.tsv") - baseline, *arrivals = writes - assert baseline["outcome"] == baseline["arrival"] == "baseline" - assert [int(row["arrival"]) for row in arrivals] == list(range(300)) - positions = [int(baseline["sequence"])] - counts = [int(baseline["count"])] - missed = 0 - for index, row in enumerate(arrivals): - assert int(row["scheduled_us"]) == index * 200_000 - delay = int(row["started_us"]) - int(row["scheduled_us"]) - assert delay >= 0 - if row["outcome"] == "scheduler_late": - assert delay >= 200_000 and row["sequence"] == row["count"] == "0" - missed += 1 - continue - assert row["outcome"] == "committed" and delay < 200_000 - assert int(row["sequence"]) > positions[-1] - assert int(row["count"]) == counts[-1] + 1 - positions.append(int(row["sequence"])) - counts.append(int(row["count"])) - assert len(positions) > 1 - reads, behind, lag = 0, 0, 0 - lanes = [] - for lane in range(8): - samples = rows(f"{label}-{nodes}-reader-{lane}.tsv") - assert samples and max(int(row["started_us"]) + int(row["elapsed_us"]) for row in samples) >= 59_000_000 - successes = 0 - for row in samples: - assert 0 <= int(row["started_us"]) < 60_000_000 and int(row["elapsed_us"]) >= 0 - minimum = int(row["minimum_sequence"]) - assert (minimum > 0) == (lane % 2 == 0) - assert counts[0] <= int(row["latest_count"]) <= counts[-1] - if minimum: - assert minimum == positions[int(row["latest_count"]) - counts[0]] - if row["outcome"] == "behind": - assert minimum > 0 and row["sequence"] == row["count"] == "0" - behind += 1 - continue - assert row["outcome"] == "ok" - sequence = int(row["sequence"]) - assert max(minimum, positions[0]) <= sequence <= positions[-1] - index = bisect.bisect_right(positions, sequence) - 1 - assert int(row["count"]) == counts[index], "snapshot value disagrees with its receipt" - lag = max(lag, int(row["latest_count"]) - int(row["count"])) - successes += 1 - assert successes > 0 - reads += successes - lanes.append(successes) - return dict(nodes=nodes, window_seconds=60, planned_writes=300, - acknowledged_writes=len(positions) - 1, missed_writes=missed, - fully_served_writes=missed == 0, successful_reads=reads, - behind_responses=behind, max_acknowledged_count_lag=lag, - successful_reads_by_lane=lanes, raw_sha256=hashes) - - -def verify_reader_loss(control: Path, killed_node: int) -> dict: - result = verify_mixed_load(control, 5, "reader_loss") - fault_path = control / "reader-loss.tsv" - with fault_path.open(newline="") as source: - events = list(csv.DictReader(source, delimiter="\t")) - assert len(events) == 1 - event = {key: int(value) for key, value in events[0].items()} - assert event["killed_node"] == killed_node - requested, killed, ready, served = [event[key] for key in - ("requested_us", "killed_us", "ready_us", "served_us")] - assert 10_000_000 <= requested <= killed <= ready <= served < 50_000_000 - result["raw_sha256"][fault_path.name] = hashlib.sha256(fault_path.read_bytes()).hexdigest() - result["fault"] = event - result["phases"] = {} - # Count calls wholly inside a phase. A pre-fault request completed after - # recovery must not masquerade as service while the reader was unavailable. - for phase, low, high in [("before", 0, requested), ("replacement", killed, served), - ("after", served, 60_000_000)]: - timings = {"writes": [], "reads": []} - lanes = [] - for suffix in ["writes", *[f"reader-{lane}" for lane in range(8)]]: - kind = "writes" if suffix == "writes" else "reads" - with (control / f"reader_loss-5-{suffix}.tsv").open(newline="") as source: - samples = list(csv.DictReader(source, delimiter="\t")) - elapsed = [int(row["elapsed_us"]) for row in samples - if row["outcome"] in ("committed", "ok") - and low <= int(row["started_us"]) - and int(row["started_us"]) + int(row["elapsed_us"]) <= high] - assert elapsed, f"{suffix} made no progress {phase} reader replacement" - timings[kind].extend(elapsed) - if kind == "reads": - lanes.append(len(elapsed)) - summary = {"read_successes_by_lane": lanes} - for kind, elapsed in timings.items(): - ordered = sorted(elapsed) - summary[kind] = dict(successes=len(elapsed), - p99_ms=ordered[(len(ordered) * 99 + 99) // 100 - 1] / 1000, - max_ms=ordered[-1] / 1000) - result["phases"][phase] = summary - return result - - -class Fleet: - def __init__(self, state: Path, project: str, overrides: list[Path], workload: str = "readers"): - assert workload in ("readers", "entities") - self.workload = workload - self.state = state.resolve(strict=True) - self.project = project - if not re.fullmatch(r"[a-z0-9][a-z0-9_-]{0,55}", project): - raise ValueError("use a short, unique lowercase Compose project name") - self.env = dict(os.environ, CRAB_REFERENCE_STATE=str(self.state)) - existing = self.run("docker", "ps", "-aq", "--filter", f"label=com.docker.compose.project={project}") - if existing.strip(): - raise ValueError("project already has containers; retain it and choose a fresh project") - self.evidence = self.state / "evidence" / ("scaling" if workload == "readers" else "entity-scaling") - self.evidence.mkdir(mode=0o1777) - self.evidence.chmod(0o1777) - self.control = self.evidence / "control" - self.control.mkdir(mode=0o1777) - self.control.chmod(0o1777) - source = self.state / "source/crates/crab-cell-app/qualification/compose.yaml" - command = ["docker", "compose", "-p", project, "-f", str(source)] - for override in overrides: - command += ["-f", str(override.resolve(strict=True))] - config = json.loads(self.run(*command, "config", "--format", "json")) - node = config["services"]["node-0"] - # Twenty live nodes plus one killed boot. New boots have new identities; - # restarting a deterministic fixture session would bypass expiry fencing. - for index in range(3, 21): - added = copy.deepcopy(node) - added["environment"].update( - CRAB_CELL_PERF_PROCESS_NODE=str(index), - CRAB_CELL_PERF_PROCESS_ADVERTISE=f"node-{index}:8080", - ) - config["services"][f"node-{index}"] = added - config["services"]["driver"]["command"] = ["scale" if workload == "readers" else "entity-scale"] - if workload == "entities": - for name, service in config["services"].items(): - if name.startswith("node-"): - service["command"] = ["entity-node"] - for service in config["services"].values(): - for volume in service.get("volumes", []): - if volume["target"] == "/evidence": - volume["source"] = str(self.evidence) - self.compose_file = self.evidence / "compose.json" - self.compose_file.write_text(json.dumps(config, indent=2) + "\n") - self.compose_command = ["docker", "compose", "-p", project, "-f", str(self.compose_file)] - self.active: dict[int, str] = {} - self.started = 0 - self.killed: dict[int, str] = {} - self.events: list[dict] = [] - self.driver = "" - - def run(self, *command: str, timeout: int = 180) -> str: - result = subprocess.run(command, env=self.env, text=True, capture_output=True, timeout=timeout) - if result.returncode: - raise RuntimeError(f"{' '.join(command[:4])} failed: {result.stderr[-4000:]}") - return result.stdout - - def compose(self, *args: str, timeout: int = 180) -> str: - return self.run(*self.compose_command, *args, timeout=timeout) - - def inspect(self, container: str) -> dict: - return json.loads(self.run("docker", "inspect", container, timeout=15))[0] - - def limits(self, container: str) -> dict: - observed = self.inspect(container) - config = observed["HostConfig"] - assert config["NanoCpus"] == 1_000_000_000 - assert config["Memory"] == config["MemorySwap"] == 1_073_741_824 - assert config["PidsLimit"] == 256 and config["ReadonlyRootfs"] - assert "ALL" in config["CapDrop"] - assert not observed["State"]["OOMKilled"] - for mount in observed["Mounts"]: - if mount["Destination"] in ("/source", "/target"): - assert not mount["RW"] - return observed - - def scale(self, count: int) -> None: - assert count in (3, 5, 10, 20) and count > len(self.active) - added = list(range(self.started, self.started + count - len(self.active))) - assert added[-1] <= 20 - # The driver's completed dependency chain already created the bucket. - self.compose("up", "-d", "--no-deps", *[f"node-{node}" for node in added]) - for node in added: - container = self.compose("ps", "-aq", f"node-{node}").strip() - assert container - self.limits(container) - self.active[node] = container - self.started += len(added) - - def kill(self, node: int) -> None: - assert node >= 3 and node in self.active and not self.killed - container = self.active[node] - assert self.limits(container)["State"]["Running"] - counters = self.run( - "docker", "exec", container, "sh", "-c", - 'for f in cpu.max cpu.stat memory.max memory.swap.max memory.peak memory.events; ' - 'do echo "$f"; cat "/sys/fs/cgroup/$f"; done', - timeout=15, - ) - (self.evidence / f"node-{node}-kernel-before-kill.txt").write_text(counters) - self.run("docker", "kill", "--signal", "KILL", container, timeout=15) - assert self.run("docker", "wait", container, timeout=15).strip() == "137" - state = self.limits(container)["State"] - assert not state["Running"] and state["ExitCode"] == 137 - self.killed[node] = self.active.pop(node) - - def execute(self) -> None: - self.driver = self.compose("run", "-d", "--name", f"{self.project}-driver", "driver").strip() - self.limits(self.driver) - sequence = 0 - deadline = time.monotonic() + 20 * 60 - while True: - containers = json.loads(self.run("docker", "inspect", self.driver, *self.active.values(), timeout=20)) - if not containers[0]["State"]["Running"]: - assert containers[0]["State"]["ExitCode"] == 0, "driver failed; inspect driver.log" - break - assert time.monotonic() < deadline, "scaling driver exceeded 20 minutes" - for node, observed in zip(self.active, containers[1:], strict=True): - state = observed["State"] - assert state["Running"] or ( - (self.control / "stop").exists() and state["ExitCode"] == 0 - ), f"node {node} exited unexpectedly: {state}" - request = self.control / f"fleet-{sequence}.request" - if request.exists(): - action, argument = request.read_text().split() - count = int(argument) - started = time.time() - if action == "scale": - self.scale(count) - elif action == "kill": - self.kill(count) - else: - raise ValueError(f"unknown controller command: {action}") - event = dict(sequence=sequence, action=action, argument=count, - started_at=started, completed_at=time.time(), active_nodes=sorted(self.active)) - self.events.append(event) - (self.evidence / "events.json").write_text(json.dumps(self.events, indent=2) + "\n") - temporary = request.with_suffix(".tmp") - temporary.write_text(" ".join(map(str, sorted(self.active)))) - temporary.replace(request.with_suffix(".done")) - print(json.dumps(event), flush=True) - sequence += 1 - time.sleep(0.5) - self.verify() - - def verify(self) -> None: - assert len(self.active) == 20 and len(self.killed) == (1 if self.workload == "readers" else 0) - assert [(event["action"], event["argument"]) for event in self.events if event["action"] == "scale"] == [ - ("scale", 3), ("scale", 5), ("scale", 10), ("scale", 20) - ] - roles = {f"node-{node}": container for node, container in self.active.items()} - roles["driver"] = self.driver - reports = {} - volumes = set() - binaries = set() - for role, container in roles.items(): - assert self.run("docker", "wait", container, timeout=30).strip() == "0", role - observed = self.limits(container) - mounts = [mount["Name"] for mount in observed["Mounts"] if mount["Destination"] == "/scratch"] - assert len(mounts) == 1 and mounts[0] not in volumes - volumes.update(mounts) - log = (self.evidence / f"{role}.log").read_text() - assert "test result: ok. 1 passed; 0 failed;" in log, role - if role != "driver": - node = role.split("-")[1] - assert re.search(rf"node_{node}_session_withdrawn: generation=\d+", log) - if self.workload == "readers": - assert f"node_{node}_reader_drained: activation_closed=1 resolver_closed=1" in log - counters = (self.evidence / f"{role}-kernel-after.txt").read_text() - assert "cpu.max\n100000 100000\n" in counters - assert "memory.max\n1073741824\n" in counters - assert "memory.swap.max\n0\n" in counters - assert re.search(r"^oom 0$", counters, re.M) and re.search(r"^oom_kill 0$", counters, re.M) - peak = int(re.search(r"memory.peak\n(\d+)", counters)[1]) - reports[role] = dict(exit_code=observed["State"]["ExitCode"], memory_peak_bytes=peak) - binaries.add((self.evidence / f"{role}-binary.sha256").read_text().split()[0]) - assert len(binaries) == 1 - source = (self.state / "evidence/source-revision.txt").read_text().strip() - if self.workload == "entities": - result = dict(workload=self.workload, source=source, binary_sha256=binaries.pop(), - roles=reports, events=self.events, **verify_entities(self.control)) - (self.evidence / "verification.json").write_text(json.dumps(result, indent=2) + "\n") - print(f"Verified entity integrity and resources at 3/5/10/20 nodes; evidence: {self.evidence}", flush=True) - return - killed = next(iter(self.killed)) - driver_log = (self.evidence / "driver.log").read_text() - for nodes, readers in [(3, 2), (5, 3), (10, 9), (20, 19)]: - assert f"PERF reader_scale: nodes={nodes} readers={readers} " in driver_log - mixed = [verify_mixed_load(self.control, nodes) for nodes in (3, 5, 10, 20)] - assert f"killed_node={killed} ready_readers=3 exact_queries=12 " in driver_log - result = dict(verified=True, source=source, binary_sha256=binaries.pop(), - roles=reports, killed_node=killed, events=self.events, mixed_load=mixed, - reader_loss_load=verify_reader_loss(self.control, killed)) - (self.evidence / "verification.json").write_text(json.dumps(result, indent=2) + "\n") - print(f"Verified 3/5/10/20 nodes and reader replacement; evidence: {self.evidence}", flush=True) - - def retain(self) -> None: - try: - (self.evidence / "compose.log").write_text(self.compose("logs", "--no-color")) - containers = self.compose("ps", "-aq").split() - if containers: - (self.evidence / "containers.json").write_text(self.run("docker", "inspect", *containers)) - rustfs = self.compose("ps", "-aq", "rustfs").strip() - if rustfs: - self.run("docker", "cp", f"{rustfs}:/data/logs", str(self.evidence / "rustfs-logs")) - finally: - self.compose("stop") - - -def main() -> None: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--state", type=Path, required=True, help="prepared source, binary and evidence directory") - parser.add_argument("--project", required=True, help="fresh Compose project; stopped containers are retained") - parser.add_argument("--compose-file", type=Path, action="append", default=[], help="explicit image/cache override") - parser.add_argument("--workload", choices=("readers", "entities"), default="readers", help="reader replacement or scheduled writable entity traffic") - args = parser.parse_args() - fleet = Fleet(args.state, args.project, args.compose_file, args.workload) - try: - fleet.execute() - finally: - fleet.retain() - - -if __name__ == "__main__": - main() diff --git a/crates/crab-cell-app/qualification/test_entities.py b/crates/crab-cell-app/qualification/test_entities.py deleted file mode 100644 index 28035d2e8..000000000 --- a/crates/crab-cell-app/qualification/test_entities.py +++ /dev/null @@ -1,79 +0,0 @@ -"""Reject false capacity and durability claims in the entity workload receipt.""" - -import csv -from pathlib import Path -import tempfile -import unittest - -from entities import verify_window - - -class EntityWindowEvidence(unittest.TestCase): - def setUp(self): - temporary = tempfile.TemporaryDirectory() - self.addCleanup(temporary.cleanup) - self.root = Path(temporary.name) - self.label = "entities-3-uniform-1" - (self.root / f"{self.label}-window.tsv").write_text( - "window_id\tnodes\tshape\trate_per_node\tconcurrency\tseconds\tstarted_ms\tended_ms\tstarted_boot_ms\tended_boot_ms\telapsed_us\n" - "0\t3\tuniform\t1\t4\t10\t100000\t110000\t100000\t110000\t10000000\n") - counts = [0] * 12 - lines = ["arrival\tscheduled_us\tstarted_us\telapsed_us\tentity\tkind\toutcome\tsequence\tread_sequence\tcount"] - for arrival in range(30): - entity = arrival % 12 - counts[entity] += 1 - sequence = counts[entity] * 2 + 1 - scheduled = arrival * 1_000_000 // 3 - lines.append(f"{arrival}\t{scheduled}\t{scheduled + 10}\t1000\t{entity}\twrite\tok\t{sequence}\t{sequence}\t{counts[entity]}") - (self.root / f"{self.label}.tsv").write_text("\n".join(lines) + "\n") - (self.root / f"{self.label}-readback.tsv").write_text( - "entity\texpected\tactual\tsequence\n" + "".join( - f"{entity}\t{count}\t{count}\t{count * 2 + 1}\n" for entity, count in enumerate(counts))) - - def change(self, suffix, update): - path = self.root / f"{self.label}{suffix}.tsv" - with path.open(newline="") as source: - rows = list(csv.DictReader(source, delimiter="\t")) - update(rows) - with path.open("w", newline="") as target: - writer = csv.DictWriter(target, fieldnames=rows[0], delimiter="\t") - writer.writeheader() - writer.writerows(rows) - - def verify(self): - return verify_window(self.root, 3, "uniform", 1, 4, 0, {}) - - def test_complete_window_has_independent_owner_work(self): - result = self.verify() - self.assertEqual((result["fully_served_arrivals"], result["acknowledged_writes_by_node"]), (True, [12, 10, 8])) - - def test_value_at_wrong_receipt_is_rejected(self): - self.change("", lambda rows: rows[0].update(count="2")) - with self.assertRaisesRegex(AssertionError, "read value disagrees"): - self.verify() - - def test_wall_clock_adjustment_does_not_shorten_the_load_window(self): - self.change("-window", lambda rows: rows[0].update(ended_ms="109898")) - result = self.verify() - self.assertEqual((result["fully_served_arrivals"], result["wall_clock_adjustment_ms"]), (True, -102)) - - def test_missing_arrival_is_rejected(self): - self.change("", lambda rows: rows.pop()) - with self.assertRaisesRegex(AssertionError, "missing or duplicate arrival"): - self.verify() - - def test_missing_persisted_write_is_rejected(self): - self.change("-readback", lambda rows: rows[0].update(actual="2", expected="2")) - with self.assertRaisesRegex(AssertionError, "lost or duplicated write"): - self.verify() - - def test_missed_arrival_is_not_fully_served(self): - self.change("", lambda rows: rows[-1].update(started_us="10000000", elapsed_us="0", - outcome="scheduler_late", sequence="0", read_sequence="0", count="0")) - self.change("-readback", lambda rows: rows[5].update(actual="2", expected="2")) - result = self.verify() - self.assertEqual((result["fully_served_arrivals"], result["completed_actions"]), (False, 29)) - - -if __name__ == "__main__": - unittest.main() diff --git a/crates/crab-cell-app/qualification/test_scale.py b/crates/crab-cell-app/qualification/test_scale.py deleted file mode 100644 index ea7a533e1..000000000 --- a/crates/crab-cell-app/qualification/test_scale.py +++ /dev/null @@ -1,125 +0,0 @@ -"""Reject incomplete or false mixed-workload receipts before qualification.""" - -import csv -from pathlib import Path -import tempfile -import unittest - -from scale import verify_mixed_load, verify_reader_loss - - -class MixedLoadEvidence(unittest.TestCase): - def setUp(self): - self.temporary = tempfile.TemporaryDirectory() - self.addCleanup(self.temporary.cleanup) - self.root = Path(self.temporary.name) - writes = ["arrival\tscheduled_us\tstarted_us\telapsed_us\toutcome\tsequence\tcount", - "baseline\t0\t0\t0\tbaseline\t100\t10"] - for index in range(300): - writes.append(f"{index}\t{index * 200_000}\t{index * 200_000 + 100}\t1000\tcommitted\t{102 + index * 2}\t{11 + index}") - (self.root / "mixed-3-writes.tsv").write_text("\n".join(writes) + "\n") - for lane in range(8): - strict = lane % 2 == 0 - lines = ["started_us\telapsed_us\tminimum_sequence\tlatest_count\toutcome\tsequence\tcount"] - if strict: - lines.append("59000000\t1000\t700\t310\tbehind\t0\t0") - lines.append(f"59999000\t3000\t{700 if strict else 0}\t310\tok\t{700 if strict else 699}\t{310 if strict else 309}") - (self.root / f"mixed-3-reader-{lane}.tsv").write_text("\n".join(lines) + "\n") - - def change(self, name, update): - path = self.root / name - with path.open(newline="") as source: - rows = list(csv.DictReader(source, delimiter="\t")) - update(rows) - with path.open("w", newline="") as target: - writer = csv.DictWriter(target, fieldnames=rows[0], delimiter="\t") - writer.writeheader() - writer.writerows(rows) - - def test_observed_lag_and_typed_behind_remain_visible(self): - result = verify_mixed_load(self.root, 3) - self.assertEqual((result["acknowledged_writes"], result["successful_reads"], - result["behind_responses"], result["max_acknowledged_count_lag"]), - (300, 8, 4, 1)) - - def test_stale_value_at_covering_receipt_is_rejected(self): - self.change("mixed-3-reader-0.tsv", lambda rows: rows[-1].update(count="309")) - with self.assertRaisesRegex(AssertionError, "snapshot value"): - verify_mixed_load(self.root, 3) - - def test_result_below_requested_minimum_is_rejected(self): - self.change("mixed-3-reader-0.tsv", lambda rows: rows[-1].update(sequence="699", count="309")) - with self.assertRaises(AssertionError): - verify_mixed_load(self.root, 3) - - def test_missing_scheduled_arrival_is_rejected(self): - self.change("mixed-3-writes.tsv", lambda rows: rows.pop()) - with self.assertRaises(AssertionError): - verify_mixed_load(self.root, 3) - - def test_duplicate_effect_is_rejected(self): - self.change("mixed-3-writes.tsv", lambda rows: rows[-1].update(count="311")) - with self.assertRaises(AssertionError): - verify_mixed_load(self.root, 3) - - def test_missed_arrival_cannot_be_reported_as_fully_served(self): - self.change("mixed-3-writes.tsv", lambda rows: rows[-1].update( - started_us="60000000", elapsed_us="0", outcome="scheduler_late", sequence="0", count="0")) - for lane in range(8): - def update(rows): - for row in rows: - row["latest_count"] = "309" - if int(row["minimum_sequence"]): - row["minimum_sequence"] = "698" - if row["outcome"] == "ok": - row["sequence"] = str(int(row["sequence"]) - 2) - row["count"] = str(int(row["count"]) - 1) - self.change(f"mixed-3-reader-{lane}.tsv", update) - result = verify_mixed_load(self.root, 3) - self.assertEqual((result["acknowledged_writes"], result["missed_writes"], result["fully_served_writes"]), - (299, 1, False)) - - -class ReaderLossEvidence(unittest.TestCase): - change = MixedLoadEvidence.change - - def setUp(self): - MixedLoadEvidence.setUp(self) - for path in self.root.glob("mixed-3-*.tsv"): - (self.root / path.name.replace("mixed-3", "reader_loss-5")).write_bytes(path.read_bytes()) - (self.root / "reader-loss.tsv").write_text( - "killed_node\trequested_us\tkilled_us\tready_us\tserved_us\n" - "3\t10000000\t10500000\t24000000\t25000000\n") - for lane in range(8): - strict = lane % 2 == 0 - def update(rows): - for start, sequence, count in [(5_000_000, 150, 35), (40_000_000, 500, 210), (15_000_000, 250, 85)]: - rows.insert(0, dict(started_us=str(start), elapsed_us="1000", - minimum_sequence=str(sequence if strict else 0), - latest_count=str(count), outcome="ok", sequence=str(sequence), - count=str(count))) - self.change(f"reader_loss-5-reader-{lane}.tsv", update) - - def test_live_fault_reports_each_phase(self): - result = verify_reader_loss(self.root, 3) - self.assertEqual([result["phases"][phase]["reads"]["successes"] - for phase in ("before", "replacement", "after")], [8, 8, 8]) - - def test_fault_after_load_is_rejected(self): - self.change("reader-loss.tsv", lambda rows: rows[0].update( - requested_us="60000000", killed_us="61000000", ready_us="70000000", served_us="71000000")) - with self.assertRaises(AssertionError): - verify_reader_loss(self.root, 3) - - def test_post_recovery_reads_do_not_prove_service_during_replacement(self): - self.change("reader_loss-5-reader-0.tsv", lambda rows: rows[0].update(elapsed_us="11000000")) - with self.assertRaisesRegex(AssertionError, "no progress replacement"): - verify_reader_loss(self.root, 3) - - def test_wrong_killed_node_is_rejected(self): - with self.assertRaises(AssertionError): - verify_reader_loss(self.root, 4) - - -if __name__ == "__main__": - unittest.main() diff --git a/crates/crab-cell-app/src/client_macro.rs b/crates/crab-cell-app/src/client_macro.rs deleted file mode 100644 index 1a5a1eacf..000000000 --- a/crates/crab-cell-app/src/client_macro.rs +++ /dev/null @@ -1,173 +0,0 @@ -//! Generated typed application client bindings. - -/// Generates scoped typed Cell accessors from explicit namespace and operation IDs. -/// -/// Each generated constructor checks its declared IDs against the compiled -/// application and typed operation traits before any call can start. -/// -/// ```no_run -/// mod proof { -/// # include!(concat!(env!("CARGO_MANIFEST_DIR"), "/tests/support/generated_compile_setup.rs")); -/// # use crab_cell_runtime::MutationIdentity; -/// fn typed(cell: &Entity, identity: MutationIdentity) { -/// let _ = cell.set(identity, ()); -/// } -/// } -/// ``` -/// -/// ```compile_fail -/// mod proof { -/// # include!(concat!(env!("CARGO_MANIFEST_DIR"), "/tests/support/generated_compile_setup.rs")); -/// # use crab_cell_runtime::MutationIdentity; -/// fn wrong_input(cell: &Entity, identity: MutationIdentity) { -/// let _ = cell.set(identity, 42_u32); -/// } -/// } -/// ``` -/// -/// ```compile_fail -/// mod proof { -/// # include!(concat!(env!("CARGO_MANIFEST_DIR"), "/tests/support/generated_compile_setup.rs")); -/// # use crab_cell_runtime::MutationIdentity; -/// fn unlisted_operation(cell: &Entity, identity: MutationIdentity) { -/// let _ = cell.delete(identity, ()); -/// } -/// } -/// ``` -/// -/// ```compile_fail -/// mod proof { -/// # include!(concat!(env!("CARGO_MANIFEST_DIR"), "/tests/support/generated_compile_setup.rs")); -/// fn wrong_key(client: &Client) { -/// let _ = client.entity(b"untyped-key"); -/// } -/// } -/// ``` -#[macro_export] -macro_rules! cell_client { - ( - $visibility:vis struct $client:ident ($application:ty) { - $( - $cell_visibility:vis fn $accessor:ident ( $scope:ident : &$key:ty ) -> $cell:ident { - namespace: $namespace:expr, - module: $module:expr, - commands: { $( $command_visibility:vis fn $command_method:ident, $prepare_method:ident : $command:ty = $command_id:expr; )* }, - queries: { $( $query_visibility:vis fn $query_method:ident : $query:ty = $query_id:expr; )* } - } - )* - } - ) => { - $visibility struct $client { - handle: $crate::ApplicationHandle<$application>, - } - - impl $client { - /// Validates generated bindings before accepting an application handle. - $visibility fn new(handle: $crate::ApplicationHandle<$application>) -> crab_cell_runtime::Result { - $( - let namespace: crab_cell_runtime::NamespaceId = const { $namespace }; - let module: &'static str = const { $module }; - let declared = handle.compiled().cell_types().iter() - .find(|cell_type| cell_type.namespace() == namespace) - .ok_or(crab_cell_runtime::Error::Registry("generated namespace is not declared"))?; - if declared.module() != module { - return Err(crab_cell_runtime::Error::Registry("generated module differs from descriptor")); - } - $( - if <$command as crab_cell_runtime::registry::Command>::MODULE != module - || <$command as crab_cell_runtime::registry::Command>::ID != const { $command_id } - { - return Err(crab_cell_runtime::Error::Registry("generated command differs from stable ID")); - } - handle.compiled().registry().command_contract::<$command>(namespace)?; - )* - $( - if <$query as crab_cell_runtime::registry::Query>::MODULE != module - || <$query as crab_cell_runtime::registry::Query>::ID != const { $query_id } - { - return Err(crab_cell_runtime::Error::Registry("generated query differs from stable ID")); - } - handle.compiled().registry().query_contract::<$query>(namespace)?; - )* - )* - Ok(Self { handle }) - } - - /// Resolves an ambiguous mutation through the application's durable request ledger. - $visibility async fn resolve( - &self, - pending: &crab_cell_runtime::client::PendingMutation, - ) -> std::result::Result< - crab_cell_runtime::cell::executor::Resolution, - crab_cell_runtime::client::InvocationError>, - > { - self.handle.resolve(pending).await - } - - $( - /// Selects a Cell using the descriptor's canonical partition function. - $cell_visibility fn $accessor(&self, $scope: &$key) -> crab_cell_runtime::Result<$cell> { - let key = <$key as $crate::CellKey>::canonical_bytes($scope); - Ok($cell { - handle: self.handle.clone(), - target: self.handle.target_for_scope(const { $namespace }, key)?, - }) - } - )* - } - - $( - $cell_visibility struct $cell { - handle: $crate::ApplicationHandle<$application>, - target: crab_cell_runtime::identity::CellTarget, - } - - impl $cell { - /// Returns the stable scoped target selected by this generated binding. - $cell_visibility fn target(&self) -> &crab_cell_runtime::identity::CellTarget { - &self.target - } - - $( - /// Invokes the declared typed command on this scoped Cell. - $command_visibility async fn $command_method( - &self, - identity: crab_cell_runtime::MutationIdentity, - input: <$command as crab_cell_runtime::registry::Command>::Input, - ) -> std::result::Result< - crab_cell_runtime::client::Committed<<$command as crab_cell_runtime::registry::Command>::Output>, - crab_cell_runtime::client::InvocationError<<$command as crab_cell_runtime::registry::Command>::Output>, - > { - self.handle.command::<$command>(&self.target, identity, input).await - } - - /// Prepares the declared command for outcome-aware execution. - $command_visibility async fn $prepare_method( - &self, - identity: crab_cell_runtime::MutationIdentity, - input: <$command as crab_cell_runtime::registry::Command>::Input, - ) -> std::result::Result< - crab_cell_runtime::client::PreparedCommand<$command>, - crab_cell_runtime::client::InvocationError<<$command as crab_cell_runtime::registry::Command>::Output>, - > { - self.handle.prepare_command::<$command>(&self.target, identity, input).await - } - )* - - $( - /// Invokes the declared typed query on this scoped Cell. - $query_visibility async fn $query_method( - &self, - minimum: Option, - input: <$query as crab_cell_runtime::registry::Query>::Input, - ) -> std::result::Result< - crab_cell_runtime::client::Observed<<$query as crab_cell_runtime::registry::Query>::Output>, - crab_cell_runtime::client::InvocationError<<$query as crab_cell_runtime::registry::Query>::Output>, - > { - self.handle.query::<$query>(&self.target, minimum, input).await - } - )* - } - )* - }; -} diff --git a/crates/crab-cell-app/src/lib.rs b/crates/crab-cell-app/src/lib.rs deleted file mode 100644 index 9b48fee86..000000000 --- a/crates/crab-cell-app/src/lib.rs +++ /dev/null @@ -1,748 +0,0 @@ -//! Stable author-facing composition for statically linked Cell applications. -//! -//! This crate compiles the existing runtime registry together with a bounded, -//! deterministic topology descriptor. Node lifecycle, storage providers, -//! authority and HTTP policy remain outside this boundary. - -#![deny(missing_docs)] -// A panic in a filter process or FUSE path corrupts a worktree, so production -// builds deny unwrap, expect, panic, todo, and unimplemented; test builds keep -// them available. -#![cfg_attr( - not(test), - deny( - clippy::unwrap_used, - clippy::expect_used, - clippy::panic, - clippy::todo, - clippy::unimplemented - ) -)] - -use std::{marker::PhantomData, sync::Arc}; - -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::executor::Resolution; -use crab_cell_runtime::client::{ - CellClient, Committed, InvocationError, Observed, PendingMutation, PreparedCommand, -}; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, TenantId, partition_for_shard, -}; -use crab_cell_runtime::primitives::blob::{BlobArtifactStore, BlobModule, BlobNamespace}; -use crab_cell_runtime::primitives::cron::{CronModule, CronNamespace}; -use crab_cell_runtime::primitives::effects::{EffectModule, EffectSource}; -use crab_cell_runtime::primitives::kv::{KvModule, KvNamespace}; -use crab_cell_runtime::primitives::queue::{QueueModule, QueueNamespace}; -use crab_cell_runtime::primitives::sql::{SqlCell, SqlModule}; -use crab_cell_runtime::primitives::workflow::{ - WorkflowActivities, WorkflowActivityModule, WorkflowModule, WorkflowNamespace, -}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, Command, Query, Registry, RegistryBuilder, -}; -use crab_cell_runtime::{Error, Result}; - -const DESCRIPTOR_MAGIC: &[u8] = b"crab.application.v1\0"; -const MAX_APPLICATION_NAME_BYTES: usize = 128; -const MAX_CELL_TYPES: usize = 128; -const MAX_DESCRIPTOR_BYTES: usize = 256 * 1024; -const MAX_PARTITION_VERSION: u32 = 2; -const ENTITY_PARTITION_VERSION: u32 = 2; -const ENTITY_PARTITION_PREFIX: u8 = 1; - -/// One application-owned Cell topology declaration. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct CellType { - module: &'static str, - name: &'static str, - namespace: NamespaceId, - role: CatalogRole, - shards: u32, - partition_version: u32, - schema_min: u32, - schema_max: u32, - database_limit_bytes: u64, - capture_limit_bytes: u64, -} - -impl CellType { - /// Creates a bounded topology declaration with conservative version one defaults. - pub fn new( - module: &'static str, - name: &'static str, - namespace: NamespaceId, - role: CatalogRole, - shards: u32, - ) -> Result { - let cell_type = Self { - module, - name, - namespace, - role, - shards, - partition_version: 1, - schema_min: 1, - schema_max: 1, - database_limit_bytes: 64 * 1024 * 1024, - capture_limit_bytes: 16 * 1024 * 1024, - }; - cell_type.validate()?; - Ok(cell_type) - } - - /// Replaces the declared application limits without changing stable identity. - /// - /// Database limits must be at least 512 bytes and capture limits at least 128 bytes. - pub fn with_limits( - mut self, - database_limit_bytes: u64, - capture_limit_bytes: u64, - ) -> Result { - self.database_limit_bytes = database_limit_bytes; - self.capture_limit_bytes = capture_limit_bytes; - self.validate()?; - Ok(self) - } - - /// Selects stable entity partitions instead of fixed shards. - /// - /// The namespace descriptor must declare one shard. The application - /// derives one target per entity with [`Self::entity_partition`]. - pub fn with_entity_partitions(mut self) -> Result { - self.partition_version = ENTITY_PARTITION_VERSION; - self.validate()?; - Ok(self) - } - - /// Replaces the schema range while retaining the stable Cell identity. - pub fn with_schema_range(mut self, schema_min: u32, schema_max: u32) -> Result { - self.schema_min = schema_min; - self.schema_max = schema_max; - self.validate()?; - Ok(self) - } - - /// Returns the stable module owner. - #[must_use] - pub const fn module(&self) -> &'static str { - self.module - } - - /// Returns the diagnostic Cell type name. - #[must_use] - pub const fn name(&self) -> &'static str { - self.name - } - - /// Returns the stable namespace identity. - #[must_use] - pub const fn namespace(&self) -> NamespaceId { - self.namespace - } - - /// Returns the catalog role pinned for this type. - #[must_use] - pub const fn role(&self) -> CatalogRole { - self.role - } - - /// Returns the fixed shard count, or one for an entity Cell type. - #[must_use] - pub const fn shards(&self) -> u32 { - self.shards - } - - /// Returns the database ceiling declared for each Cell of this type. - #[must_use] - pub const fn database_limit_bytes(&self) -> u64 { - self.database_limit_bytes - } - - /// Returns the capture ceiling declared for each Cell of this type. - #[must_use] - pub const fn capture_limit_bytes(&self) -> u64 { - self.capture_limit_bytes - } - - /// Maps one bounded application scope to its stable fixed shard number. - pub fn shard_for_scope(&self, scope: &[u8]) -> Result { - if self.partition_version == ENTITY_PARTITION_VERSION { - return Err(Error::Identity("entity Cell type has no fixed shard")); - } - crab_cell_runtime::shard_for_scope(self.namespace, scope, self.shards) - } - - /// Returns the canonical fixed-shard partition for one application scope. - pub fn partition_for_scope(&self, scope: &[u8]) -> Result<[u8; 4]> { - Ok(crab_cell_runtime::partition_for_shard( - self.shard_for_scope(scope)?, - )) - } - - /// Derives the canonical partition bytes for one entity identity. - /// - /// The scope must be nonempty and at most 1,024 bytes. Its digest, not the - /// raw identity, enters the catalog partition key. - pub fn entity_partition(&self, scope: &[u8]) -> Result<[u8; 33]> { - if self.partition_version != ENTITY_PARTITION_VERSION { - return Err(Error::Identity("Cell type uses fixed shards")); - } - if scope.is_empty() || scope.len() > 1_024 { - return Err(Error::Identity("invalid entity scope length")); - } - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.entity.partition.v1\0"); - hasher.update(self.namespace.as_bytes()); - hasher.update(&(scope.len() as u32).to_be_bytes()); - hasher.update(scope); - let mut partition = [0_u8; 33]; - partition[0] = ENTITY_PARTITION_PREFIX; - partition[1..].copy_from_slice(hasher.finalize().as_bytes()); - Ok(partition) - } - - fn valid_partition(&self, partition: &[u8]) -> bool { - if self.partition_version == ENTITY_PARTITION_VERSION { - return partition.len() == 33 && partition[0] == ENTITY_PARTITION_PREFIX; - } - let Ok(shard) = <[u8; 4]>::try_from(partition) else { - return false; - }; - let shard = u32::from_be_bytes(shard); - shard < self.shards && partition_for_shard(shard).as_slice() == partition - } - - fn validate(&self) -> Result<()> { - if !valid_name(self.module) - || !valid_name(self.name) - || self.namespace.as_bytes().iter().all(|byte| *byte == 0) - || !(1..=4096).contains(&self.shards) - || !self.shards.is_power_of_two() - || self.partition_version == 0 - || self.partition_version > MAX_PARTITION_VERSION - || (self.partition_version == ENTITY_PARTITION_VERSION && self.shards != 1) - || self.schema_min == 0 - || self.schema_min > self.schema_max - || self.database_limit_bytes < 512 - || self.capture_limit_bytes < 128 - { - return Err(Error::Registry("invalid Cell type declaration")); - } - Ok(()) - } -} - -/// Startup-only compiler for one static application. -pub struct ApplicationBuilder { - name: &'static str, - registry: RegistryBuilder, - cell_types: Vec, -} - -impl ApplicationBuilder { - /// Creates an application compiler from the same build evidence as the runtime registry. - pub fn new(name: &'static str, build: BuildDescriptor) -> Result { - if !valid_name(name) { - return Err(Error::Registry("invalid application name")); - } - Ok(Self { - name, - registry: RegistryBuilder::new(build), - cell_types: Vec::new(), - }) - } - - /// Registers one statically linked runtime module. - pub fn register(&mut self, module: M) -> Result<()> { - self.registry.register(module) - } - - /// Adds one topology declaration and rejects duplicate stable identities. - pub fn cell_type(&mut self, cell_type: CellType) -> Result<()> { - cell_type.validate()?; - if self.cell_types.iter().any(|existing| { - existing.namespace == cell_type.namespace || existing.name == cell_type.name - }) { - return Err(Error::Registry("duplicate Cell type identity")); - } - self.cell_types.push(cell_type); - Ok(()) - } - - /// Freezes the registry and emits deterministic application descriptor bytes. - pub fn finish(mut self) -> Result { - if self.cell_types.is_empty() || self.cell_types.len() > MAX_CELL_TYPES { - return Err(Error::Registry("Cell type count must be in 1..=128")); - } - self.cell_types - .sort_by_key(|cell_type| *cell_type.namespace.as_bytes()); - let registry = Arc::new(self.registry.finish()?); - if registry.namespace_count() != self.cell_types.len() { - return Err(Error::Registry( - "application topology does not declare every registry namespace", - )); - } - for cell_type in &self.cell_types { - let Some((module, namespace)) = registry.namespace_contract(cell_type.namespace) else { - return Err(Error::Registry("Cell type namespace is not registered")); - }; - if module != cell_type.module - || namespace.role != cell_type.role - || namespace.shards != cell_type.shards - { - return Err(Error::Registry("Cell type differs from compiled namespace")); - } - if registry.module_schema_range(cell_type.module) - != Some((cell_type.schema_min, cell_type.schema_max)) - { - return Err(Error::Registry( - "Cell type schema range differs from compiled module", - )); - } - } - let bytes = encode_descriptor(self.name, ®istry, &self.cell_types)?; - let digest = Digest::from_bytes(*blake3::hash(&bytes).as_bytes()); - Ok(CompiledApplication { - registry, - descriptor: bytes, - digest, - name: self.name, - cell_types: self.cell_types, - }) - } -} - -/// Immutable application artifact consumed by node composition. -pub struct CompiledApplication { - registry: Arc, - descriptor: Vec, - digest: Digest, - name: &'static str, - cell_types: Vec, -} - -impl CompiledApplication { - /// Returns the runtime registry used by every application invocation. - #[must_use] - pub fn registry(&self) -> Arc { - Arc::clone(&self.registry) - } - - /// Returns canonical topology and release descriptor bytes. - #[must_use] - pub fn descriptor_bytes(&self) -> &[u8] { - &self.descriptor - } - - /// Returns the digest of [`Self::descriptor_bytes`]. - #[must_use] - pub const fn descriptor_digest(&self) -> Digest { - self.digest - } - - /// Returns the stable diagnostic application name. - #[must_use] - pub const fn name(&self) -> &'static str { - self.name - } - - /// Returns the sorted topology inventory. - #[must_use] - pub fn cell_types(&self) -> &[CellType] { - &self.cell_types - } - - fn validate_module(&self, namespace: NamespaceId, module: &'static str) -> Result<()> { - let Some(cell_type) = self - .cell_types - .iter() - .find(|cell_type| cell_type.namespace == namespace) - else { - return Err(Error::Registry("namespace is not declared by application")); - }; - let Some((registered_module, _)) = self.registry.namespace_contract(namespace) else { - return Err(Error::Registry("namespace contract is missing")); - }; - if registered_module != module { - return Err(Error::Registry("namespace module differs from capability")); - } - if cell_type.module != module { - return Err(Error::Registry("Cell type module differs from capability")); - } - Ok(()) - } -} - -/// Trait implemented by a statically linked application root. -pub trait CellApplication: Send + Sync + 'static { - /// Stable application name carried into the compiled release descriptor. - const NAME: &'static str; - - /// Registers every module, namespace, and relationship this application - /// exposes. The builder rejects a registration that does not match the - /// compiled function bindings. - fn register(builder: &mut ApplicationBuilder) -> Result<()>; - - /// Compiles this application using deterministic build evidence. - fn compile(build: BuildDescriptor) -> Result { - let mut builder = ApplicationBuilder::new(Self::NAME, build)?; - Self::register(&mut builder)?; - builder.finish() - } -} - -/// Stable key encoding for one generated Cell accessor. -/// -/// Implementations must return the same bounded bytes for the same logical key -/// across compatible releases; changing them moves the key to another Cell. -pub trait CellKey { - /// Returns the stored canonical bytes used by the declared partition function. - fn canonical_bytes(&self) -> &[u8]; -} - -/// Tenant/application-bound typed capability over an already-started client. -pub struct ApplicationHandle { - client: CellClient, - tenant: TenantId, - application: ApplicationId, - compiled: Arc, - marker: PhantomData A>, -} - -impl Clone for ApplicationHandle { - fn clone(&self) -> Self { - Self { - client: self.client.clone(), - tenant: self.tenant, - application: self.application, - compiled: Arc::clone(&self.compiled), - marker: PhantomData, - } - } -} - -impl ApplicationHandle { - /// Binds a compiled application and matching client to one tenant and application identity. - /// - /// Returns an error before any invocation when the author type or client - /// registry does not match the hosted application artifact. - pub fn new( - client: CellClient, - compiled: Arc, - tenant: TenantId, - application: ApplicationId, - ) -> Result { - if compiled.name() != A::NAME { - return Err(Error::Registry( - "application type differs from compiled application", - )); - } - if client.registry_digest() != compiled.registry.release_digest() { - return Err(Error::Registry( - "client registry differs from compiled application", - )); - } - Ok(Self { - client, - tenant, - application, - compiled, - marker: PhantomData, - }) - } - - /// Returns an application capability with an explicit typed-query read policy. - /// - /// Replica policy requires host-configured readers and never falls back to - /// owner reads. Commands, outcome resolution, streams and primitive lease - /// validation retain owner order. - #[must_use] - pub fn with_read_policy(&self, policy: crab_cell_runtime::client::ReadPolicy) -> Self { - let mut handle = self.clone(); - handle.client = handle.client.with_read_policy(policy); - handle - } - - /// Returns a handle whose Blob capability uses the configured object store. - #[must_use] - pub fn with_blob_artifact_store(&self, store: BlobArtifactStore) -> Self { - let mut handle = self.clone(); - handle.client = handle.client.with_blob_artifact_store(store); - handle - } - - /// Returns the immutable compiled artifact. - #[must_use] - pub fn compiled(&self) -> &CompiledApplication { - &self.compiled - } - - /// Derives a scoped target using the declared entity or fixed-shard topology. - pub fn target_for_scope(&self, namespace: NamespaceId, scope: &[u8]) -> Result { - let cell_type = self - .compiled - .cell_types - .iter() - .find(|cell_type| cell_type.namespace == namespace) - .ok_or(Error::Registry("namespace is not declared by application"))?; - if cell_type.partition_version == ENTITY_PARTITION_VERSION { - let partition = cell_type.entity_partition(scope)?; - return CellTarget::new(self.tenant, self.application, namespace, &partition); - } - let partition = cell_type.partition_for_scope(scope)?; - CellTarget::new(self.tenant, self.application, namespace, &partition) - } - - /// Executes one statically typed command after enforcing application scope. - pub async fn command( - &self, - target: &CellTarget, - identity: crab_cell_runtime::MutationIdentity, - input: C::Input, - ) -> std::result::Result, InvocationError> { - if let Err(error) = self.validate_target_module(target, C::MODULE) { - return Err(InvocationError::NotStarted(error)); - } - self.client.command::(target, identity, input).await - } - - /// Prepares a scoped typed command whose evidence can be resolved after cancellation. - /// - /// Rejects a target outside this application before preparing the mutation. - pub async fn prepare_command( - &self, - target: &CellTarget, - identity: crab_cell_runtime::MutationIdentity, - input: C::Input, - ) -> std::result::Result, InvocationError> { - self.validate_target_module(target, C::MODULE) - .map_err(InvocationError::NotStarted)?; - self.client - .prepare_command::(target, identity, input) - .await - } - - /// Executes one statically typed query after enforcing application scope. - pub async fn query( - &self, - target: &CellTarget, - minimum: Option, - input: Q::Input, - ) -> std::result::Result, InvocationError> { - if let Err(error) = self.validate_target_module(target, Q::MODULE) { - return Err(InvocationError::NotStarted(error)); - } - self.client.query::(target, minimum, input).await - } - - /// Resolves a pending command against the current owner after checking its application scope. - /// - /// The caller must keep the pending mutation from its original typed invocation; - /// resolution can remain unknown until the owner recovers or the identity expires. - pub async fn resolve( - &self, - pending: &PendingMutation, - ) -> std::result::Result>> { - self.validate_target(pending.target()) - .map_err(InvocationError::NotStarted)?; - self.client.resolve(pending).await - } - - /// Returns the typed KV capability for a compiled fixed-shard KV module. - pub fn kv(&self, namespace: NamespaceId) -> Result> { - self.validate_sharded_namespace(namespace, M::MODULE, CatalogRole::Kv)?; - KvNamespace::new( - self.client.clone(), - self.tenant, - self.application, - namespace, - ) - } - - /// Returns the typed SQL capability for one explicitly selected Cell. - pub fn sql(&self, target: CellTarget) -> Result> { - self.validate_target(&target)?; - self.validate_namespace(target.namespace(), M::MODULE, CatalogRole::Sql)?; - SqlCell::new(self.client.clone(), target) - } - - /// Returns the typed Blob capability for a compiled fixed-shard Blob module. - pub fn blob(&self) -> Result> { - self.validate_sharded_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Blob)?; - BlobNamespace::new(self.client.clone(), self.tenant, self.application) - } - - /// Returns the typed Queue capability for a compiled fixed-shard Queue module. - pub fn queue(&self) -> Result> { - self.validate_sharded_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Queue)?; - QueueNamespace::new(self.client.clone(), self.tenant, self.application) - } - - /// Returns the typed Cron capability for a compiled fixed-shard Cron module. - pub fn cron(&self) -> Result> { - self.validate_sharded_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Cron)?; - CronNamespace::new(self.client.clone(), self.tenant, self.application) - } - - /// Returns the typed Workflow capability for a compiled fixed-shard Workflow module. - pub fn workflow(&self) -> Result> { - self.validate_sharded_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Workflow)?; - WorkflowNamespace::new(self.client.clone(), self.tenant, self.application) - } - - /// Returns the native activity capability for one compiled fixed-shard Workflow module. - pub fn activities(&self) -> Result> { - self.validate_sharded_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Workflow)?; - if !self.compiled.registry.has_activity_runner(M::NAMESPACE) { - return Err(Error::Registry("activity runner is not registered")); - } - WorkflowActivities::new(self.client.clone(), self.tenant, self.application) - } - - /// Returns the source effect capability for one explicitly selected Cell. - /// - /// Effects are emitted by commands through [`crab_cell_runtime::registry::CommandContext::emit_effect`]; - /// this capability is for claiming and acknowledging the resulting source - /// ledger. The destination still owns external idempotency. - pub fn effects(&self, target: CellTarget) -> Result> { - self.validate_target(&target)?; - self.compiled - .validate_module(target.namespace(), M::MODULE)?; - if !self.compiled.registry.has_effect_runner(target.namespace()) { - return Err(Error::Registry("effect runner is not registered")); - } - self.client.effect_source::(target) - } - - fn validate_target(&self, target: &CellTarget) -> Result<()> { - if target.tenant() != self.tenant || target.application() != self.application { - return Err(Error::Identity("Cell target is outside application scope")); - } - let Some(cell_type) = self - .compiled - .cell_types - .iter() - .find(|cell_type| cell_type.namespace == target.namespace()) - else { - return Err(Error::Registry("namespace is not declared by application")); - }; - if !cell_type.valid_partition(target.partition()) { - return Err(Error::Identity( - "Cell target partition is outside the declared topology", - )); - } - Ok(()) - } - - fn validate_target_module(&self, target: &CellTarget, module: &'static str) -> Result<()> { - self.validate_target(target)?; - self.compiled.validate_module(target.namespace(), module) - } - - fn validate_sharded_namespace( - &self, - namespace: NamespaceId, - module: &'static str, - role: CatalogRole, - ) -> Result<()> { - let cell_type = self.validate_namespace(namespace, module, role)?; - // Namespace helpers derive four-byte shard targets internally. Accepting - // an entity declaration here would bypass the application's topology. - if cell_type.partition_version == ENTITY_PARTITION_VERSION { - return Err(Error::Identity( - "namespace capability requires fixed shards", - )); - } - Ok(()) - } - - fn validate_namespace( - &self, - namespace: NamespaceId, - module: &'static str, - role: CatalogRole, - ) -> Result<&CellType> { - let Some(cell_type) = self - .compiled - .cell_types - .iter() - .find(|cell_type| cell_type.namespace == namespace) - else { - return Err(Error::Registry("namespace is not declared by application")); - }; - if cell_type.role != role { - return Err(Error::Registry("namespace role differs from capability")); - } - self.compiled.validate_module(namespace, module)?; - let Some((_, contract)) = self.compiled.registry.namespace_contract(namespace) else { - return Err(Error::Registry("namespace contract is missing")); - }; - if contract.role != role { - return Err(Error::Registry("namespace module differs from capability")); - } - Ok(cell_type) - } -} - -fn encode_descriptor(name: &str, registry: &Registry, cell_types: &[CellType]) -> Result> { - let mut bytes = Vec::with_capacity(128); - bytes.extend_from_slice(DESCRIPTOR_MAGIC); - push_text(&mut bytes, name)?; - bytes.extend_from_slice(registry.release_digest().as_bytes()); - bytes.extend_from_slice(&(cell_types.len() as u16).to_be_bytes()); - for cell_type in cell_types { - push_text(&mut bytes, cell_type.module)?; - push_text(&mut bytes, cell_type.name)?; - bytes.extend_from_slice(cell_type.namespace.as_bytes()); - bytes.push(role_code(cell_type.role)); - bytes.extend_from_slice(&cell_type.shards.to_be_bytes()); - bytes.extend_from_slice(&cell_type.partition_version.to_be_bytes()); - bytes.extend_from_slice(&cell_type.schema_min.to_be_bytes()); - bytes.extend_from_slice(&cell_type.schema_max.to_be_bytes()); - bytes.extend_from_slice(&cell_type.database_limit_bytes.to_be_bytes()); - bytes.extend_from_slice(&cell_type.capture_limit_bytes.to_be_bytes()); - } - if bytes.len() > MAX_DESCRIPTOR_BYTES { - return Err(Error::Registry("application descriptor exceeds 256 KiB")); - } - Ok(bytes) -} - -fn push_text(bytes: &mut Vec, value: &str) -> Result<()> { - if value.len() > MAX_APPLICATION_NAME_BYTES { - return Err(Error::Registry( - "application descriptor text exceeds 128 bytes", - )); - } - let length = u16::try_from(value.len()) - .map_err(|_| Error::Registry("application descriptor text length overflow"))?; - bytes.extend_from_slice(&length.to_be_bytes()); - bytes.extend_from_slice(value.as_bytes()); - Ok(()) -} - -fn valid_name(value: &str) -> bool { - !value.is_empty() - && value.len() <= MAX_APPLICATION_NAME_BYTES - && value - .bytes() - .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_' | b'.')) -} - -fn role_code(role: CatalogRole) -> u8 { - match role { - CatalogRole::Repository => 0, - CatalogRole::Sql => 1, - CatalogRole::Kv => 2, - CatalogRole::Queue => 3, - CatalogRole::Workflow => 4, - CatalogRole::Blob => 5, - CatalogRole::Cron => 6, - } -} - -mod client_macro; - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-app/src/tests.rs b/crates/crab-cell-app/src/tests.rs deleted file mode 100644 index cddc346a6..000000000 --- a/crates/crab-cell-app/src/tests.rs +++ /dev/null @@ -1,380 +0,0 @@ -use super::*; -use crab_cell_runtime::identity::Digest; -use crab_cell_runtime::registry::{CellModule, ModuleDescriptor, NamespaceDescriptor}; - -struct SqlModule; - -impl CellModule for SqlModule { - const NAME: &'static str = "app-sql"; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: ModuleDescriptor = ModuleDescriptor { - name: "app-sql", - source_digest: Digest::from_bytes([1; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: &[crab_cell_runtime::MigrationDescriptor { - version: 1, - sql: "-- app-sql migration v1", - digest: Digest::from_bytes([ - 0x43, 0x1d, 0x99, 0x74, 0x5c, 0x8f, 0x39, 0x00, 0x9d, 0x2e, 0xe2, 0x00, 0x7c, - 0x75, 0x42, 0xe3, 0xa5, 0xf0, 0x0c, 0x92, 0x21, 0x8b, 0xf7, 0x3b, 0x8f, 0x6f, - 0x6b, 0xc3, 0x38, 0xe4, 0xf5, 0xca, - ]), - }], - commands: &[], - queries: &[], - workflow_definitions: &[], - activity_types: &[], - namespaces: &[ - NamespaceDescriptor { - id: NamespaceId::from_bytes([2; 16]), - name: "app-sql", - role: CatalogRole::Sql, - shards: 1, - effect_targets: &[], - dead_letter: None, - }, - NamespaceDescriptor { - id: NamespaceId::from_bytes([3; 16]), - name: "app-sql-2", - role: CatalogRole::Sql, - shards: 1, - effect_targets: &[], - dead_letter: None, - }, - ], - }; - &DESCRIPTOR - } - - fn register(self, _registry: &mut RegistryBuilder) -> Result<()> { - Ok(()) - } -} - -fn build(reverse: bool) -> CompiledApplication { - let mut builder = ApplicationBuilder::new( - "app", - BuildDescriptor { - source_revision: "source".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }, - ) - .unwrap(); - builder.register(SqlModule).unwrap(); - let first = CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(); - let second = CellType::new( - "app-sql", - "inventory", - NamespaceId::from_bytes([3; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(); - if reverse { - builder.cell_type(second).unwrap(); - builder.cell_type(first).unwrap(); - } else { - builder.cell_type(first).unwrap(); - builder.cell_type(second).unwrap(); - } - builder.finish().unwrap() -} - -#[test] -fn descriptor_is_stable_and_scope_is_checked() { - let first = build(false); - let second = build(true); - assert_eq!(first.descriptor_bytes(), second.descriptor_bytes()); - assert_eq!(first.descriptor_digest(), second.descriptor_digest()); - assert_eq!( - first.registry().release_digest(), - first.registry().release_digest() - ); -} - -#[test] -fn compiled_application_rejects_cross_module_invocation_targets() { - let application = build(false); - let sql_namespace = NamespaceId::from_bytes([2; 16]); - - assert!( - application - .validate_module(sql_namespace, "app-sql") - .is_ok() - ); - assert!( - application - .validate_module(sql_namespace, "unrelated-module") - .is_err() - ); - assert!( - application - .validate_module(NamespaceId::from_bytes([99; 16]), "app-sql") - .is_err() - ); -} - -#[test] -fn duplicate_cell_namespace_and_mismatched_module_fail_closed() { - let mut builder = ApplicationBuilder::new( - "app", - BuildDescriptor { - source_revision: "source".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }, - ) - .unwrap(); - let cell_type = CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(); - builder.cell_type(cell_type).unwrap(); - assert!(builder.cell_type(cell_type).is_err()); - builder.register(SqlModule).unwrap(); - let mut wrong = ApplicationBuilder::new( - "app", - BuildDescriptor { - source_revision: "source".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }, - ) - .unwrap(); - wrong - .cell_type( - CellType::new( - "other", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(), - ) - .unwrap(); - wrong.register(SqlModule).unwrap(); - assert!(wrong.finish().is_err()); -} - -#[test] -fn every_registered_namespace_requires_a_cell_type() { - let mut builder = ApplicationBuilder::new( - "app", - BuildDescriptor { - source_revision: "source".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }, - ) - .unwrap(); - builder.register(SqlModule).unwrap(); - builder - .cell_type( - CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(), - ) - .unwrap(); - - assert!(builder.finish().is_err()); -} - -#[test] -fn cell_type_schema_range_must_match_compiled_module() { - let mut builder = ApplicationBuilder::new( - "app", - BuildDescriptor { - source_revision: "source".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }, - ) - .unwrap(); - builder.register(SqlModule).unwrap(); - builder - .cell_type( - CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap() - .with_schema_range(1, 2) - .unwrap(), - ) - .unwrap(); - - assert!(builder.finish().is_err()); -} - -#[test] -fn duplicate_cell_type_name_fails_closed() { - let mut builder = ApplicationBuilder::new( - "app", - BuildDescriptor { - source_revision: "source".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }, - ) - .unwrap(); - let first = CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(); - let second = CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([3; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(); - builder.cell_type(first).unwrap(); - assert!(builder.cell_type(second).is_err()); -} - -#[test] -fn cell_type_limits_and_partition_bounds_fail_closed() { - for shards in [0, 3, 8_192] { - assert!( - CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - shards, - ) - .is_err() - ); - } - assert!( - CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([0; 16]), - CatalogRole::Sql, - 1, - ) - .is_err() - ); - - let cell_type = CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(); - assert!(cell_type.with_limits(0, 1).is_err()); - assert!(cell_type.with_limits(1, 0).is_err()); - assert!(cell_type.with_limits(511, 128).is_err()); - assert!(cell_type.with_limits(512, 127).is_err()); - assert!(cell_type.with_schema_range(0, 1).is_err()); - assert!(cell_type.with_schema_range(2, 1).is_err()); -} - -#[test] -fn entity_partitions_are_stable_and_distinct_from_fixed_shards() { - let fixed = CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(); - assert!(fixed.entity_partition(b"order-1").is_err()); - assert!(fixed.with_entity_partitions().is_ok()); - assert!( - CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 2, - ) - .unwrap() - .with_entity_partitions() - .is_err() - ); - - let entity = fixed.with_entity_partitions().unwrap(); - let first = entity.entity_partition(b"order-1").unwrap(); - assert_eq!(first, entity.entity_partition(b"order-1").unwrap()); - assert_ne!(first, entity.entity_partition(b"order-2").unwrap()); - assert!(entity.valid_partition(&first)); - assert!(!fixed.valid_partition(&first)); - assert!(!entity.valid_partition(&partition_for_shard(0))); - let mut invalid_prefix = first; - invalid_prefix[0] = 0; - assert!(!entity.valid_partition(&invalid_prefix)); - assert!(!entity.valid_partition(&first[..32])); - assert!(entity.shard_for_scope(b"order-1").is_err()); - assert!(entity.entity_partition(b"").is_err()); - assert!(entity.entity_partition(&[0; 1_025]).is_err()); -} - -#[test] -fn entity_topology_changes_application_descriptor() { - let fixed = build(false); - let mut builder = ApplicationBuilder::new( - "app", - BuildDescriptor { - source_revision: "source".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }, - ) - .unwrap(); - builder.register(SqlModule).unwrap(); - builder - .cell_type( - CellType::new( - "app-sql", - "orders", - NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap() - .with_entity_partitions() - .unwrap(), - ) - .unwrap(); - builder - .cell_type( - CellType::new( - "app-sql", - "inventory", - NamespaceId::from_bytes([3; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(), - ) - .unwrap(); - let entity = builder.finish().unwrap(); - assert_ne!(fixed.descriptor_digest(), entity.descriptor_digest()); -} diff --git a/crates/crab-cell-app/tests-allow-list.txt b/crates/crab-cell-app/tests-allow-list.txt deleted file mode 100644 index 273a362d7..000000000 --- a/crates/crab-cell-app/tests-allow-list.txt +++ /dev/null @@ -1,2 +0,0 @@ -tests.rs # drives the private registry validation path through ApplicationBuilder -lib.rs # declares crate-private test modules diff --git a/crates/crab-cell-app/tests/contracts.rs b/crates/crab-cell-app/tests/contracts.rs deleted file mode 100644 index 23ff57514..000000000 --- a/crates/crab-cell-app/tests/contracts.rs +++ /dev/null @@ -1,362 +0,0 @@ -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::identity::{Digest, NamespaceId}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, ModuleDescriptor, NamespaceDescriptor, Registry, RegistryBuilder, -}; -use crab_cell_runtime::registry::{MigrationDescriptor, OperationDescriptor}; - -const MIGRATION_SQL: &str = "-- contract migration v1"; -const WORKFLOW_DEFINITION: Digest = Digest::from_bytes([41; 32]); - -fn build() -> BuildDescriptor { - BuildDescriptor { - source_revision: "contract-validation".into(), - cargo_lock_digest: Digest::from_bytes([42; 32]), - } -} - -fn migration(version: u32, sql: &'static str, digest: Digest) -> &'static [MigrationDescriptor] { - Box::leak(Box::new([MigrationDescriptor { - version, - sql, - digest, - }])) -} - -fn namespace( - namespace: u8, - role: CatalogRole, - shards: u32, - effect_targets: &'static [NamespaceId], - dead_letter: Option, -) -> &'static [NamespaceDescriptor] { - Box::leak(Box::new([NamespaceDescriptor { - id: NamespaceId::from_bytes([namespace; 16]), - name: "contract-namespace", - role, - shards, - effect_targets, - dead_letter, - }])) -} - -fn one_namespace_id(namespace: u8) -> &'static [NamespaceId] { - Box::leak(Box::new([NamespaceId::from_bytes([namespace; 16])])) -} - -fn descriptor( - name: &'static str, - namespace: &'static [NamespaceDescriptor], - migrations: &'static [MigrationDescriptor], - commands: &'static [OperationDescriptor], - workflow_definitions: &'static [Digest], - activity_types: &'static [&'static str], -) -> &'static ModuleDescriptor { - Box::leak(Box::new(ModuleDescriptor { - name, - source_digest: Digest::from_bytes([7; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations, - commands, - queries: &[], - workflow_definitions, - activity_types, - namespaces: namespace, - })) -} - -macro_rules! module { - ($type:ident, $name:literal, $descriptor:ident) => { - struct $type; - - impl CellModule for $type { - const NAME: &'static str = $name; - - fn descriptor(&self) -> &'static ModuleDescriptor { - $descriptor() - } - - fn register(self, _registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - Ok(()) - } - } - }; -} - -fn valid_descriptor() -> &'static ModuleDescriptor { - descriptor( - "valid", - namespace(1, CatalogRole::Repository, 1, &[], None), - migration( - 1, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[], - &[], - &[], - ) -} - -module!(ValidModule, "valid", valid_descriptor); - -fn finishes_with_error(module: M) -> bool { - let mut builder = RegistryBuilder::new(build()); - builder.register(module).is_ok() && builder.finish().is_err() -} - -fn invalid_migration_digest_descriptor() -> &'static ModuleDescriptor { - descriptor( - "invalid-migration-digest", - namespace(2, CatalogRole::Repository, 1, &[], None), - migration(1, MIGRATION_SQL, Digest::from_bytes([8; 32])), - &[], - &[], - &[], - ) -} - -module!( - InvalidMigrationDigest, - "invalid-migration-digest", - invalid_migration_digest_descriptor -); - -fn invalid_migration_order_descriptor() -> &'static ModuleDescriptor { - descriptor( - "invalid-migration-order", - namespace(3, CatalogRole::Repository, 1, &[], None), - migration( - 2, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[], - &[], - &[], - ) -} - -module!( - InvalidMigrationOrder, - "invalid-migration-order", - invalid_migration_order_descriptor -); - -fn invalid_effect_target_descriptor() -> &'static ModuleDescriptor { - descriptor( - "invalid-effect-target", - namespace(4, CatalogRole::Repository, 1, one_namespace_id(99), None), - migration( - 1, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[], - &[], - &[], - ) -} - -module!( - InvalidEffectTarget, - "invalid-effect-target", - invalid_effect_target_descriptor -); - -fn invalid_dead_letter_descriptor() -> &'static ModuleDescriptor { - descriptor( - "invalid-dead-letter", - namespace( - 5, - CatalogRole::Repository, - 1, - &[], - Some(NamespaceId::from_bytes([5; 16])), - ), - migration( - 1, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[], - &[], - &[], - ) -} - -module!( - InvalidDeadLetter, - "invalid-dead-letter", - invalid_dead_letter_descriptor -); - -fn missing_workflow_binding_descriptor() -> &'static ModuleDescriptor { - descriptor( - "missing-workflow-binding", - namespace(6, CatalogRole::Workflow, 1, &[], None), - migration( - 1, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[], - &[WORKFLOW_DEFINITION], - &[], - ) -} - -module!( - MissingWorkflowBinding, - "missing-workflow-binding", - missing_workflow_binding_descriptor -); - -fn activity_without_workflow_descriptor() -> &'static ModuleDescriptor { - descriptor( - "activity-without-workflow", - namespace(7, CatalogRole::Workflow, 1, &[], None), - migration( - 1, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[], - &[], - &["contract-activity"], - ) -} - -module!( - ActivityWithoutWorkflow, - "activity-without-workflow", - activity_without_workflow_descriptor -); - -fn invalid_shards_descriptor() -> &'static ModuleDescriptor { - descriptor( - "invalid-shards", - namespace(8, CatalogRole::Repository, 3, &[], None), - migration( - 1, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[], - &[], - &[], - ) -} - -module!(InvalidShards, "invalid-shards", invalid_shards_descriptor); - -fn invalid_operation_limits_descriptor() -> &'static ModuleDescriptor { - const COMMAND: OperationDescriptor = OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 0, - output_limit: 1, - }; - descriptor( - "invalid-operation-limits", - namespace(9, CatalogRole::Repository, 1, &[], None), - migration( - 1, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[COMMAND], - &[], - &[], - ) -} - -module!( - InvalidOperationLimits, - "invalid-operation-limits", - invalid_operation_limits_descriptor -); - -fn duplicate_a_descriptor() -> &'static ModuleDescriptor { - descriptor( - "duplicate-module", - namespace(10, CatalogRole::Repository, 1, &[], None), - migration( - 1, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[], - &[], - &[], - ) -} - -fn duplicate_b_descriptor() -> &'static ModuleDescriptor { - descriptor( - "duplicate-module", - namespace(11, CatalogRole::Repository, 1, &[], None), - migration( - 1, - MIGRATION_SQL, - Digest::from_bytes(*blake3::hash(MIGRATION_SQL.as_bytes()).as_bytes()), - ), - &[], - &[], - &[], - ) -} - -module!(DuplicateModuleA, "duplicate-module", duplicate_a_descriptor); -module!(DuplicateModuleB, "duplicate-module", duplicate_b_descriptor); - -type ContractCase = (&'static str, fn() -> bool); - -#[test] -fn descriptor_contract_rejects_each_declared_relationship_and_limit() { - let cases: [ContractCase; 8] = [ - ("migration digest", || { - finishes_with_error(InvalidMigrationDigest) - }), - ("migration ordering", || { - finishes_with_error(InvalidMigrationOrder) - }), - ("effect target", || finishes_with_error(InvalidEffectTarget)), - ("dead-letter target", || { - finishes_with_error(InvalidDeadLetter) - }), - ("workflow binding", || { - finishes_with_error(MissingWorkflowBinding) - }), - ("activity inventory", || { - finishes_with_error(ActivityWithoutWorkflow) - }), - ("namespace shards", || finishes_with_error(InvalidShards)), - ("operation limits", || { - finishes_with_error(InvalidOperationLimits) - }), - ]; - for (name, rejects) in cases { - assert!(rejects(), "invalid {name} descriptor was accepted"); - } -} - -#[test] -fn descriptor_contract_rejects_duplicate_module_names() { - let mut builder = RegistryBuilder::new(build()); - builder.register(DuplicateModuleA).unwrap(); - builder.register(DuplicateModuleB).unwrap(); - assert!(builder.finish().is_err()); -} - -#[test] -fn valid_descriptor_still_compiles_after_contract_fixtures() { - let mut builder = RegistryBuilder::new(build()); - builder.register(ValidModule).unwrap(); - let registry: Registry = builder.finish().unwrap(); - assert_eq!(registry.namespace_count(), 1); -} diff --git a/crates/crab-cell-app/tests/reference_application.rs b/crates/crab-cell-app/tests/reference_application.rs deleted file mode 100644 index 33cc05c0c..000000000 --- a/crates/crab-cell-app/tests/reference_application.rs +++ /dev/null @@ -1,179 +0,0 @@ -use std::{future::Future, pin::Pin, sync::Arc, sync::OnceLock, time::UNIX_EPOCH}; - -use crab_cell_app::{ApplicationBuilder, ApplicationHandle, CellApplication, CellType}; -use crab_cell_runtime::cell::actor::CellHandle; -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogEntry; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::executor::MutationIdentity; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::client::{CellClient, InvocationError}; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, TenantId, partition_for_shard, -}; -use crab_cell_runtime::identity::{IncarnationId, NodeId, RequestId}; -use crab_cell_runtime::ltx::CellStorageLayout; -use crab_cell_runtime::node::{ - FencedNodeSession, NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain, -}; -use crab_cell_runtime::primitives::blob::{BlobArtifactStore, BlobModule}; -use crab_cell_runtime::primitives::blob::{ - BlobCondition, BlobMutation, BlobMutationOutcome, BlobQuery, BlobQueryResult, - install_blob_schema, register_blob, -}; -use crab_cell_runtime::primitives::cron::CronModule; -use crab_cell_runtime::primitives::cron::{ - CronInvocation, CronMutation, CronQueryResult, CronTarget, install_cron_schema, register_cron, -}; -use crab_cell_runtime::primitives::effects::EffectModule; -use crab_cell_runtime::primitives::effects::{ - EffectClaimRequest, EffectLeaseOutcome, register_effect_delivery, -}; -use crab_cell_runtime::primitives::kv::KvModule; -use crab_cell_runtime::primitives::kv::{ - KvAtomicCommand, KvAtomicRequest, KvGetQuery, KvGetRequest, KvMutation, install_kv_schema, - register_kv, -}; -use crab_cell_runtime::primitives::maintenance::{MaintenanceModule, register_maintenance}; -use crab_cell_runtime::primitives::queue::QueueModule; -use crab_cell_runtime::primitives::queue::{ - QueueClaimRequest, QueueDeadLetterTarget, QueueLeaseOutcome, QueueSendRequest, - install_queue_schema, register_queue, -}; -use crab_cell_runtime::primitives::sql::SqlModule; -use crab_cell_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue, register_sql}; -use crab_cell_runtime::primitives::workflow::{ - ActivityContext, ActivityExecution, ActivityHandler, ActivityRunOutcome, WorkflowAction, - WorkflowContext, WorkflowDecision, WorkflowDefinition, WorkflowStatus, install_workflow_schema, - register_activity, register_workflow, register_workflow_activities, -}; -use crab_cell_runtime::primitives::workflow::{WorkflowActivityModule, WorkflowModule}; -use crab_cell_runtime::qualification::{ - QualificationExecution, QualificationOperation, QualificationOperationExecutor, - QualificationProfile, QualificationWorkload, -}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, Command, ModuleDescriptor, NamespaceDescriptor, Registry, - RegistryBuilder, -}; -use crab_cell_runtime::registry::{CommandContext, CommandResult, OperationDescriptor}; -use crab_cell_runtime::{Error, Result}; -use crab_ltx::{CellReplica, DiskBudget, Host, Limits}; -use crab_storage::Store; -use ed25519_dalek::SigningKey; -use object_store::memory::InMemory; - -mod reference_application { - pub mod application; - pub mod commit; - pub mod entities; - pub mod fleet; - pub mod harness; - pub mod performance; - pub mod performance_fixture; - pub mod primitives; - pub mod process_node; - pub mod process_performance; - pub mod process_recruitment; - pub mod process_replica; - pub mod process_scaling; - pub mod public_host; -} - -pub(crate) use reference_application::application::*; -pub(crate) use reference_application::harness::*; - -#[test] -fn application_descriptor_is_stable_when_modules_register_in_reverse_order() { - let forward = compile_reference_in_order(false); - let reverse = compile_reference_in_order(true); - assert_eq!(forward.descriptor_bytes(), reverse.descriptor_bytes()); - assert_eq!(forward.descriptor_digest(), reverse.descriptor_digest()); -} - -#[test] -fn generated_client_release_descriptor_matches_independent_stable_ids() { - fn push_name(bytes: &mut Vec, value: &str) { - bytes.extend_from_slice(&(value.len() as u16).to_be_bytes()); - bytes.extend_from_slice(value.as_bytes()); - } - - let application = compiled(); - let mut registry = RegistryBuilder::new(BuildDescriptor { - source_revision: "reference-source".into(), - cargo_lock_digest: Digest::from_bytes([42; 32]), - }); - registry.register(ReferenceSql).unwrap(); - registry.register(ReferenceKv).unwrap(); - registry.register(ReferenceBlob).unwrap(); - registry.register(ReferenceQueue).unwrap(); - registry.register(ReferenceDeadLetter).unwrap(); - registry.register(ReferenceCron).unwrap(); - registry.register(ReferenceWorkflow).unwrap(); - let registry = registry.finish().unwrap(); - - let mut expected = b"crab.application.v1\0".to_vec(); - push_name(&mut expected, "reference-application"); - expected.extend_from_slice(registry.release_digest().as_bytes()); - expected.extend_from_slice(&7_u16.to_be_bytes()); - for (module, cell_name, namespace, role) in [ - (SQL_MODULE, "sql", SQL_NAMESPACE, 1_u8), - (KV_MODULE, "kv", KV_NAMESPACE, 2), - (BLOB_MODULE, "blob", BLOB_NAMESPACE, 5), - (QUEUE_MODULE, "queue", QUEUE_NAMESPACE, 3), - (DEAD_LETTER_MODULE, "dead-letter", DEAD_LETTER_NAMESPACE, 3), - (CRON_MODULE, "cron", CRON_NAMESPACE, 6), - (WORKFLOW_MODULE, "workflow", WORKFLOW_NAMESPACE, 4), - ] { - push_name(&mut expected, module); - push_name(&mut expected, cell_name); - expected.extend_from_slice(namespace.as_bytes()); - expected.push(role); - expected.extend_from_slice(&1_u32.to_be_bytes()); - expected.extend_from_slice(&1_u32.to_be_bytes()); - expected.extend_from_slice(&1_u32.to_be_bytes()); - expected.extend_from_slice(&1_u32.to_be_bytes()); - expected.extend_from_slice(&(64_u64 * 1024 * 1024).to_be_bytes()); - expected.extend_from_slice(&(16_u64 * 1024 * 1024).to_be_bytes()); - } - assert_eq!(application.descriptor_bytes(), expected); -} - -#[test] -fn reference_application_registers_every_primitive_and_relationship() { - let application = compiled(); - assert_eq!(application.cell_types().len(), 7); - assert!(application.registry().has_effect_runner(SQL_NAMESPACE)); - assert_eq!( - application - .registry() - .namespace_contract(QUEUE_NAMESPACE) - .unwrap() - .1 - .dead_letter, - Some(DEAD_LETTER_NAMESPACE) - ); - assert_eq!( - application - .registry() - .namespace_contract(WORKFLOW_NAMESPACE) - .unwrap() - .1 - .effect_targets, - &[SQL_NAMESPACE] - ); - assert!(!application.descriptor_bytes().is_empty()); -} - -#[test] -fn descriptor_digest_changes_when_build_identity_changes() { - let first = compiled(); - let second = ReferenceApplication::compile(BuildDescriptor { - source_revision: "different-source".into(), - cargo_lock_digest: Digest::from_bytes([42; 32]), - }) - .unwrap(); - assert_ne!(first.descriptor_digest(), second.descriptor_digest()); -} diff --git a/crates/crab-cell-app/tests/reference_application/application.rs b/crates/crab-cell-app/tests/reference_application/application.rs deleted file mode 100644 index 56005de4c..000000000 --- a/crates/crab-cell-app/tests/reference_application/application.rs +++ /dev/null @@ -1,659 +0,0 @@ -//! The reference application every suite module compiles and drives. - -use crate::*; - -pub(crate) const SQL_NAMESPACE: NamespaceId = NamespaceId::from_bytes([1; 16]); -pub(crate) const KV_NAMESPACE: NamespaceId = NamespaceId::from_bytes([2; 16]); -pub(crate) const BLOB_NAMESPACE: NamespaceId = NamespaceId::from_bytes([3; 16]); -pub(crate) const QUEUE_NAMESPACE: NamespaceId = NamespaceId::from_bytes([4; 16]); -pub(crate) const DEAD_LETTER_NAMESPACE: NamespaceId = NamespaceId::from_bytes([5; 16]); -pub(crate) const CRON_NAMESPACE: NamespaceId = NamespaceId::from_bytes([6; 16]); -pub(crate) const WORKFLOW_NAMESPACE: NamespaceId = NamespaceId::from_bytes([7; 16]); -pub(crate) const WORKFLOW_DIGEST: Digest = Digest::from_bytes([8; 32]); - -pub(crate) const SQL_MODULE: &str = "reference-sql"; -pub(crate) const KV_MODULE: &str = "reference-kv"; -pub(crate) const BLOB_MODULE: &str = "reference-blob"; -pub(crate) const QUEUE_MODULE: &str = "reference-queue"; -pub(crate) const DEAD_LETTER_MODULE: &str = "reference-dead-letter"; -pub(crate) const CRON_MODULE: &str = "reference-cron"; -pub(crate) const WORKFLOW_MODULE: &str = "reference-workflow"; - -pub(crate) fn operation(id: u32) -> OperationDescriptor { - OperationDescriptor { - id, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 1 << 20, - output_limit: 1 << 20, - } -} - -pub(crate) fn migration() -> &'static [crab_cell_runtime::registry::MigrationDescriptor] { - static MIGRATION: OnceLock<&'static [crab_cell_runtime::registry::MigrationDescriptor]> = - OnceLock::new(); - MIGRATION.get_or_init(|| { - Box::leak(Box::new([ - crab_cell_runtime::registry::MigrationDescriptor { - version: 1, - sql: "-- reference application migration v1", - digest: Digest::from_bytes( - *blake3::hash(b"-- reference application migration v1").as_bytes(), - ), - }, - ])) - }) -} - -pub(crate) fn descriptor( - module: &'static str, - commands: &'static [u32], - queries: &'static [u32], - namespaces: &'static [NamespaceDescriptor], - workflows: &'static [Digest], - activities: &'static [&'static str], -) -> &'static ModuleDescriptor { - static DESCRIPTORS: OnceLock>> = - OnceLock::new(); - let descriptors = DESCRIPTORS.get_or_init(|| std::sync::Mutex::new(Vec::new())); - let mut descriptors = descriptors.lock().expect("reference descriptor lock"); - if let Some(descriptor) = descriptors - .iter() - .copied() - .find(|descriptor| descriptor.name == module) - { - return descriptor; - } - let command_descriptors = Box::leak( - commands - .iter() - .copied() - .map(|id| { - let mut descriptor = operation(id); - if module == KV_MODULE { - descriptor.input_limit = 4 * 1024 * 1024 + 64 * 1024; - } - descriptor - }) - .collect::>() - .into_boxed_slice(), - ); - let query_descriptors = Box::leak( - queries - .iter() - .copied() - .map(|id| { - let mut descriptor = operation(id); - if module == KV_MODULE { - descriptor.output_limit = 4 * 1024 * 1024 + 64 * 1024; - } - descriptor - }) - .collect::>() - .into_boxed_slice(), - ); - let descriptor = Box::leak(Box::new(ModuleDescriptor { - name: module, - source_digest: Digest::from_bytes(*blake3::hash(module.as_bytes()).as_bytes()), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: migration(), - commands: command_descriptors, - queries: query_descriptors, - workflow_definitions: workflows, - activity_types: activities, - namespaces, - })); - descriptors.push(descriptor); - descriptor -} - -pub(crate) struct ReferenceSql; -impl SqlModule for ReferenceSql { - const MODULE: &'static str = SQL_MODULE; - const BATCH_COMMAND_ID: u32 = 1; - const BATCH_QUERY_ID: u32 = 2; -} -impl EffectModule for ReferenceSql { - const MODULE: &'static str = SQL_MODULE; - const CLAIM_COMMAND_ID: u32 = 3; - const LEASE_COMMAND_ID: u32 = 4; - const VALIDATE_QUERY_ID: u32 = 5; - const STATUS_QUERY_ID: u32 = 6; -} -impl CellModule for ReferenceSql { - const NAME: &'static str = SQL_MODULE; - fn descriptor(&self) -> &'static ModuleDescriptor { - static NAMESPACES: &[NamespaceDescriptor] = &[NamespaceDescriptor { - id: SQL_NAMESPACE, - name: "reference-sql", - role: CatalogRole::Sql, - shards: 1, - effect_targets: &[], - dead_letter: None, - }]; - descriptor( - SQL_MODULE, - &[1, 3, 4, 6], - &[2, 5, 6, 7], - NAMESPACES, - &[], - &[], - ) - } - fn register(self, registry: &mut RegistryBuilder) -> Result<()> { - register_sql::(registry)?; - registry.bind_command::()?; - registry.bind_query::()?; - register_effect_delivery::(registry) - } -} - -pub(crate) struct ReferenceReceiptCount; - -impl crab_cell_runtime::registry::Query for ReferenceReceiptCount { - const MODULE: &'static str = SQL_MODULE; - const ID: u32 = 7; - const CODEC_VERSION: u32 = 1; - type Input = (); - type Output = u64; - - fn execute( - context: &mut crab_cell_runtime::registry::QueryContext<'_>, - _input: Self::Input, - ) -> Result { - let results = context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT COUNT(*) FROM invoice_receipts".into(), - parameters: Vec::new(), - }], - })?; - match results[0].rows.first().and_then(|row| row.first()) { - Some(SqlValue::Integer(count)) => { - u64::try_from(*count).map_err(|_| Error::Command("negative invoice receipt count")) - } - _ => Err(Error::Command("invoice receipt count is unavailable")), - } - } -} - -pub(crate) struct ReferenceCronReceiver; - -pub(crate) struct OrderId(pub(crate) Vec); - -impl crab_cell_app::CellKey for OrderId { - fn canonical_bytes(&self) -> &[u8] { - &self.0 - } -} - -impl Command for ReferenceCronReceiver { - const MODULE: &'static str = SQL_MODULE; - const ID: u32 = 6; - const CODEC_VERSION: u32 = 1; - type Input = CronInvocation; - type Output = (); - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> Result> { - context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "INSERT INTO invoice_receipts(schedule_id, occurrence, payload) VALUES (?1, ?2, ?3)" - .into(), - parameters: vec![ - SqlValue::Blob(input.schedule_id.to_vec()), - SqlValue::Integer(i64::try_from(input.occurrence).map_err(|_| { - Error::Command("cron occurrence exceeds SQL integer range") - })?), - SqlValue::Blob(input.payload), - ], - }], - })?; - Ok(CommandResult::Success(())) - } -} - -crab_cell_app::cell_client! { - pub(crate) struct ReferenceClient (ReferenceApplication) { - pub(crate) fn orders(scope: &OrderId) -> ReferenceOrderCell { - namespace: SQL_NAMESPACE, - module: SQL_MODULE, - commands: { pub(crate) fn receive_cron, prepare_receive_cron: ReferenceCronReceiver = 6; }, - queries: { pub(crate) fn receipt_count: ReferenceReceiptCount = 7; } - } - } -} - -pub(crate) struct ReferenceKv; -impl KvModule for ReferenceKv { - const MODULE: &'static str = KV_MODULE; - const ATOMIC_COMMAND_ID: u32 = 1; - const GET_QUERY_ID: u32 = 2; - const LIST_QUERY_ID: u32 = 3; -} - -pub(crate) struct UnregisteredEffects; -impl EffectModule for UnregisteredEffects { - const MODULE: &'static str = KV_MODULE; - const CLAIM_COMMAND_ID: u32 = 20; - const LEASE_COMMAND_ID: u32 = 21; - const VALIDATE_QUERY_ID: u32 = 22; - const STATUS_QUERY_ID: u32 = 23; -} -impl CellModule for ReferenceKv { - const NAME: &'static str = KV_MODULE; - fn descriptor(&self) -> &'static ModuleDescriptor { - static NAMESPACES: &[NamespaceDescriptor] = &[NamespaceDescriptor { - id: KV_NAMESPACE, - name: "reference-kv", - role: CatalogRole::Kv, - shards: 1, - effect_targets: &[], - dead_letter: None, - }]; - descriptor(KV_MODULE, &[1], &[2, 3], NAMESPACES, &[], &[]) - } - fn register(self, registry: &mut RegistryBuilder) -> Result<()> { - register_kv::(registry) - } -} - -pub(crate) struct ReferenceBlob; -impl MaintenanceModule for ReferenceBlob { - const MODULE: &'static str = BLOB_MODULE; - const TICK_COMMAND_ID: u32 = 3; -} -impl BlobModule for ReferenceBlob { - const NAMESPACE: NamespaceId = BLOB_NAMESPACE; - const MUTATE_COMMAND_ID: u32 = 1; - const QUERY_ID: u32 = 2; -} -impl CellModule for ReferenceBlob { - const NAME: &'static str = BLOB_MODULE; - fn descriptor(&self) -> &'static ModuleDescriptor { - static NAMESPACES: &[NamespaceDescriptor] = &[NamespaceDescriptor { - id: BLOB_NAMESPACE, - name: "reference-blob", - role: CatalogRole::Blob, - shards: 1, - effect_targets: &[], - dead_letter: None, - }]; - descriptor(BLOB_MODULE, &[1, 3], &[2], NAMESPACES, &[], &[]) - } - fn register(self, registry: &mut RegistryBuilder) -> Result<()> { - register_blob::(registry) - } -} - -pub(crate) struct ReferenceQueue; -impl MaintenanceModule for ReferenceQueue { - const MODULE: &'static str = QUEUE_MODULE; - const TICK_COMMAND_ID: u32 = 7; - const QUEUE_DEAD_LETTER: Option = Some(QueueDeadLetterTarget::new( - DEAD_LETTER_MODULE, - DEAD_LETTER_NAMESPACE, - 1, - 1, - 1, - )); -} -impl QueueModule for ReferenceQueue { - const NAMESPACE: NamespaceId = QUEUE_NAMESPACE; - const SEND_COMMAND_ID: u32 = 1; - const CLAIM_COMMAND_ID: u32 = 2; - const LEASE_COMMAND_ID: u32 = 3; - const VALIDATE_QUERY_ID: u32 = 4; - const CONTROL_COMMAND_ID: u32 = 5; - const INFO_QUERY_ID: u32 = 6; -} -impl CellModule for ReferenceQueue { - const NAME: &'static str = QUEUE_MODULE; - fn descriptor(&self) -> &'static ModuleDescriptor { - static NAMESPACES: &[NamespaceDescriptor] = &[NamespaceDescriptor { - id: QUEUE_NAMESPACE, - name: "reference-queue", - role: CatalogRole::Queue, - shards: 1, - effect_targets: &[DEAD_LETTER_NAMESPACE], - dead_letter: Some(DEAD_LETTER_NAMESPACE), - }]; - descriptor( - QUEUE_MODULE, - &[1, 2, 3, 5, 7], - &[4, 6], - NAMESPACES, - &[], - &[], - ) - } - fn register(self, registry: &mut RegistryBuilder) -> Result<()> { - register_queue::(registry) - } -} - -pub(crate) struct ReferenceDeadLetter; -impl MaintenanceModule for ReferenceDeadLetter { - const MODULE: &'static str = DEAD_LETTER_MODULE; - const TICK_COMMAND_ID: u32 = 2; -} -impl QueueModule for ReferenceDeadLetter { - const NAMESPACE: NamespaceId = DEAD_LETTER_NAMESPACE; - const SEND_COMMAND_ID: u32 = 1; - const CLAIM_COMMAND_ID: u32 = 3; - const LEASE_COMMAND_ID: u32 = 4; - const VALIDATE_QUERY_ID: u32 = 5; - const CONTROL_COMMAND_ID: u32 = 6; - const INFO_QUERY_ID: u32 = 7; -} -impl CellModule for ReferenceDeadLetter { - const NAME: &'static str = DEAD_LETTER_MODULE; - fn descriptor(&self) -> &'static ModuleDescriptor { - static NAMESPACES: &[NamespaceDescriptor] = &[NamespaceDescriptor { - id: DEAD_LETTER_NAMESPACE, - name: "reference-dead-letter", - role: CatalogRole::Queue, - shards: 1, - effect_targets: &[], - dead_letter: None, - }]; - descriptor( - DEAD_LETTER_MODULE, - &[1, 2, 3, 4, 6], - &[5, 7], - NAMESPACES, - &[], - &[], - ) - } - fn register(self, registry: &mut RegistryBuilder) -> Result<()> { - register_queue::(registry) - } -} - -pub(crate) struct ReferenceCron; -impl MaintenanceModule for ReferenceCron { - const MODULE: &'static str = CRON_MODULE; - const TICK_COMMAND_ID: u32 = 3; - const CRON_TARGETS: &'static [CronTarget] = - &[CronTarget::new(SQL_MODULE, SQL_NAMESPACE, 6, 1, 1 << 20)]; -} -impl CronModule for ReferenceCron { - const NAMESPACE: NamespaceId = CRON_NAMESPACE; - const MUTATE_COMMAND_ID: u32 = 1; - const QUERY_ID: u32 = 2; -} -impl EffectModule for ReferenceCron { - const MODULE: &'static str = CRON_MODULE; - const CLAIM_COMMAND_ID: u32 = 4; - const LEASE_COMMAND_ID: u32 = 5; - const VALIDATE_QUERY_ID: u32 = 6; - const STATUS_QUERY_ID: u32 = 7; -} -impl CellModule for ReferenceCron { - const NAME: &'static str = CRON_MODULE; - fn descriptor(&self) -> &'static ModuleDescriptor { - static NAMESPACES: &[NamespaceDescriptor] = &[NamespaceDescriptor { - id: CRON_NAMESPACE, - name: "reference-cron", - role: CatalogRole::Cron, - shards: 1, - effect_targets: &[SQL_NAMESPACE], - dead_letter: None, - }]; - descriptor(CRON_MODULE, &[1, 3, 4, 5], &[2, 6, 7], NAMESPACES, &[], &[]) - } - fn register(self, registry: &mut RegistryBuilder) -> Result<()> { - register_cron::(registry)?; - register_effect_delivery::(registry) - } -} - -pub(crate) struct ReferenceWorkflow; -pub(crate) struct ReferenceDefinition; -impl WorkflowDefinition for ReferenceDefinition { - fn digest(&self) -> Digest { - WORKFLOW_DIGEST - } - fn effect_targets(&self) -> &'static [NamespaceId] { - &[SQL_NAMESPACE] - } - fn transition( - &self, - state: &[u8], - event: &[u8], - context: WorkflowContext, - ) -> Result { - if event == b"activity" { - return Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state: b"activity-pending".to_vec(), - result: None, - actions: vec![WorkflowAction::Activity { - activity_type: "reference-activity".into(), - input: b"activity-result".to_vec(), - due_at_ms: context.now_ms(), - expires_at_ms: context.now_ms() + 60_000, - }], - }); - } - if event == b"effect" { - return Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: b"effect-published".to_vec(), - result: Some(b"effect-scheduled".to_vec()), - actions: vec![WorkflowAction::Effect { - intent: crab_cell_runtime::primitives::effects::EffectCommandIntent { - target: CellTarget::new( - context.source().tenant(), - context.source().application(), - SQL_NAMESPACE, - &partition_for_shard(0), - )?, - command_id: ReferenceSql::BATCH_COMMAND_ID, - codec_version: ::CODEC_VERSION, - input: b"qualification-effect".to_vec(), - expires_at_ms: context.now_ms() + 60_000, - }, - }], - }); - } - if event.starts_with(b"activity\0") { - return Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: event.to_vec(), - result: Some(event.to_vec()), - actions: Vec::new(), - }); - } - Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: state.to_vec(), - result: Some(event.to_vec()), - actions: Vec::new(), - }) - } -} -pub(crate) static REFERENCE_DEFINITION: ReferenceDefinition = ReferenceDefinition; -pub(crate) static REFERENCE_DEFINITIONS: &[&'static dyn WorkflowDefinition] = - &[&REFERENCE_DEFINITION]; -pub(crate) static REFERENCE_ACTIVITY_TYPES: &[&str] = &["reference-activity"]; - -pub(crate) struct ReferenceActivity; -impl ActivityHandler for ReferenceActivity { - const TYPE: &'static str = "reference-activity"; - fn execute( - _context: ActivityContext, - input: Vec, - ) -> Pin + Send + 'static>> { - Box::pin(async move { ActivityExecution::Completed(input) }) - } -} -impl MaintenanceModule for ReferenceWorkflow { - const MODULE: &'static str = WORKFLOW_MODULE; - const TICK_COMMAND_ID: u32 = 9; - const WORKFLOW_DEFINITIONS: &'static [&'static dyn WorkflowDefinition] = REFERENCE_DEFINITIONS; -} -impl WorkflowModule for ReferenceWorkflow { - const MODULE: &'static str = WORKFLOW_MODULE; - const NAMESPACE: NamespaceId = WORKFLOW_NAMESPACE; - const CURRENT_DEFINITION: &'static dyn WorkflowDefinition = &REFERENCE_DEFINITION; - const DEFINITIONS: &'static [&'static dyn WorkflowDefinition] = REFERENCE_DEFINITIONS; - const START_COMMAND_ID: u32 = 1; - const SIGNAL_COMMAND_ID: u32 = 2; - const CANCEL_COMMAND_ID: u32 = 3; - const CONTROL_COMMAND_ID: u32 = 4; - const GET_QUERY_ID: u32 = 5; -} -impl WorkflowActivityModule for ReferenceWorkflow { - const ACTIVITY_TYPES: &'static [&'static str] = REFERENCE_ACTIVITY_TYPES; - const ACTIVITY_CLAIM_COMMAND_ID: u32 = 6; - const ACTIVITY_COMPLETE_COMMAND_ID: u32 = 7; - const ACTIVITY_EXTEND_COMMAND_ID: u32 = 8; - const ACTIVITY_VALIDATE_QUERY_ID: u32 = 10; -} -impl EffectModule for ReferenceWorkflow { - const MODULE: &'static str = WORKFLOW_MODULE; - const CLAIM_COMMAND_ID: u32 = 11; - const LEASE_COMMAND_ID: u32 = 12; - const VALIDATE_QUERY_ID: u32 = 13; - const STATUS_QUERY_ID: u32 = 14; -} -impl CellModule for ReferenceWorkflow { - const NAME: &'static str = WORKFLOW_MODULE; - fn descriptor(&self) -> &'static ModuleDescriptor { - static NAMESPACES: &[NamespaceDescriptor] = &[NamespaceDescriptor { - id: WORKFLOW_NAMESPACE, - name: "reference-workflow", - role: CatalogRole::Workflow, - shards: 1, - effect_targets: &[SQL_NAMESPACE], - dead_letter: None, - }]; - descriptor( - WORKFLOW_MODULE, - &[1, 2, 3, 4, 6, 7, 8, 9, 11, 12], - &[5, 10, 13, 14], - NAMESPACES, - &[WORKFLOW_DIGEST], - REFERENCE_ACTIVITY_TYPES, - ) - } - fn register(self, registry: &mut RegistryBuilder) -> Result<()> { - register_workflow::(registry)?; - register_workflow_activities::(registry)?; - register_activity::(registry)?; - register_effect_delivery::(registry)?; - register_maintenance::(registry) - } -} - -pub(crate) struct ReferenceApplication; -impl CellApplication for ReferenceApplication { - const NAME: &'static str = "reference-application"; - - fn register(builder: &mut ApplicationBuilder) -> Result<()> { - Self::register_with_sql(builder, ReferenceSql) - } -} - -impl ReferenceApplication { - pub(super) fn register_with_sql( - builder: &mut ApplicationBuilder, - sql: M, - ) -> Result<()> { - builder.register(sql)?; - builder.register(ReferenceKv)?; - builder.register(ReferenceBlob)?; - builder.register(ReferenceQueue)?; - builder.register(ReferenceDeadLetter)?; - builder.register(ReferenceCron)?; - builder.register(ReferenceWorkflow)?; - for (module, name, namespace, role) in [ - (SQL_MODULE, "sql", SQL_NAMESPACE, CatalogRole::Sql), - (KV_MODULE, "kv", KV_NAMESPACE, CatalogRole::Kv), - (BLOB_MODULE, "blob", BLOB_NAMESPACE, CatalogRole::Blob), - (QUEUE_MODULE, "queue", QUEUE_NAMESPACE, CatalogRole::Queue), - ( - DEAD_LETTER_MODULE, - "dead-letter", - DEAD_LETTER_NAMESPACE, - CatalogRole::Queue, - ), - (CRON_MODULE, "cron", CRON_NAMESPACE, CatalogRole::Cron), - ( - WORKFLOW_MODULE, - "workflow", - WORKFLOW_NAMESPACE, - CatalogRole::Workflow, - ), - ] { - builder.cell_type(CellType::new(module, name, namespace, role, 1)?)?; - } - Ok(()) - } -} - -pub(crate) fn compiled() -> crab_cell_app::CompiledApplication { - ReferenceApplication::compile(BuildDescriptor { - source_revision: "reference-source".into(), - cargo_lock_digest: Digest::from_bytes([42; 32]), - }) - .expect("reference application compiles") -} - -pub(crate) fn compile_reference_in_order(reverse: bool) -> crab_cell_app::CompiledApplication { - let mut builder = ApplicationBuilder::new( - ReferenceApplication::NAME, - BuildDescriptor { - source_revision: "reference-source".into(), - cargo_lock_digest: Digest::from_bytes([42; 32]), - }, - ) - .unwrap(); - if reverse { - builder.register(ReferenceWorkflow).unwrap(); - builder.register(ReferenceCron).unwrap(); - builder.register(ReferenceDeadLetter).unwrap(); - builder.register(ReferenceQueue).unwrap(); - builder.register(ReferenceBlob).unwrap(); - builder.register(ReferenceKv).unwrap(); - builder.register(ReferenceSql).unwrap(); - } else { - builder.register(ReferenceSql).unwrap(); - builder.register(ReferenceKv).unwrap(); - builder.register(ReferenceBlob).unwrap(); - builder.register(ReferenceQueue).unwrap(); - builder.register(ReferenceDeadLetter).unwrap(); - builder.register(ReferenceCron).unwrap(); - builder.register(ReferenceWorkflow).unwrap(); - } - for (module, name, namespace, role) in [ - (SQL_MODULE, "sql", SQL_NAMESPACE, CatalogRole::Sql), - (KV_MODULE, "kv", KV_NAMESPACE, CatalogRole::Kv), - (BLOB_MODULE, "blob", BLOB_NAMESPACE, CatalogRole::Blob), - (QUEUE_MODULE, "queue", QUEUE_NAMESPACE, CatalogRole::Queue), - ( - DEAD_LETTER_MODULE, - "dead-letter", - DEAD_LETTER_NAMESPACE, - CatalogRole::Queue, - ), - (CRON_MODULE, "cron", CRON_NAMESPACE, CatalogRole::Cron), - ( - WORKFLOW_MODULE, - "workflow", - WORKFLOW_NAMESPACE, - CatalogRole::Workflow, - ), - ] { - builder - .cell_type(CellType::new(module, name, namespace, role, 1).unwrap()) - .unwrap(); - } - builder.finish().unwrap() -} diff --git a/crates/crab-cell-app/tests/reference_application/commit.rs b/crates/crab-cell-app/tests/reference_application/commit.rs deleted file mode 100644 index bc0af5ba4..000000000 --- a/crates/crab-cell-app/tests/reference_application/commit.rs +++ /dev/null @@ -1,442 +0,0 @@ -//! The typed-handle commit path through the reference application. - -use super::performance_fixture::PerfFixture; -use crate::*; - -mod wrong_generated_ids { - use super::*; - - crab_cell_app::cell_client! { - pub(super) struct WrongClient (ReferenceApplication) { - pub(super) fn orders(scope: &OrderId) -> WrongOrderCell { - namespace: SQL_NAMESPACE, - module: SQL_MODULE, - commands: { pub(super) fn receive, prepare_receive: ReferenceCronReceiver = 99; }, - queries: { } - } - } - } -} - -struct WrongApplication; - -impl CellApplication for WrongApplication { - const NAME: &'static str = "wrong-application"; - - fn register(_builder: &mut ApplicationBuilder) -> Result<()> { - Ok(()) - } -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn application_handle_rejects_mismatched_author_and_client_registry() { - let fixture = PerfFixture::start(1).await; - let tenant = fixture.sql_target.tenant(); - let application_id = fixture.sql_target.application(); - let compiled = Arc::new(compiled()); - assert!( - ApplicationHandle::::new( - fixture.client.clone(), - Arc::clone(&compiled), - tenant, - application_id, - ) - .is_err() - ); - let other_release = ReferenceApplication::compile(BuildDescriptor { - source_revision: "different-source".into(), - cargo_lock_digest: Digest::from_bytes([42; 32]), - }) - .unwrap(); - assert!( - ApplicationHandle::::new( - fixture.client.clone(), - Arc::new(other_release), - tenant, - application_id, - ) - .is_err() - ); - fixture.shutdown().await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn generated_client_rejects_command_id_outside_descriptor() { - let fixture = PerfFixture::start(1).await; - assert!(matches!( - wrong_generated_ids::WrongClient::new(fixture.typed.clone()), - Err(Error::Registry("generated command differs from stable ID")) - )); - fixture.shutdown().await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn generated_client_derives_target_and_commits_bound_operation() { - let fixture = PerfFixture::start(1).await; - let client = ReferenceClient::new(fixture.typed.clone()).unwrap(); - let order = client.orders(&OrderId(b"order-42".to_vec())).unwrap(); - assert_eq!(order.target(), &fixture.sql_target); - let committed = order - .receive_cron( - reference_identity(31, super::performance_fixture::now_ms()), - CronInvocation { - schedule_id: [32; 16], - generation: 1, - occurrence: 1, - scheduled_at_ms: super::performance_fixture::now_ms(), - payload: b"generated-client".to_vec(), - }, - ) - .await - .unwrap(); - let observed = order - .receipt_count(Some(committed.receipt), ()) - .await - .unwrap(); - assert_eq!(observed.output, 1); - fixture.shutdown().await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn generated_replica_policy_never_falls_back_and_commands_still_use_owner() { - use crab_cell_runtime::client::ReadPolicy; - let fixture = PerfFixture::start(1).await; - let client = ReferenceClient::new(fixture.typed.with_read_policy(ReadPolicy::Replica)).unwrap(); - let order = client.orders(&OrderId(b"order-42".to_vec())).unwrap(); - let committed = order - .receive_cron( - reference_identity(61, super::performance_fixture::now_ms()), - CronInvocation { - schedule_id: [62; 16], - generation: 1, - occurrence: 1, - scheduled_at_ms: super::performance_fixture::now_ms(), - payload: b"replica-policy".to_vec(), - }, - ) - .await - .unwrap(); - assert!(matches!( - order.receipt_count(Some(committed.receipt), ()).await, - Err(InvocationError::NotStarted(Error::ReplicaUnavailable)) - )); - let owner = ReferenceClient::new(fixture.typed.clone()).unwrap(); - assert_eq!( - owner - .orders(&OrderId(b"order-42".to_vec())) - .unwrap() - .receipt_count(Some(committed.receipt), ()) - .await - .unwrap() - .output, - 1 - ); - fixture.shutdown().await; -} - -struct LocalReader(crab_cell_runtime::client::CellReadReplica); - -impl crab_cell_runtime::peer::PeerReplicaResolver for LocalReader { - fn resolve( - &self, - _target: CellTarget, - ) -> Pin< - Box< - dyn Future> - + Send - + 'static, - >, - > { - let reader = self.0.clone(); - Box::pin(async move { Ok(reader) }) - } -} - -struct NoReplicaPeer; - -impl crab_cell_runtime::peer::PeerRoundTrip for NoReplicaPeer { - fn send( - &self, - _target: CellTarget, - _request: Vec, - _remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - Box::pin(async { Err(Error::Peer("test has no peer route")) }) - } -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn generated_queries_report_snapshot_position_while_commands_advance_owner() { - use super::performance_fixture::{node_session, now_ms}; - use crab_cell_runtime::client::{CellReadReplica, ReadPolicy, ReplicaReadRouter}; - use crab_cell_runtime::peer::{PeerPrincipal, PeerSigner, ReplicaPeerClient}; - - let fixture = PerfFixture::start(1).await; - let layout = fixture.layout.clone().unwrap(); - let target = fixture.sql_target.clone(); - let registry = &fixture.registry; - let fleet = Digest::from_bytes([71; 32]); - let image = Digest::from_bytes([72; 32]); - let directory = NodeDirectory::new(layout.clone(), fleet, image, registry.release_digest()); - let reader_session = crab_cell_runtime::SessionId::from_bytes([73; 16]); - let key = SigningKey::from_bytes(&[74; 32]); - for (id, session) in [(75, node_session(0)), (76, reader_session)] { - let now = now_ms(); - directory - .create( - NodeAdvertisement::sign( - NodeId::from_bytes([id; 16]), - session, - format!("https://node-{id}.internal:8081"), - fleet, - Digest::from_bytes([id; 32]), - image, - registry.release_digest(), - &key, - 1, - now, - now + 15_000, - registry.module_digests(), - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 32 << 20, - free_disk_bytes: 1 << 20, - job_credits: 2, - ..NodeCapacity::default() - }, - ) - .unwrap(), - now, - ) - .await - .unwrap(); - } - let control = CellAuthority::new(layout.clone()) - .load(target.cell_id()) - .await - .unwrap() - .unwrap(); - let incarnation = control.value().incarnation; - crab_cell_runtime::read_policy::ReadPolicyStore::new(layout.clone()) - .create(target.cell_id(), incarnation, 1) - .await - .unwrap(); - let reader_runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 512).unwrap(), - 16 << 20, - reader_session, - reference_host(), - ) - .unwrap(); - let files = tempfile::TempDir::new().unwrap(); - let reader = CellReadReplica::open( - reader_runtime.clone(), - Arc::clone(registry), - CellAuthority::new(layout.clone()), - directory.clone(), - CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(), - target.clone(), - &files.path().join("snapshot.sqlite"), - ) - .await - .unwrap(); - let peer = ReplicaPeerClient::new( - Arc::clone(registry), - Arc::new(PeerSigner::new( - reader_session, - registry.release_digest(), - key, - )), - PeerPrincipal { - issuer: "reference".into(), - subject: "reader".into(), - actions: vec!["cell.read".into()], - }, - Arc::new(NoReplicaPeer), - ); - let configured = fixture - .client - .with_read_replicas( - ReplicaReadRouter::new( - CellAuthority::with_telemetry(layout, reader_runtime.telemetry_handle()), - directory, - ), - peer, - Some((reader_session, Arc::new(LocalReader(reader.clone())))), - ) - .unwrap(); - let application = fixture.nodes[0] - .application_handle::( - configured, - target.tenant(), - target.application(), - ) - .unwrap(); - let client = ReferenceClient::new(application.with_read_policy(ReadPolicy::Replica)).unwrap(); - let order = client.orders(&OrderId(b"order-42".to_vec())).unwrap(); - let before = order.receipt_count(None, ()).await.unwrap(); - assert_eq!(before.output, 0); - let committed = order - .receive_cron( - reference_identity(81, now_ms()), - CronInvocation { - schedule_id: [82; 16], - generation: 1, - occurrence: 1, - scheduled_at_ms: now_ms(), - payload: Vec::new(), - }, - ) - .await - .unwrap(); - let stale = order.receipt_count(None, ()).await.unwrap(); - assert_eq!(stale, before); - assert!(matches!( - order.receipt_count(Some(committed.receipt), ()).await, - Err(InvocationError::NotStarted(Error::ReplicaBehind { .. })) - )); - reader - .refresh(&files.path().join("refreshed.sqlite")) - .await - .unwrap(); - let refreshed = order - .receipt_count(Some(committed.receipt), ()) - .await - .unwrap(); - assert_eq!(refreshed.output, 1); - assert!(refreshed.receipt.commit_sequence >= committed.receipt.commit_sequence); - reader.close(); - drop(order); - drop(client); - drop(application); - drop(reader); - reader_runtime.shutdown().await.unwrap(); - fixture.shutdown().await; -} - -#[allow(dead_code)] -fn typed_capability_surface( - handle: &ApplicationHandle, - target: CellTarget, -) -> Result<()> { - let _sql = handle.sql::(target.clone())?; - let _kv = handle.kv::(KV_NAMESPACE)?; - let _blob = handle.blob::()?; - let _queue = handle.queue::()?; - let _cron = handle.cron::()?; - let _workflow = handle.workflow::()?; - let _activities = handle.activities::()?; - let _effects = handle.effects::(target)?; - Ok(()) -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn reference_application_uses_typed_handle_for_a_real_commit() { - let application = Arc::new(compiled()); - let tenant = TenantId::from_bytes([21; 16]); - let application_id = crab_cell_runtime::ApplicationId::from_bytes([22; 16]); - let target = CellTarget::new( - tenant, - application_id, - SQL_NAMESPACE, - &crab_cell_runtime::partition_for_shard(0), - ) - .unwrap(); - let incarnation = IncarnationId::from_bytes([23; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new( - store.clone(), - object_store::path::Path::from("reference-application"), - *application_id.as_bytes(), - ); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new(layout.clone(), tenant); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Sql, - application.registry().module_code(SQL_MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let session = crab_cell_runtime::SessionId::from_bytes([24; 16]); - let observed = authority - .create_initial( - &proof, - incarnation, - Owner { - session, - endpoint: "https://reference.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 4).unwrap(), - 16 * 1024 * 1024, - session, - reference_host(), - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof, - CellReplica::new( - layout, - *target.cell_id().as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(), - authority, - observed, - directory.path().join("reference.sqlite"), - |_| Ok(()), - ) - .await - .unwrap(); - let client = CellClient::local(application.registry(), handle); - let typed = - ApplicationHandle::::new(client, application, tenant, application_id) - .unwrap() - .with_blob_artifact_store(BlobArtifactStore::new(store)); - typed_capability_surface(&typed, target.clone()).unwrap(); - let now_ms = i64::try_from( - std::time::SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap(); - let sql = typed.sql::(target).unwrap(); - let committed = sql - .batch( - MutationIdentity { - request_id: RequestId::from_bytes([25; 16]), - issued_at_ms: now_ms, - expires_at_ms: now_ms + 60_000, - }, - SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT ?1".into(), - parameters: vec![SqlValue::Integer(7)], - }], - }, - ) - .await - .unwrap(); - assert_eq!(committed.output[0].rows.len(), 1); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-app/tests/reference_application/entities.rs b/crates/crab-cell-app/tests/reference_application/entities.rs deleted file mode 100644 index 7b91f6c89..000000000 --- a/crates/crab-cell-app/tests/reference_application/entities.rs +++ /dev/null @@ -1,265 +0,0 @@ -//! Entity-scoped invoice receipts through generated application clients. - -use super::fleet::peer_round_trip; -use crate::*; -use crab_cell_runtime::peer::{PeerPrincipal, PeerRoundTrip, PeerSigner}; -use std::collections::HashMap; - -const ENTITIES_PER_NODE: usize = 4; - -pub(crate) struct EntityReferenceApplication; - -impl CellApplication for EntityReferenceApplication { - const NAME: &'static str = "entity-reference-application"; - - fn register(builder: &mut ApplicationBuilder) -> Result<()> { - builder.register(ReferenceSql)?; - builder.cell_type( - CellType::new(SQL_MODULE, "orders", SQL_NAMESPACE, CatalogRole::Sql, 1)? - .with_entity_partitions()?, - ) - } -} - -crab_cell_app::cell_client! { - pub(crate) struct EntityReferenceClient (EntityReferenceApplication) { - pub(crate) fn orders(scope: &OrderId) -> EntityOrderCell { - namespace: SQL_NAMESPACE, - module: SQL_MODULE, - commands: { pub(crate) fn receive_cron, prepare_receive_cron: ReferenceCronReceiver = 6; }, - queries: { pub(crate) fn receipt_count: ReferenceReceiptCount = 7; } - } - } -} - -pub(super) fn compiled_entities() -> Arc { - Arc::new( - EntityReferenceApplication::compile(BuildDescriptor { - source_revision: "entity-reference-source".into(), - cargo_lock_digest: Digest::from_bytes([42; 32]), - }) - .unwrap(), - ) -} - -fn entity_key(entity: usize) -> OrderId { - OrderId(format!("order-{entity}").into_bytes()) -} - -fn entity_target(application: &crab_cell_app::CompiledApplication, entity: usize) -> CellTarget { - let partition = application.cell_types()[0] - .entity_partition(&entity_key(entity).0) - .unwrap(); - CellTarget::new( - TenantId::from_bytes([81; 16]), - ApplicationId::from_bytes([82; 16]), - SQL_NAMESPACE, - &partition, - ) - .unwrap() -} - -async fn provision_entity( - host: &crab_cell_host::CellNode, - layout: &CellStorageLayout, - directory: &std::path::Path, - node: usize, - entity: usize, - endpoint: String, -) -> CellHandle { - let application = host.application(); - let registry = application.registry(); - let cell_type = application.cell_types()[0]; - let target = entity_target(application, entity); - let catalog = - crab_cell_runtime::cell::catalog::CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - cell_type.role(), - registry.module_code(cell_type.module()).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let incarnation = IncarnationId::from_bytes([u8::try_from(entity + 1).unwrap(); 16]); - let observed = authority - .create_initial( - &proof, - incarnation, - Owner { - session: super::performance_fixture::node_session(node), - endpoint, - }, - ) - .await - .unwrap(); - let replica = CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *incarnation.as_bytes(), - Limits { - max_database_bytes: cell_type.database_limit_bytes(), - max_capture_bytes: cell_type.capture_limit_bytes(), - ..Limits::default() - }, - ) - .unwrap(); - host.runtime() - .bootstrap( - proof, - replica, - authority, - observed, - directory.join(format!("order-{entity}.sqlite")), - super::performance_fixture::install_sql_tables, - ) - .await - .unwrap() -} - -fn application_handle( - application: Arc, - round_trip: Arc, -) -> ApplicationHandle { - let client = peer_client(&application, round_trip); - ApplicationHandle::::new( - client, - application, - TenantId::from_bytes([81; 16]), - ApplicationId::from_bytes([82; 16]), - ) - .unwrap() - .with_blob_artifact_store(BlobArtifactStore::new(Store::new( - Arc::new(InMemory::new()), - ))) -} - -fn peer_client( - application: &crab_cell_app::CompiledApplication, - round_trip: Arc, -) -> CellClient { - let registry = application.registry(); - CellClient::peer( - registry.clone(), - Arc::new(PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - )), - PeerPrincipal { - issuer: "reference".into(), - subject: "entity-client".into(), - actions: vec!["cell.read".into(), "cell.write".into()], - }, - round_trip, - ) -} - -#[test] -fn generated_client_uses_the_declared_entity_partition() { - let application = compiled_entities(); - let handle = application_handle(application.clone(), peer_round_trip(HashMap::new())); - let client = EntityReferenceClient::new(handle).unwrap(); - let first = client.orders(&OrderId(b"order-1".to_vec())).unwrap(); - let second = client.orders(&OrderId(b"order-2".to_vec())).unwrap(); - let repeated = client.orders(&OrderId(b"order-1".to_vec())).unwrap(); - assert_eq!(first.target(), repeated.target()); - assert_ne!(first.target().cell_id(), second.target().cell_id()); - assert_eq!( - first.target().partition(), - application.cell_types()[0] - .entity_partition(b"order-1") - .unwrap(), - ); - for invalid in [Vec::new(), vec![0; 1_025]] { - assert!(client.orders(&OrderId(invalid)).is_err()); - } -} - -struct EntityPrimitiveApplication; - -impl CellApplication for EntityPrimitiveApplication { - const NAME: &'static str = "entity-primitive-application"; - - fn register(builder: &mut ApplicationBuilder) -> Result<()> { - builder.register(ReferenceSql)?; - builder.register(ReferenceKv)?; - builder.register(ReferenceBlob)?; - builder.register(ReferenceQueue)?; - builder.register(ReferenceDeadLetter)?; - builder.register(ReferenceCron)?; - builder.register(ReferenceWorkflow)?; - for cell_type in compiled().cell_types() { - builder.cell_type(cell_type.with_entity_partitions()?)?; - } - Ok(()) - } -} - -#[test] -fn namespace_primitive_handles_reject_entity_topology() { - let application = Arc::new( - EntityPrimitiveApplication::compile(BuildDescriptor { - source_revision: "entity-reference-source".into(), - cargo_lock_digest: Digest::from_bytes([42; 32]), - }) - .unwrap(), - ); - let handle = application_handle::( - application, - peer_round_trip(HashMap::new()), - ); - let results = [ - ("kv", handle.kv::(KV_NAMESPACE).map(|_| ())), - ("blob", handle.blob::().map(|_| ())), - ("queue", handle.queue::().map(|_| ())), - ("cron", handle.cron::().map(|_| ())), - ( - "workflow", - handle.workflow::().map(|_| ()), - ), - ( - "activities", - handle.activities::().map(|_| ()), - ), - ]; - for (name, result) in results { - assert!( - matches!( - result, - Err(Error::Identity( - "namespace capability requires fixed shards" - )) - ), - "{name}: {result:?}", - ); - } -} - -#[test] -fn explicit_sql_and_effect_capabilities_accept_entity_targets() { - let application = compiled_entities(); - let target = CellTarget::new( - TenantId::from_bytes([81; 16]), - ApplicationId::from_bytes([82; 16]), - SQL_NAMESPACE, - &application.cell_types()[0] - .entity_partition(b"order-1") - .unwrap(), - ) - .unwrap(); - let handle = application_handle::( - application, - peer_round_trip(HashMap::new()), - ); - assert!(handle.sql::(target.clone()).is_ok()); - assert!(handle.effects::(target).is_ok()); -} - -mod hosts; -mod process; diff --git a/crates/crab-cell-app/tests/reference_application/entities/hosts.rs b/crates/crab-cell-app/tests/reference_application/entities/hosts.rs deleted file mode 100644 index ba059022e..000000000 --- a/crates/crab-cell-app/tests/reference_application/entities/hosts.rs +++ /dev/null @@ -1,211 +0,0 @@ -//! Independent entity ledgers reached through every public host and the balancer. - -use super::super::fleet::{ - GatewayStats, balancer_round_trip, start_balancer, start_gateway_peer_server, -}; -use super::super::performance_fixture::{node_session, now_ms, rustfs_store}; -use super::super::process_node; -use super::*; -use crab_cell_runtime::cell::executor::Resolution; -use crab_cell_runtime::peer::PeerVerifier; -use futures_util::future::join_all; -use tokio::net::TcpListener; - -const NODES: usize = 3; - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn entity_ledgers_are_isolated_across_three_public_hosts() { - run(Store::new(Arc::new(InMemory::new())), "entity-hosts".into()).await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "real RustFS endpoint and unique CRAB_CELL_TEST_PREFIX required"] -async fn entity_ledgers_are_isolated_across_three_rustfs_hosts() { - let root = std::env::var("CRAB_CELL_TEST_PREFIX").unwrap(); - run(rustfs_store(), root.into()).await; -} - -async fn run(store: Store, root: object_store::path::Path) { - let application = compiled_entities(); - let registry = application.registry(); - let tenant = TenantId::from_bytes([81; 16]); - let application_id = ApplicationId::from_bytes([82; 16]); - let layout = CellStorageLayout::new(store, root, *application_id.as_bytes()); - let authority = CellAuthority::new(layout.clone()); - let directory = tempfile::TempDir::new().unwrap(); - let node_directories = (0..NODES) - .map(|node| { - let path = directory.path().join(format!("node-{node}")); - std::fs::create_dir(&path).unwrap(); - path - }) - .collect::>(); - let mut listeners = Vec::new(); - for _ in 0..NODES { - listeners.push(TcpListener::bind("127.0.0.1:0").await.unwrap()); - } - let addresses = listeners - .iter() - .map(|listener| listener.local_addr().unwrap()) - .collect::>(); - let nodes = join_all(addresses.iter().enumerate().map(|(node, address)| { - process_node::start( - node, - application.clone(), - &layout, - &node_directories[node], - format!("https://{address}"), - ) - })) - .await; - - // Provision from the descriptor independently of the generated accessor: - // otherwise a shared routing error could make both sides agree on one Cell. - let mut owned = vec![Vec::new(); NODES]; - let mut targets = Vec::new(); - let mut routes = HashMap::new(); - for entity in 0..NODES * ENTITIES_PER_NODE { - let node = entity % NODES; - let target = entity_target(&application, entity); - let endpoint = format!("https://{}", addresses[node]); - let handle = provision_entity( - &nodes[node].0, - &layout, - &node_directories[node], - node, - entity, - endpoint, - ) - .await; - owned[node].push(handle); - routes.insert(target.cell_id(), addresses[node]); - targets.push(target); - } - let signer = PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - signer.verifying_key(), - )); - let stats = (0..NODES) - .map(|_| Arc::new(GatewayStats::default())) - .collect::>(); - let routes = Arc::new(std::sync::RwLock::new(routes)); - let mut servers = Vec::new(); - for (node, listener) in listeners.into_iter().enumerate() { - servers.push(start_gateway_peer_server( - listener, - ®istry, - verifier.clone(), - owned[node].clone(), - routes.clone(), - stats[node].clone(), - Some(( - Arc::new(nodes[node].2.clone()), - process_node::directory(&layout, ®istry), - )), - )); - } - let (address, balancer, ingress) = start_balancer(addresses).await; - servers.push(balancer); - let clients = nodes - .iter() - .map(|(host, _, _)| { - let handle = host - .application_handle::( - peer_client(&application, balancer_round_trip(address)), - tenant, - application_id, - ) - .unwrap(); - EntityReferenceClient::new(handle).unwrap() - }) - .collect::>(); - - // Reuse the same mutation identity and payload across different entities. - // The durable request ledger must be scoped to the selected Cell. - let mutation = reference_identity(115, now_ms()); - let input = CronInvocation { - schedule_id: [116; 16], - generation: 1, - occurrence: 1, - scheduled_at_ms: now_ms(), - payload: b"entity-invoice".to_vec(), - }; - let mut receipts = Vec::new(); - for (entity, target) in targets.iter().enumerate() { - let client = &clients[entity % NODES]; - let order = client.orders(&entity_key(entity)).unwrap(); - assert_eq!(order.target(), target); - let prepared = order - .prepare_receive_cron(mutation, input.clone()) - .await - .unwrap(); - let pending = prepared.evidence().clone(); - let first = prepared.execute().await.unwrap(); - assert!(matches!( - client.resolve(&pending).await.unwrap(), - Resolution::Committed(_) - )); - let replay = order.receive_cron(mutation, input.clone()).await.unwrap(); - assert_eq!(first.receipt, replay.receipt); - receipts.push(first.receipt); - } - // Every host-bound author client must see every entity's independent state. - for client in &clients { - for (entity, receipt) in receipts.iter().enumerate() { - let order = client.orders(&entity_key(entity)).unwrap(); - assert_eq!( - order - .receipt_count(Some(*receipt), ()) - .await - .unwrap() - .output, - 1 - ); - } - } - let mut owners = vec![0; NODES]; - for (target, receipt) in targets.iter().zip(&receipts) { - let control = authority.load(target.cell_id()).await.unwrap().unwrap(); - let node = (0..NODES) - .find(|node| control.value().owner.as_ref().unwrap().session == node_session(*node)) - .unwrap(); - owners[node] += 1; - assert!(control.value().root.as_ref().unwrap().commit_sequence >= receipt.commit_sequence); - } - assert_eq!(owners, vec![ENTITIES_PER_NODE; NODES]); - let ingress_counts = ingress.counts(); - assert!(ingress_counts.iter().all(|count| *count > 0)); - assert!(ingress_counts.iter().max().unwrap() - ingress_counts.iter().min().unwrap() <= 1); - for counts in &stats { - let (local, forwarded) = counts.counts(); - assert!( - local > 0 && forwarded > 0, - "local={local} forwarded={forwarded}" - ); - } - println!( - "ENTITY_PROOF nodes={NODES} cells={} owners={owners:?} ingress={ingress_counts:?} visible_receipts={}", - targets.len(), - receipts.len() - ); - for (node, _, _) in nodes { - node.shutdown().await.unwrap(); - } - assert!( - process_node::directory(&layout, ®istry) - .live(now_ms(), 32) - .await - .unwrap() - .is_empty() - ); - for server in servers { - server.abort(); - let _ = server.await; - } -} diff --git a/crates/crab-cell-app/tests/reference_application/entities/process.rs b/crates/crab-cell-app/tests/reference_application/entities/process.rs deleted file mode 100644 index 1429a5cb9..000000000 --- a/crates/crab-cell-app/tests/reference_application/entities/process.rs +++ /dev/null @@ -1,177 +0,0 @@ -//! Private-disk entity owners for the constrained Compose workload. - -use super::super::fleet::{GatewayStats, start_gateway_peer_server}; -use super::super::performance_fixture::{node_session, now_ms, rustfs_store}; -use super::super::process_node; -use super::super::process_performance::{publish_address, wait_for_marker}; -use super::*; -use crab_cell_runtime::peer::PeerVerifier; -use std::{ - env, - net::SocketAddr, - path::Path, - sync::RwLock, - time::{Duration, Instant}, -}; -use tokio::net::TcpListener; - -mod driver; -mod observation; - -async fn endpoints(sync: &Path, count: usize) -> Vec { - let mut endpoints = Vec::new(); - for node in 0..count { - let ready = sync.join(format!("node-{node}.ready")); - wait_for_marker(&ready).await; - endpoints.push(std::fs::read_to_string(ready).unwrap().parse().unwrap()); - } - endpoints -} - -fn publish_marker(path: &Path, value: impl AsRef<[u8]>) { - let temporary = path.with_extension("tmp"); - std::fs::write(&temporary, value).unwrap(); - std::fs::rename(temporary, path).unwrap(); -} - -fn boot_ms() -> u64 { - // Linux exposes the shared boot clock with centisecond precision. Wall - // time can step during a window and cannot align resource observations. - let uptime = std::fs::read_to_string("/proc/uptime").unwrap(); - let (seconds, centiseconds) = uptime - .split_whitespace() - .next() - .unwrap() - .split_once('.') - .unwrap(); - assert_eq!(centiseconds.len(), 2); - seconds.parse::().unwrap() * 1_000 + centiseconds.parse::().unwrap() * 10 -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "Compose entity node: Linux cgroups, private disk and real RustFS required"] -async fn entity_process_node() { - tracing_subscriber::fmt() - .with_max_level(tracing_subscriber::filter::LevelFilter::WARN) - .with_ansi(false) - .try_init() - .unwrap(); - let node: usize = env::var("CRAB_CELL_PERF_PROCESS_NODE") - .unwrap() - .parse() - .unwrap(); - assert!(node < 20); - let sync = env::var("CRAB_CELL_PERF_PROCESS_SYNC").unwrap(); - let sync = Path::new(&sync); - let application = compiled_entities(); - let registry = application.registry(); - let storage = Arc::new(observation::StorageCounters::default()); - let store = rustfs_store().with_storage_observer(storage.clone()); - let layout = CellStorageLayout::new( - store, - env::var("CRAB_CELL_PERF_PROCESS_ROOT").unwrap().into(), - *ApplicationId::from_bytes([82; 16]).as_bytes(), - ); - let directory = tempfile::TempDir::new().unwrap(); - let listener = TcpListener::bind(env::var("CRAB_CELL_PERF_PROCESS_BIND").unwrap()) - .await - .unwrap(); - let address = tokio::net::lookup_host(env::var("CRAB_CELL_PERF_PROCESS_ADVERTISE").unwrap()) - .await - .unwrap() - .next() - .unwrap(); - let endpoint = format!("https://{address}"); - let (host, durability, readers) = process_node::start( - node, - application.clone(), - &layout, - directory.path(), - endpoint.clone(), - ) - .await; - let mut handles = Vec::new(); - for entity in node * ENTITIES_PER_NODE..(node + 1) * ENTITIES_PER_NODE { - handles.push( - provision_entity( - &host, - &layout, - directory.path(), - node, - entity, - endpoint.clone(), - ) - .await, - ); - } - let signer = PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - signer.verifying_key(), - )); - let routes = Arc::new(RwLock::new(HashMap::new())); - let stats = Arc::new(GatewayStats::default()); - let server = start_gateway_peer_server( - listener, - ®istry, - verifier, - handles, - routes.clone(), - stats.clone(), - Some(( - Arc::new(readers), - process_node::directory(&layout, ®istry), - )), - ); - publish_address(&sync.join(format!("node-{node}.ready")), address); - publish_marker(&sync.join(format!("node-{node}.serving")), []); - let mut observations = observation::NodeObservations::new(sync, node); - let mut next_sample = Instant::now(); - let mut stage = 0; - while !sync.join("stop").exists() { - let requested = sync.join("entity-stage.request"); - if requested.exists() { - let count: usize = std::fs::read_to_string(requested).unwrap().parse().unwrap(); - if count > stage && node < count { - assert!([3, 5, 10, 20].contains(&count)); - let addresses = endpoints(sync, count).await; - let expanded = (0..count * ENTITIES_PER_NODE) - .map(|entity| { - ( - entity_target(&application, entity).cell_id(), - addresses[entity / ENTITIES_PER_NODE], - ) - }) - .collect(); - // Publish one complete routing table before admitting the next - // stage, so old gateways can reach newly provisioned owners. - *routes.write().unwrap() = expanded; - stage = count; - publish_marker(&sync.join(format!("node-{node}-stage-{stage}.ready")), []); - } - } - if Instant::now() >= next_sample { - observations.sample(stage, &host, directory.path(), &storage, &stats); - next_sample = Instant::now() + Duration::from_secs(1); - } - tokio::time::sleep(Duration::from_millis(50)).await; - } - observations.sample(stage, &host, directory.path(), &storage, &stats); - assert_eq!(host.stats().active_cells(), ENTITIES_PER_NODE); - host.shutdown().await.unwrap(); - observations.finish(&storage, &durability.object_waits()); - let (local, forwarded) = stats.counts(); - assert!(local > 0 && forwarded > 0); - publish_marker( - &sync.join(format!("node-{node}.counts")), - format!("{local} {forwarded}"), - ); - publish_marker(&sync.join(format!("node-{node}.done")), []); - server.abort(); - let _ = server.await; -} diff --git a/crates/crab-cell-app/tests/reference_application/entities/process/driver.rs b/crates/crab-cell-app/tests/reference_application/entities/process/driver.rs deleted file mode 100644 index 0a0399358..000000000 --- a/crates/crab-cell-app/tests/reference_application/entities/process/driver.rs +++ /dev/null @@ -1,419 +0,0 @@ -//! Scheduled arrivals retain overload and verify every resulting Cell ledger. - -use super::*; -use crate::reference_application::fleet::{balancer_round_trip, start_balancer}; -use crate::reference_application::process_performance::Controller; -use crab_cell_runtime::{ - Receipt, - cell::executor::{Resolution, StoredOutcome}, -}; -use std::{ - fs::File, - io::{BufWriter, Write}, -}; -use tokio::task::JoinSet; - -const WINDOW_SECONDS: usize = 10; - -struct Window { - id: usize, - nodes: usize, - shape: &'static str, - rate_per_node: usize, - concurrency: usize, -} - -impl Window { - fn label(&self) -> String { - format!( - "entities-{}-{}-{}", - self.nodes, self.shape, self.rate_per_node - ) - } -} - -struct Sample { - arrival: usize, - scheduled_us: u64, - started_us: u64, - elapsed_us: u64, - entity: usize, - write: bool, - outcome: &'static str, - sequence: u64, - read_sequence: u64, - count: u64, -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "Compose controller required for scheduled entity traffic on 3/5/10/20 nodes"] -async fn entity_process_scaling() { - let sync = env::var("CRAB_CELL_PERF_PROCESS_SYNC").unwrap(); - let sync = Path::new(&sync); - let mut controller = Controller::new(sync); - let application = compiled_entities(); - let layout = CellStorageLayout::new( - rustfs_store(), - env::var("CRAB_CELL_PERF_PROCESS_ROOT").unwrap().into(), - *ApplicationId::from_bytes([82; 16]).as_bytes(), - ); - let authority = CellAuthority::new(layout.clone()); - let mut owners = BufWriter::new(File::create(sync.join("entity-owners.tsv")).unwrap()); - writeln!(owners, "stage\tentity\tcell\towner\tepoch\tincarnation").unwrap(); - let mut expected = Vec::new(); - let mut window_id = 0; - for nodes in [3, 5, 10, 20] { - assert_eq!( - controller.command("scale", nodes).await, - (0..nodes).collect::>() - ); - for node in 0..nodes { - wait_for_marker(&sync.join(format!("node-{node}.serving"))).await; - } - let addresses = endpoints(sync, nodes).await; - publish_marker(&sync.join("entity-stage.request"), nodes.to_string()); - for node in 0..nodes { - wait_for_marker(&sync.join(format!("node-{node}-stage-{nodes}.ready"))).await; - } - assert_eq!( - process_node::directory(&layout, &application.registry()) - .live(now_ms(), 32) - .await - .unwrap() - .len(), - nodes - ); - let (address, server, ingress) = start_balancer(addresses.clone()).await; - let handle = ApplicationHandle::new( - peer_client(&application, balancer_round_trip(address)), - application.clone(), - TenantId::from_bytes([81; 16]), - ApplicationId::from_bytes([82; 16]), - ) - .unwrap(); - let client = Arc::new(EntityReferenceClient::new(handle).unwrap()); - expected.resize(nodes * ENTITIES_PER_NODE, 0_u64); - for (entity, count) in expected.iter().enumerate() { - let target = entity_target(&application, entity); - let control = authority.load(target.cell_id()).await.unwrap().unwrap(); - let owner = control.value().owner.as_ref().unwrap(); - let node = entity / ENTITIES_PER_NODE; - assert_eq!(owner.session, node_session(node)); - assert_eq!(owner.endpoint, format!("https://{}", addresses[node])); - writeln!( - owners, - "{nodes}\t{entity}\t{:?}\t{node}\t{}\t{:?}", - target.cell_id(), - control.value().epoch, - control.value().incarnation - ) - .unwrap(); - let order = client.orders(&entity_key(entity)).unwrap(); - assert_eq!(order.target(), &target); - assert_eq!(order.receipt_count(None, ()).await.unwrap().output, *count); - } - owners.flush().unwrap(); - for shape in ["uniform", "hot", "skewed"] { - for (rate_per_node, concurrency) in [(1, 4), (4, 16), (16, 64)] { - let window = Window { - id: window_id, - nodes, - shape, - rate_per_node, - concurrency, - }; - run_window(sync, &window, client.clone(), &mut expected).await; - window_id += 1; - } - } - let mut roots = - BufWriter::new(File::create(sync.join(format!("entity-roots-{nodes}.tsv"))).unwrap()); - writeln!( - roots, - "entity\tcell\towner\tepoch\tincarnation\troot_sequence\troot_digest" - ) - .unwrap(); - for entity in 0..expected.len() { - let target = entity_target(&application, entity); - let control = authority.load(target.cell_id()).await.unwrap().unwrap(); - let owner = control.value().owner.as_ref().unwrap(); - let node = entity / ENTITIES_PER_NODE; - assert_eq!(owner.session, node_session(node)); - let root = control.value().root.as_ref().unwrap(); - writeln!( - roots, - "{entity}\t{:?}\t{node}\t{}\t{:?}\t{}\t{:?}", - target.cell_id(), - control.value().epoch, - control.value().incarnation, - root.commit_sequence, - root.digest - ) - .unwrap(); - } - roots.flush().unwrap(); - let counts = ingress.counts(); - assert!(counts.iter().all(|count| *count > 0)); - assert!(counts.iter().max().unwrap() - counts.iter().min().unwrap() <= 1); - publish_marker( - &sync.join(format!("entity-ingress-{nodes}.txt")), - counts - .iter() - .map(usize::to_string) - .collect::>() - .join(" "), - ); - println!( - "ENTITY_SCALE nodes={nodes} cells={} ingress={counts:?}", - expected.len() - ); - server.abort(); - let _ = server.await; - } - publish_marker(&sync.join("stop"), []); - for node in 0..20 { - wait_for_marker(&sync.join(format!("node-{node}.done"))).await; - } - assert!( - process_node::directory(&layout, &application.registry()) - .live(now_ms(), 32) - .await - .unwrap() - .is_empty() - ); -} - -fn destination(shape: &str, arrival: usize, cells: usize) -> (usize, bool) { - match shape { - "uniform" => (arrival % cells, true), - "hot" => ( - if arrival % 5 == 4 { - 1 + (arrival / 5) % (cells - 1) - } else { - 0 - }, - true, - ), - "skewed" => { - if arrival.is_multiple_of(5) { - ((arrival / 5) % cells, true) - } else { - (arrival % ENTITIES_PER_NODE, false) - } - } - _ => panic!("undeclared traffic shape"), - } -} - -async fn run_window( - sync: &Path, - window: &Window, - client: Arc, - expected: &mut [u64], -) { - let rate = window.nodes * window.rate_per_node; - let planned = rate * WINDOW_SECONDS; - let label = window.label(); - let mut output = BufWriter::new(File::create(sync.join(format!("{label}.tsv"))).unwrap()); - writeln!( - output, - "arrival\tscheduled_us\tstarted_us\telapsed_us\tentity\tkind\toutcome\tsequence\tread_sequence\tcount" - ) - .unwrap(); - output.flush().unwrap(); - let started_ms = now_ms(); - let started_boot_ms = boot_ms(); - let started = Instant::now(); - let mut jobs = JoinSet::new(); - let mut samples = Vec::with_capacity(planned); - for arrival in 0..planned { - let scheduled_us = (arrival as u64 * 1_000_000) / rate as u64; - tokio::time::sleep_until((started + Duration::from_micros(scheduled_us)).into()).await; - while let Some(result) = jobs.try_join_next() { - retain_sample(&mut output, &mut samples, result.unwrap()); - } - let started_us = started.elapsed().as_micros() as u64; - let (entity, write) = destination(window.shape, arrival, expected.len()); - let mut sample = Sample { - arrival, - scheduled_us, - started_us, - elapsed_us: 0, - entity, - write, - outcome: "client_full", - sequence: 0, - read_sequence: 0, - count: 0, - }; - if started_us >= (arrival as u64 + 1) * 1_000_000 / rate as u64 { - sample.outcome = "scheduler_late"; - } - if sample.outcome == "scheduler_late" || jobs.len() >= window.concurrency { - retain_sample(&mut output, &mut samples, sample); - continue; - } - let client = client.clone(); - let request = window.id * 1_000_000 + arrival; - let admitted = started + Duration::from_micros(sample.started_us); - jobs.spawn(async move { execute(client, sample, request, admitted).await }); - } - tokio::time::sleep_until((started + Duration::from_secs(WINDOW_SECONDS as u64)).into()).await; - while let Some(result) = jobs.join_next().await { - retain_sample(&mut output, &mut samples, result.unwrap()); - } - let elapsed_us = started.elapsed().as_micros() as u64; - let ended_ms = now_ms(); - let ended_boot_ms = boot_ms(); - samples.sort_by_key(|sample| sample.arrival); - assert_eq!(samples.len(), planned); - for sample in &samples { - if sample.write && sample.sequence > 0 { - expected[sample.entity] += 1; - } - } - let mut checks = - BufWriter::new(File::create(sync.join(format!("{label}-readback.tsv"))).unwrap()); - writeln!(checks, "entity\texpected\tactual\tsequence").unwrap(); - for (entity, expected) in expected.iter().enumerate() { - let actual = client - .orders(&entity_key(entity)) - .unwrap() - .receipt_count(None, ()) - .await - .unwrap(); - writeln!( - checks, - "{entity}\t{expected}\t{}\t{}", - actual.output, actual.receipt.commit_sequence - ) - .unwrap(); - assert_eq!( - actual.output, *expected, - "{label}: Cell {entity} lost or duplicated a mutation" - ); - } - checks.flush().unwrap(); - publish_marker( - &sync.join(format!("{label}-window.tsv")), - format!( - "window_id\tnodes\tshape\trate_per_node\tconcurrency\tseconds\tstarted_ms\tended_ms\tstarted_boot_ms\tended_boot_ms\telapsed_us\n{}\t{}\t{}\t{}\t{}\t{WINDOW_SECONDS}\t{started_ms}\t{ended_ms}\t{started_boot_ms}\t{ended_boot_ms}\t{elapsed_us}\n", - window.id, window.nodes, window.shape, window.rate_per_node, window.concurrency - ), - ); - let complete = samples - .iter() - .filter(|sample| sample.outcome == "ok" || sample.outcome == "resolved") - .count(); - println!( - "ENTITY_WINDOW label={label} planned={planned} complete={complete} elapsed_us={elapsed_us}" - ); -} - -fn retain_sample(output: &mut BufWriter, samples: &mut Vec, sample: Sample) { - writeln!( - output, - "{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}\t{}", - sample.arrival, - sample.scheduled_us, - sample.started_us, - sample.elapsed_us, - sample.entity, - if sample.write { "write" } else { "read" }, - sample.outcome, - sample.sequence, - sample.read_sequence, - sample.count - ) - .unwrap(); - output.flush().unwrap(); - samples.push(sample); -} - -async fn execute( - client: Arc, - mut sample: Sample, - request: usize, - started: Instant, -) -> Sample { - let order = client.orders(&entity_key(sample.entity)).unwrap(); - let minimum = if sample.write { - let input = CronInvocation { - schedule_id: [117; 16], - generation: 1, - occurrence: request as u64 + 1, - scheduled_at_ms: now_ms(), - payload: b"distributed-entity-invoice".to_vec(), - }; - match order - .receive_cron(qualification_identity(request as u64, now_ms()), input) - .await - { - Ok(committed) => { - sample.outcome = "ok"; - Some(committed.receipt) - } - Err(InvocationError::NotStarted(error)) => { - eprintln!("ENTITY_NOT_STARTED request={request} error={error}"); - sample.outcome = "not_started"; - sample.elapsed_us = started.elapsed().as_micros() as u64; - return sample; - } - Err(InvocationError::Pending(pending)) => { - let deadline = Instant::now() + Duration::from_secs(30); - loop { - match client.resolve(&pending).await { - Ok(Resolution::Committed(StoredOutcome::Success { - commit_sequence, - .. - })) => { - sample.outcome = "resolved"; - break Some(Receipt { - cell: pending.target().cell_id(), - incarnation: pending.incarnation(), - commit_sequence, - }); - } - Ok(Resolution::Absent) => { - sample.outcome = "absent"; - sample.elapsed_us = started.elapsed().as_micros() as u64; - return sample; - } - Ok(Resolution::Unknown) | Err(InvocationError::NotStarted(_)) => { - assert!( - Instant::now() < deadline, - "unresolved request {request}: {pending:?}" - ); - tokio::time::sleep(Duration::from_millis(100)).await; - } - other => panic!("unexpected request resolution {request}: {other:?}"), - } - } - } - other => panic!("unexpected mutation outcome {request}: {other:?}"), - } - } else { - sample.outcome = "ok"; - None - }; - if let Some(receipt) = minimum { - sample.sequence = receipt.commit_sequence; - } - match order.receipt_count(minimum, ()).await { - Ok(observed) => { - sample.count = observed.output; - sample.read_sequence = observed.receipt.commit_sequence; - } - Err(InvocationError::NotStarted(error)) => { - eprintln!("ENTITY_READ_FAILED request={request} error={error}"); - sample.outcome = if sample.write { - "write_only" - } else { - "read_failed" - }; - } - other => panic!("unexpected read outcome {request}: {other:?}"), - } - sample.elapsed_us = started.elapsed().as_micros() as u64; - sample -} diff --git a/crates/crab-cell-app/tests/reference_application/entities/process/observation.rs b/crates/crab-cell-app/tests/reference_application/entities/process/observation.rs deleted file mode 100644 index 885cd38aa..000000000 --- a/crates/crab-cell-app/tests/reference_application/entities/process/observation.rs +++ /dev/null @@ -1,157 +0,0 @@ -//! Cumulative provider operations and per-node Linux resource samples. - -use super::*; -use crab_storage::{StorageObservation, StorageObserver, StorageOperation, StorageOutcome}; -use std::{ - fs::File, - io::{BufWriter, Write}, - sync::atomic::{AtomicU64, Ordering}, -}; - -#[derive(Default)] -pub(super) struct StorageCounters { - started: AtomicU64, - finished: AtomicU64, - outcomes: [[AtomicU64; 9]; 11], - bytes_read: AtomicU64, - bytes_written: AtomicU64, -} - -impl StorageObserver for StorageCounters { - fn started(&self, _: StorageOperation) { - self.started.fetch_add(1, Ordering::Relaxed); - } - - fn finished(&self, observation: StorageObservation) { - self.outcomes[observation.operation.index()][observation.outcome.index()] - .fetch_add(1, Ordering::Relaxed); - self.bytes_read - .fetch_add(observation.bytes_read, Ordering::Relaxed); - self.bytes_written - .fetch_add(observation.bytes_written, Ordering::Relaxed); - self.finished.fetch_add(1, Ordering::Relaxed); - } -} - -pub(super) struct NodeObservations { - resources: BufWriter, - objects: BufWriter, - durability: BufWriter, -} - -impl NodeObservations { - pub(super) fn new(sync: &Path, node: usize) -> Self { - let create = |name| { - BufWriter::new(File::create(sync.join(format!("node-{node}-{name}.tsv"))).unwrap()) - }; - let mut resources = create("resources"); - writeln!(resources, "at_ms\tboot_ms\tstage\tactive_cells\tworker_jobs\tprimitive_jobs\thydration_jobs\tretained_bytes\tdisk_reserved_bytes\tdisk_bytes\tcpu_usage_us\tthrottled_us\tmemory_current_bytes\tmemory_peak_bytes\tgateway_local\tgateway_forwarded\tobject_started\tobject_finished\tbytes_read\tbytes_written").unwrap(); - Self { - resources, - objects: create("objects"), - durability: create("durability"), - } - } - - pub(super) fn sample( - &mut self, - stage: usize, - host: &crab_cell_host::CellNode, - root: &Path, - storage: &StorageCounters, - gateway: &GatewayStats, - ) { - let cpu = std::fs::read_to_string("/sys/fs/cgroup/cpu.stat").unwrap(); - let cpu_value = |name| { - cpu.lines() - .find_map(|line| { - let (key, value) = line.split_once(' ')?; - (key == name).then(|| value.parse::().unwrap()) - }) - .unwrap() - }; - let memory = |name| { - std::fs::read_to_string(format!("/sys/fs/cgroup/{name}")) - .unwrap() - .trim() - .parse::() - .unwrap() - }; - let stats = host.stats(); - let (local, forwarded) = gateway.counts(); - let values = [ - now_ms() as u64, - boot_ms(), - stage as u64, - stats.active_cells() as u64, - stats.worker_jobs() as u64, - stats.primitive_jobs() as u64, - stats.hydration_jobs() as u64, - stats.retained_bytes() as u64, - stats.local_disk_reserved_bytes(), - disk_bytes(root), - cpu_value("usage_usec"), - cpu_value("throttled_usec"), - memory("memory.current"), - memory("memory.peak"), - local as u64, - forwarded as u64, - storage.started.load(Ordering::Relaxed), - storage.finished.load(Ordering::Relaxed), - storage.bytes_read.load(Ordering::Relaxed), - storage.bytes_written.load(Ordering::Relaxed), - ]; - writeln!( - self.resources, - "{}", - values.map(|value| value.to_string()).join("\t") - ) - .unwrap(); - self.resources.flush().unwrap(); - } - - pub(super) fn finish(&mut self, storage: &StorageCounters, waits: &[Duration]) { - writeln!(self.objects, "operation\toutcome\tcount").unwrap(); - for operation in StorageOperation::ALL { - for outcome in StorageOutcome::ALL { - let count = - storage.outcomes[operation.index()][outcome.index()].load(Ordering::Relaxed); - writeln!( - self.objects, - "{}\t{}\t{count}", - operation.label(), - outcome.label() - ) - .unwrap(); - } - } - writeln!(self.durability, "object_wait_us").unwrap(); - assert!(!waits.is_empty()); - for wait in waits { - writeln!(self.durability, "{}", wait.as_micros()).unwrap(); - } - self.objects.flush().unwrap(); - self.durability.flush().unwrap(); - } -} - -fn disk_bytes(root: &Path) -> u64 { - std::fs::read_dir(root) - .unwrap() - .map(|entry| { - let entry = entry.unwrap(); - // Publication removes temporary files while this sampler walks. - // A disappeared file contributes no retained bytes to this sample. - let metadata = match entry.metadata() { - Ok(metadata) => metadata, - Err(error) if error.kind() == std::io::ErrorKind::NotFound => return 0, - Err(error) => panic!("cannot sample node disk: {error}"), - }; - if metadata.is_dir() { - disk_bytes(&entry.path()) - } else { - metadata.len() - } - }) - .sum() -} diff --git a/crates/crab-cell-app/tests/reference_application/fleet.rs b/crates/crab-cell-app/tests/reference_application/fleet.rs deleted file mode 100644 index 48ef7197f..000000000 --- a/crates/crab-cell-app/tests/reference_application/fleet.rs +++ /dev/null @@ -1,412 +0,0 @@ -use super::performance_fixture::now_ms; -use crate::*; -use std::{ - collections::{HashMap, HashSet}, - net::SocketAddr, - sync::{ - RwLock, - atomic::{AtomicUsize, Ordering}, - }, - time::{Duration, Instant}, -}; - -use crab_cell_runtime::identity::CellId; -use crab_cell_runtime::peer::{ - MAX_PEER_REQUEST_BYTES, PeerAuthorizer, PeerCellResolver, PeerDispatcher, PeerRoundTrip, - PeerVerifier, UnverifiedPeerRequest, VerifiedPeerRequest, wire, -}; -use tokio::io::{AsyncReadExt, AsyncWriteExt}; -use tokio::net::{TcpListener, TcpStream}; - -type Readers = ( - Arc, - NodeDirectory, -); - -struct FleetResolver(Arc>); - -impl PeerCellResolver for FleetResolver { - fn resolve( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>> { - let handles = Arc::clone(&self.0); - Box::pin(async move { - handles - .get(&target.cell_id()) - .cloned() - .ok_or(Error::CellNotActive) - }) - } -} - -struct FleetAuthorizer; - -impl PeerAuthorizer for FleetAuthorizer { - fn authorize(&self, request: &VerifiedPeerRequest) -> Result<()> { - let action = match request.operation_tag() { - 10 | 12 => "cell.write", - 11 => match request.operation() { - Some(wire::peer_request::Operation::Read(read)) => match read.operation { - Some(wire::read_request::Operation::ReplicaActivate(_)) => { - "cell.replica.activate" - } - Some(wire::read_request::Operation::ReplicaStatus(_)) => "cell.replica.status", - _ => "cell.read", - }, - _ => return Err(Error::PeerAuthorization("unsupported read operation")), - }, - 13 | 14 => "reference.cron.deliver", - _ => return Err(Error::PeerAuthorization("unsupported fleet operation")), - }; - if request.permits(action) { - Ok(()) - } else { - Err(Error::PeerAuthorization("fleet action is missing")) - } - } -} - -struct TcpRoundTrip(Arc>); - -struct BalancerRoundTrip(SocketAddr); - -pub(super) struct BalancerStats(Vec); - -impl BalancerStats { - pub(super) fn counts(&self) -> Vec { - self.0 - .iter() - .map(|count| count.load(Ordering::Relaxed)) - .collect() - } -} - -#[derive(Default)] -pub(super) struct GatewayStats { - local: AtomicUsize, - forwarded: AtomicUsize, -} - -impl GatewayStats { - pub(super) fn counts(&self) -> (usize, usize) { - ( - self.local.load(Ordering::Relaxed), - self.forwarded.load(Ordering::Relaxed), - ) - } -} - -pub(super) fn peer_round_trip(endpoints: HashMap) -> Arc { - Arc::new(TcpRoundTrip(Arc::new(endpoints))) -} - -pub(super) fn balancer_round_trip(address: SocketAddr) -> Arc { - Arc::new(BalancerRoundTrip(address)) -} - -impl PeerRoundTrip for TcpRoundTrip { - fn send( - &self, - target: CellTarget, - request: Vec, - remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let address = self.0.get(&target.cell_id()).copied(); - Box::pin(async move { - let address = address.ok_or(Error::CellNotActive)?; - send_tcp(address, request, remaining_ms).await - }) - } -} - -impl PeerRoundTrip for BalancerRoundTrip { - fn send( - &self, - _target: CellTarget, - request: Vec, - remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - Box::pin(send_tcp(self.0, request, remaining_ms)) - } -} - -pub(super) async fn start_balancer( - nodes: impl Into>, -) -> (SocketAddr, tokio::task::JoinHandle<()>, Arc) { - let nodes = nodes.into(); - assert!(!nodes.is_empty()); - let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); - let address = listener.local_addr().unwrap(); - let stats = Arc::new(BalancerStats( - nodes.iter().map(|_| AtomicUsize::new(0)).collect(), - )); - let counts = Arc::clone(&stats); - let next = Arc::new(AtomicUsize::new(0)); - let server = tokio::spawn(async move { - while let Ok((socket, _)) = listener.accept().await { - let node = next.fetch_add(1, Ordering::Relaxed) % nodes.len(); - let entry = nodes[node]; - counts.0[node].fetch_add(1, Ordering::Relaxed); - tokio::spawn(async move { - let _ = serve_balancer(socket, entry).await; - }); - } - }); - (address, server, stats) -} - -async fn serve_balancer(mut socket: TcpStream, entry: SocketAddr) -> Result<()> { - let mut length = [0; 4]; - socket.read_exact(&mut length).await.map_err(peer_io)?; - let length = u32::from_be_bytes(length) as usize; - if length > MAX_PEER_REQUEST_BYTES { - return Err(Error::Peer("fleet request exceeds peer byte limit")); - } - let mut request = vec![0; length]; - socket.read_exact(&mut request).await.map_err(peer_io)?; - // The balancer is a transport hop: the entry node authenticates the - // unchanged request and may use the protocol's one forwarding hop. - let reply = send_tcp(entry, request, 60_000).await?; - socket - .write_all(&(reply.len() as u32).to_be_bytes()) - .await - .map_err(peer_io)?; - socket.write_all(&reply).await.map_err(peer_io)?; - Ok(()) -} - -pub(super) async fn send_tcp( - address: SocketAddr, - request: Vec, - remaining_ms: u32, -) -> Result> { - tokio::time::timeout(Duration::from_millis(u64::from(remaining_ms)), async { - let mut socket = - TcpStream::connect(address) - .await - .map_err(|source| Error::PeerTransport { - context: "fleet peer connect", - source: Box::new(source), - })?; - socket - .write_all(&(request.len() as u32).to_be_bytes()) - .await - .map_err(peer_io)?; - socket.write_all(&request).await.map_err(peer_io)?; - let mut length = [0; 4]; - socket.read_exact(&mut length).await.map_err(peer_io)?; - let length = u32::from_be_bytes(length) as usize; - if length > MAX_PEER_REQUEST_BYTES { - return Err(Error::Peer("fleet reply exceeds peer byte limit")); - } - let mut reply = vec![0; length]; - socket.read_exact(&mut reply).await.map_err(peer_io)?; - Ok(reply) - }) - .await - .map_err(|source| Error::PeerTransportUnknown { - context: "fleet peer deadline", - source: Box::new(source), - })? -} - -fn peer_io(source: std::io::Error) -> Error { - Error::PeerTransportUnknown { - context: "fleet peer round trip", - source: Box::new(source), - } -} - -pub(super) async fn start_peer_servers( - verifier: Arc, - owned: Vec<(Arc, Vec)>, -) -> (Arc, Vec>) { - let mut endpoints = HashMap::new(); - let mut servers = Vec::new(); - for (registry, handles) in owned { - let ids = handles.iter().map(CellHandle::cell_id).collect::>(); - let (address, server) = - start_peer_server(®istry, Arc::clone(&verifier), handles, None).await; - // Pin routing for this run; the receiving resolver rejects a target - // it does not own, so a wrong route cannot pass the action checks. - for id in ids { - endpoints.insert(id, address); - } - servers.push(server); - } - (peer_round_trip(endpoints), servers) -} - -pub(super) async fn start_peer_server( - registry: &Arc, - verifier: Arc, - handles: Vec, - replicas: Option, -) -> (SocketAddr, tokio::task::JoinHandle<()>) { - let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); - let address = listener.local_addr().unwrap(); - let server = start_bound_peer_server(listener, registry, verifier, handles, replicas); - (address, server) -} - -pub(super) fn start_bound_peer_server( - listener: TcpListener, - registry: &Arc, - verifier: Arc, - handles: Vec, - replicas: Option, -) -> tokio::task::JoinHandle<()> { - let directory = replicas.as_ref().map(|(_, directory)| directory.clone()); - serve_listener( - listener, - verifier, - dispatcher(registry, handles, replicas), - None, - directory, - ) -} - -fn dispatcher( - registry: &Arc, - handles: Vec, - replicas: Option, -) -> Arc { - let dispatcher = PeerDispatcher::new( - Arc::clone(registry), - Arc::new(FleetResolver(Arc::new( - handles - .into_iter() - .map(|handle| (handle.cell_id(), handle)) - .collect(), - ))), - Arc::new(FleetAuthorizer), - ); - Arc::new(match replicas { - Some((replicas, _)) => dispatcher - .with_replica_resolver(replicas.clone()) - .with_replica_control(replicas), - None => dispatcher, - }) -} - -struct Gateway { - local: HashSet, - owners: Arc>>, - stats: Arc, -} - -pub(super) fn start_gateway_peer_server( - listener: TcpListener, - registry: &Arc, - verifier: Arc, - handles: Vec, - owners: Arc>>, - stats: Arc, - replicas: Option, -) -> tokio::task::JoinHandle<()> { - let directory = replicas.as_ref().map(|(_, directory)| directory.clone()); - let gateway = Arc::new(Gateway { - local: handles.iter().map(CellHandle::cell_id).collect(), - owners, - stats, - }); - serve_listener( - listener, - verifier, - dispatcher(registry, handles, replicas), - Some(gateway), - directory, - ) -} - -fn serve_listener( - listener: TcpListener, - verifier: Arc, - dispatcher: Arc, - gateway: Option>, - directory: Option, -) -> tokio::task::JoinHandle<()> { - tokio::spawn(async move { - while let Ok((socket, _)) = listener.accept().await { - let verifier = Arc::clone(&verifier); - let dispatcher = Arc::clone(&dispatcher); - let gateway = gateway.clone(); - let directory = directory.clone(); - tokio::spawn(async move { - let _ = serve_peer(socket, verifier, dispatcher, gateway, directory).await; - }); - } - }) -} - -async fn serve_peer( - mut socket: TcpStream, - verifier: Arc, - dispatcher: Arc, - gateway: Option>, - directory: Option, -) -> Result<()> { - let started = Instant::now(); - let mut length = [0; 4]; - socket.read_exact(&mut length).await.map_err(peer_io)?; - let length = u32::from_be_bytes(length) as usize; - if length > MAX_PEER_REQUEST_BYTES { - return Err(Error::Peer("fleet request exceeds peer byte limit")); - } - let mut request = vec![0; length]; - socket.read_exact(&mut request).await.map_err(peer_io)?; - let now = now_ms(); - let decoded = UnverifiedPeerRequest::decode(&request)?; - let verified = if let Some(directory) = directory - .filter(|_| decoded.session() != crab_cell_runtime::SessionId::from_bytes([77; 16])) - { - let enrolled = tokio::time::timeout( - Duration::from_millis(u64::from(decoded.remaining_ms())), - directory.load(decoded.session(), now), - ) - .await - .map_err(|_| Error::Deadline)?? - .ok_or(Error::Fenced)?; - let node = enrolled.advertisement(); - PeerVerifier::new(node.session(), node.release(), node.verifying_key()?) - .verify_decoded(decoded, now_ms())? - } else { - verifier.verify_decoded(decoded, now)? - }; - let replica = matches!(verified.operation(), Some(wire::peer_request::Operation::Read(read)) - if matches!(read.operation, Some(wire::read_request::Operation::ReplicaQuery(_) - | wire::read_request::Operation::ReplicaActivate(_) - | wire::read_request::Operation::ReplicaStatus(_)))); - let reply = match gateway { - // A selected secondary must execute its snapshot locally. Forwarding - // an explicit replica query would hide missing reader admission. - Some(gateway) if !replica && !gateway.local.contains(&verified.target().cell_id()) => { - let address = gateway - .owners - .read() - .unwrap() - .get(&verified.target().cell_id()) - .copied() - .ok_or(Error::CellNotActive)?; - let elapsed = u32::try_from(started.elapsed().as_millis()).unwrap_or(u32::MAX); - let remaining = verified.remaining_ms().saturating_sub(elapsed.max(1)); - let forwarded = verified.forward(remaining)?; - let reply = send_tcp(address, forwarded, remaining).await?; - gateway.stats.forwarded.fetch_add(1, Ordering::Relaxed); - reply - } - Some(gateway) => { - let reply = dispatcher.dispatch_bytes(&verified, now).await?; - gateway.stats.local.fetch_add(1, Ordering::Relaxed); - reply - } - None => dispatcher.dispatch_bytes(&verified, now).await?, - }; - socket - .write_all(&(reply.len() as u32).to_be_bytes()) - .await - .map_err(peer_io)?; - socket.write_all(&reply).await.map_err(peer_io)?; - Ok(()) -} diff --git a/crates/crab-cell-app/tests/reference_application/harness.rs b/crates/crab-cell-app/tests/reference_application/harness.rs deleted file mode 100644 index 164f5ed18..000000000 --- a/crates/crab-cell-app/tests/reference_application/harness.rs +++ /dev/null @@ -1,320 +0,0 @@ -//! Runtime, session, and identity fixtures shared by the suite. - -use crate::*; - -#[expect( - clippy::too_many_arguments, - reason = "the qualification fixture keeps each Cell contract explicit" -)] -pub(crate) async fn bootstrap_reference_cell( - runtime: &CellRuntime, - registry: &Arc, - layout: &CellStorageLayout, - directory: &tempfile::TempDir, - tenant: TenantId, - application: ApplicationId, - session: crab_cell_runtime::SessionId, - namespace: NamespaceId, - role: CatalogRole, - module: &'static str, - incarnation_byte: u8, - initialize: F, -) -> crab_cell_runtime::Result -where - F: for<'connection> FnOnce( - &crab_ltx::rusqlite::Transaction<'connection>, - ) -> crab_cell_runtime::Result<()> - + Send - + 'static, -{ - let target = CellTarget::new(tenant, application, namespace, &partition_for_shard(0))?; - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new(layout.clone(), tenant); - let proof = catalog - .provision(CatalogEntry::new( - &target, - role, - registry - .module_code(module) - .ok_or(crab_cell_runtime::Error::Registry("module code is missing"))?, - 1, - )?) - .await?; - let authority = CellAuthority::new(layout.clone()); - let incarnation = IncarnationId::from_bytes([incarnation_byte; 16]); - let observed = authority - .create_initial( - &proof, - incarnation, - Owner { - session, - endpoint: format!("https://{module}.internal:8081"), - }, - ) - .await?; - let limits = reference_limits(namespace)?; - runtime - .bootstrap( - proof, - CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *incarnation.as_bytes(), - limits, - )?, - authority, - observed, - directory.path().join(format!("{module}.sqlite")), - initialize, - ) - .await -} - -pub(crate) fn reference_limits(namespace: NamespaceId) -> Result { - let application = compiled(); - let cell_type = application - .cell_types() - .iter() - .find(|cell_type| cell_type.namespace() == namespace) - .ok_or(Error::Registry("reference Cell type is missing"))?; - Ok(Limits { - max_database_bytes: cell_type.database_limit_bytes(), - max_capture_bytes: cell_type.capture_limit_bytes(), - ..Limits::default() - }) -} - -pub(crate) fn reference_host() -> Host { - Host::default().with_local_disk_budget(DiskBudget::new(1 << 30)) -} - -pub(crate) async fn seed_reference_session( - layout: &CellStorageLayout, - session: crab_cell_runtime::SessionId, -) { - let directory = NodeDirectory::new( - layout.clone(), - Digest::from_bytes([90; 32]), - Digest::from_bytes([91; 32]), - Digest::from_bytes([92; 32]), - ); - let key = SigningKey::from_bytes(&[93; 32]); - directory - .create( - NodeAdvertisement::sign( - NodeId::from_bytes(*session.as_bytes()), - session, - "https://reference-expired.internal:8081".into(), - directory.fleet(), - Digest::from_bytes([94; 32]), - Digest::from_bytes([91; 32]), - Digest::from_bytes([92; 32]), - &key, - 1, - 1, - 10_001, - vec![Digest::from_bytes([95; 32])], - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 1, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .unwrap(), - 1, - ) - .await - .unwrap(); -} - -pub(crate) async fn fence_reference_session( - layout: &CellStorageLayout, - session: crab_cell_runtime::SessionId, - claimant: crab_cell_runtime::SessionId, -) -> FencedNodeSession { - let directory = NodeDirectory::new( - layout.clone(), - Digest::from_bytes([90; 32]), - Digest::from_bytes([91; 32]), - Digest::from_bytes([92; 32]), - ); - let key = SigningKey::from_bytes(&[93; 32]); - directory - .create( - NodeAdvertisement::sign( - NodeId::from_bytes(*claimant.as_bytes()), - claimant, - "https://reference-claimant.internal:8081".into(), - directory.fleet(), - Digest::from_bytes([94; 32]), - Digest::from_bytes([91; 32]), - Digest::from_bytes([92; 32]), - &key, - 10_000, - 10_000, - 20_000, - vec![Digest::from_bytes([95; 32])], - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 1, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .unwrap(), - 10_000, - ) - .await - .unwrap(); - directory - .claim_expired(session, claimant, 10_001) - .await - .unwrap() -} - -pub(crate) fn reference_identity(byte: u8, now_ms: i64) -> MutationIdentity { - MutationIdentity { - request_id: RequestId::from_bytes([byte; 16]), - issued_at_ms: now_ms, - expires_at_ms: now_ms + 60_000, - } -} - -pub(crate) fn qualification_identity(index: u64, now_ms: i64) -> MutationIdentity { - let mut request_id = [0; 16]; - request_id[..8].copy_from_slice(&index.to_be_bytes()); - MutationIdentity { - request_id: RequestId::from_bytes(request_id), - issued_at_ms: now_ms, - expires_at_ms: now_ms + 60_000, - } -} - -pub(crate) struct TypedQualificationExecutor { - pub(crate) handle: ApplicationHandle, - pub(crate) tenant: TenantId, - pub(crate) application: ApplicationId, - pub(crate) now_ms: i64, -} - -impl QualificationOperationExecutor for TypedQualificationExecutor { - type Future<'a> = Pin> + Send + 'a>>; - - fn execute<'a>(&'a mut self, operation: QualificationOperation) -> Self::Future<'a> { - let handle = self.handle.clone(); - let tenant = self.tenant; - let application = self.application; - let now_ms = self.now_ms; - Box::pin(async move { - let identity = qualification_identity(operation.index(), now_ms); - match operation.primitive() { - "sql" => { - let target = CellTarget::new( - tenant, - application, - SQL_NAMESPACE, - &partition_for_shard(0), - )?; - let sql = handle.sql::(target)?; - sql.query( - None, - SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT ?1".into(), - parameters: vec![SqlValue::Integer(operation.nonce() as i64)], - }], - }, - ) - .await - .map_err(|_| Error::Control("qualification SQL invocation failed"))?; - } - "kv" => { - let kv = handle.kv::(KV_NAMESPACE)?; - kv.atomic( - identity, - KvAtomicRequest { - scope: b"qualification-driver".to_vec(), - checks: Vec::new(), - mutations: vec![KvMutation::Put { - key: operation.index().to_be_bytes().to_vec(), - value: operation.nonce().to_be_bytes().to_vec(), - expires_at_ms: None, - }], - }, - ) - .await - .map_err(|_| Error::Control("qualification KV invocation failed"))?; - } - "blob" => { - let blob = handle.blob::()?; - blob.query( - BlobQuery::Read { - key: b"qualification/blob".to_vec(), - offset: 0, - limit: 128, - }, - None, - ) - .await - .map_err(|_| Error::Control("qualification Blob invocation failed"))?; - } - "queue" => { - let queue = handle.queue::()?; - queue - .info(0, None) - .await - .map_err(|_| Error::Control("qualification Queue invocation failed"))?; - } - "cron" => { - let cron = handle.cron::()?; - cron.get([58; 16], None) - .await - .map_err(|_| Error::Control("qualification Cron invocation failed"))?; - } - "workflow" => { - let workflow = handle.workflow::()?; - workflow - .state(b"effect-run".to_vec(), None) - .await - .map_err(|_| Error::Control("qualification Workflow invocation failed"))?; - } - "activity" => { - let activities = handle.activities::()?; - let supervisor = - crab_cell_runtime::primitives::workflow::ActivitySupervisor::new( - activities, 5_000, - )?; - supervisor - .run_once(0, None) - .await - .map_err(|_| Error::Control("qualification Activity invocation failed"))?; - } - "effects" => { - let target = CellTarget::new( - tenant, - application, - WORKFLOW_NAMESPACE, - &partition_for_shard(0), - )?; - let effects = handle.effects::(target)?; - effects - .claim( - identity, - EffectClaimRequest { - limit: 1, - lease_ms: 5_000, - }, - ) - .await - .map_err(|_| Error::Control("qualification Effects invocation failed"))?; - } - _ => return Err(Error::Control("qualification primitive is not registered")), - } - Ok(QualificationExecution::acknowledged(true)) - }) - } -} diff --git a/crates/crab-cell-app/tests/reference_application/performance.rs b/crates/crab-cell-app/tests/reference_application/performance.rs deleted file mode 100644 index baa85c916..000000000 --- a/crates/crab-cell-app/tests/reference_application/performance.rs +++ /dev/null @@ -1,400 +0,0 @@ -use super::performance_fixture::{PerfFixture, identity, item_id, now_ms}; -use crate::*; -use std::time::{Duration, Instant}; - -use crab_cell_runtime::primitives::effects::EffectRunOutcome; -use crab_cell_runtime::primitives::maintenance::{MaintenanceTickOutcome, MaintenanceTickRequest}; - -async fn measure(name: &str, iterations: usize, mut action: F) -> Vec -where - F: FnMut(usize) -> Fut, - Fut: Future, -{ - let mut samples = Vec::with_capacity(iterations); - let started = Instant::now(); - for index in 0..iterations { - let operation_started = Instant::now(); - action(index).await; - samples.push(operation_started.elapsed()); - } - let elapsed = started.elapsed(); - report_samples(name, &mut samples, elapsed); - samples -} - -pub(super) fn report_samples(name: &str, samples: &mut [Duration], elapsed: Duration) { - samples.sort_unstable(); - let iterations = samples.len(); - let percentile = |percent: usize| { - samples[(iterations * percent).div_ceil(100).saturating_sub(1)].as_secs_f64() * 1_000.0 - }; - println!( - "PERF {name}: count={iterations} elapsed_s={:.3} ops_per_s={:.2} p50_ms={:.3} p95_ms={:.3} p99_ms={:.3} max_ms={:.3}", - elapsed.as_secs_f64(), - iterations as f64 / elapsed.as_secs_f64(), - percentile(50), - percentile(95), - percentile(99), - percentile(100), - ); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "manual end-to-end performance run"] -async fn reference_primitive_end_to_end_performance() { - let fixture = PerfFixture::start(1).await; - run_reference_primitive_performance(&fixture, false, "single_runtime").await; - fixture.shutdown().await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 8)] -#[ignore = "manual three-node end-to-end performance run"] -async fn reference_three_node_fleet_end_to_end_performance() { - let fixture = PerfFixture::start(3).await; - run_reference_primitive_performance(&fixture, true, "fleet_three_runtime_mixed").await; - fixture.shutdown().await; -} - -pub(super) async fn run_reference_primitive_performance( - fixture: &PerfFixture, - concurrent: bool, - fleet_label: &str, -) { - let iterations = std::env::var("CRAB_CELL_PERF_ITERATIONS") - .ok() - .map(|value| value.parse::().unwrap()) - .unwrap_or(30); - assert!((1..=1_000).contains(&iterations)); - let sql = fixture - .typed - .sql::(fixture.sql_target.clone()) - .unwrap(); - let kv = fixture.typed.kv::(KV_NAMESPACE).unwrap(); - let blob = fixture.typed.blob::().unwrap(); - let queue = fixture.typed.queue::().unwrap(); - let workflow = fixture.typed.workflow::().unwrap(); - let activity = crab_cell_runtime::primitives::workflow::ActivitySupervisor::new( - fixture.typed.activities::().unwrap(), - 5_000, - ) - .unwrap(); - let cron = fixture.typed.cron::().unwrap(); - let peer = fixture.cron_peer(); - let sql = &sql; - let kv = &kv; - let blob = &blob; - let queue = &queue; - let workflow = &workflow; - let activity = &activity; - let cron = &cron; - let peer = &peer; - - let sql_run = measure("sql_order_insert_read", iterations, |index| async move { - let committed = sql - .batch( - identity(1, index, 0), - SqlBatch { - statements: vec![SqlStatement { - sql: "INSERT INTO orders(id, total_cents) VALUES (?1, ?2)".into(), - parameters: vec![ - SqlValue::Integer(index as i64), - SqlValue::Integer(1_999 + index as i64), - ], - }], - }, - ) - .await - .unwrap(); - let observed = sql - .query( - Some(committed.receipt), - SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT total_cents FROM orders WHERE id = ?1".into(), - parameters: vec![SqlValue::Integer(index as i64)], - }], - }, - ) - .await - .unwrap(); - assert_eq!( - observed.output[0].rows, - vec![vec![SqlValue::Integer(1_999 + index as i64)]] - ); - }); - - let kv_run = measure("kv_cart_put_get", iterations, |index| async move { - let key = format!("cart/{index}").into_bytes(); - let value = format!("{{\"sku\":\"book\",\"quantity\":{}}}", index + 1).into_bytes(); - let committed = kv - .atomic( - identity(2, index, 0), - KvAtomicRequest { - scope: b"shop".to_vec(), - checks: Vec::new(), - mutations: vec![KvMutation::Put { - key: key.clone(), - value: value.clone(), - expires_at_ms: None, - }], - }, - ) - .await - .unwrap(); - let observed = kv - .get(b"shop".to_vec(), key, Some(committed.receipt)) - .await - .unwrap(); - assert_eq!(observed.output.unwrap().value, value); - }); - - let blob_run = measure( - "blob_attachment_upload_read_32k", - iterations, - |index| async move { - let key = format!("attachments/{index}.bin").into_bytes(); - let mut payload = vec![0; 32 * 1024]; - blake3::Hasher::new() - .update(&(index as u64).to_be_bytes()) - .finalize_xof() - .fill(&mut payload); - let upload_id = item_id(index); - blob.mutate( - identity(3, index, 0), - BlobMutation::Begin { - key: key.clone(), - upload_id, - condition: BlobCondition::Missing, - content_type: Some("application/octet-stream".into()), - metadata: Vec::new(), - expires_at_ms: now_ms() + 300_000, - }, - ) - .await - .unwrap(); - blob.mutate( - identity(3, index, 1), - BlobMutation::PutPart { - key: key.clone(), - upload_id, - part_number: 1, - payload: payload.clone(), - }, - ) - .await - .unwrap(); - let committed = blob - .mutate( - identity(3, index, 2), - BlobMutation::Complete { - key: key.clone(), - upload_id, - part_count: 1, - }, - ) - .await - .unwrap(); - assert!(matches!( - committed.output, - BlobMutationOutcome::Committed { size: 32_768, .. } - )); - let observed = blob - .query( - BlobQuery::Read { - key, - offset: 0, - limit: 32 * 1024, - }, - Some(committed.receipt), - ) - .await - .unwrap(); - match observed.output { - BlobQueryResult::Read(Some(read)) => assert_eq!(read.bytes, payload), - other => panic!("unexpected Blob read: {other:?}"), - } - }, - ); - - let queue_run = measure( - "queue_notification_send_claim_ack", - iterations, - |index| async move { - let payload = format!("notify-order-{index}").into_bytes(); - let sent = queue - .send( - identity(4, index, 0), - QueueSendRequest { - producer_id: item_id(index), - payload: payload.clone(), - available_at_ms: now_ms(), - }, - ) - .await - .unwrap(); - assert!(matches!( - sent.output, - crab_cell_runtime::primitives::queue::QueueSendOutcome::Sent { .. } - )); - let claimed = queue - .claim( - identity(4, index, 1), - 0, - QueueClaimRequest { - limit: 1, - lease_ms: 5_000, - }, - ) - .await - .unwrap(); - assert_eq!(claimed.output.len(), 1); - assert_eq!(claimed.output[0].payload, payload); - assert!( - queue - .validate_claim(0, claimed.output.clone(), Some(claimed.receipt)) - .await - .unwrap() - .output - ); - let message = &claimed.output[0]; - let ack = queue - .ack(identity(4, index, 2), 0, message.message_id, message.token) - .await - .unwrap(); - assert!(matches!(ack.output, QueueLeaseOutcome::Applied { .. })); - }, - ); - - let workflow_run = measure( - "workflow_fulfillment_activity", - iterations, - |index| async move { - let workflow_id = format!("fulfillment/{index}").into_bytes(); - let started = workflow - .start( - identity(5, index, 0), - workflow_id.clone(), - b"activity".to_vec(), - ) - .await - .unwrap(); - let run_id = match started.output { - crab_cell_runtime::primitives::workflow::WorkflowOutcome::Applied { - run_id, - .. - } => run_id, - other => panic!("unexpected Workflow start: {other:?}"), - }; - let outcome = activity.run_once(0, None).await.unwrap(); - assert!( - matches!(outcome, ActivityRunOutcome::Completed { .. }), - "{outcome:?}" - ); - let observed = workflow - .state(workflow_id, None) - .await - .unwrap() - .output - .unwrap(); - assert_eq!(observed.run_id, run_id); - assert_eq!(observed.status, WorkflowStatus::Completed); - }, - ); - - let cron_run = measure( - "cron_invoice_schedule_deliver", - iterations, - |index| async move { - let schedule_id = item_id(index); - let payload = format!("invoice/{index}").into_bytes(); - let scheduled = cron - .mutate( - identity(6, index, 0), - CronMutation::Upsert { - schedule_id, - target_index: 0, - target_partition: partition_for_shard(0).to_vec(), - payload: payload.clone(), - interval_ms: 60_000, - next_due_ms: now_ms() + 5, - }, - ) - .await - .unwrap(); - tokio::time::sleep(Duration::from_millis(6)).await; - let tick = fixture - .registry - .run_maintenance_once( - fixture.client.clone(), - fixture.cron_target.clone(), - identity(6, index, 1), - MaintenanceTickRequest { - expected_commit_sequence: scheduled.receipt.commit_sequence, - }, - ) - .await - .unwrap(); - assert!(matches!( - tick.output, - MaintenanceTickOutcome::Applied { processed: 1 } - )); - let fired = cron - .get(schedule_id, Some(tick.receipt)) - .await - .unwrap() - .output; - assert!( - matches!(fired, CronQueryResult::Get(Some(schedule)) if schedule.occurrence == 1) - ); - let delivered = fixture - .registry - .run_effect_once( - fixture.client.clone(), - fixture.cron_target.clone(), - (*peer).clone(), - 5_000, - ) - .await - .unwrap(); - assert!( - matches!(delivered, EffectRunOutcome::Delivered { .. }), - "{delivered:?}" - ); - let observed = sql.query(None, SqlBatch { statements: vec![SqlStatement { - sql: "SELECT payload FROM invoice_receipts WHERE schedule_id = ?1 AND occurrence = 1".into(), - parameters: vec![SqlValue::Blob(schedule_id.to_vec())], - }] }).await.unwrap(); - assert_eq!(observed.output[0].rows, vec![vec![SqlValue::Blob(payload)]]); - }, - ); - - if concurrent { - let started = Instant::now(); - let (sql, kv, blob, queue, workflow, cron) = - tokio::join!(sql_run, kv_run, blob_run, queue_run, workflow_run, cron_run); - let elapsed = started.elapsed(); - let mut samples = [sql, kv, blob, queue, workflow, cron].concat(); - samples.sort_unstable(); - let percentile = |percent: usize| { - samples[(samples.len() * percent).div_ceil(100).saturating_sub(1)].as_secs_f64() - * 1_000.0 - }; - println!( - "PERF {fleet_label}: count={} elapsed_s={:.3} ops_per_s={:.2} p50_ms={:.3} p95_ms={:.3} p99_ms={:.3} max_ms={:.3}", - samples.len(), - elapsed.as_secs_f64(), - samples.len() as f64 / elapsed.as_secs_f64(), - percentile(50), - percentile(95), - percentile(99), - percentile(100), - ); - } else { - sql_run.await; - kv_run.await; - blob_run.await; - queue_run.await; - workflow_run.await; - cron_run.await; - } -} diff --git a/crates/crab-cell-app/tests/reference_application/performance_fixture.rs b/crates/crab-cell-app/tests/reference_application/performance_fixture.rs deleted file mode 100644 index de65919ad..000000000 --- a/crates/crab-cell-app/tests/reference_application/performance_fixture.rs +++ /dev/null @@ -1,531 +0,0 @@ -use super::fleet::{balancer_round_trip, peer_round_trip, start_peer_servers}; -use crate::*; -use std::{ - collections::HashMap, - net::SocketAddr, - sync::Mutex, - time::{Duration, SystemTime}, -}; - -use crab_cell_host::{CellNode, CellNodeBuilder}; -use crab_cell_runtime::fleet::telemetry::CellTelemetry; -use crab_cell_runtime::node::lease::NodeLeaseGuard; -use crab_cell_runtime::node::log::DurabilitySource; -use crab_cell_runtime::peer::{ - EffectPeerClient, PeerAuthorizer, PeerCellResolver, PeerDispatcher, PeerPrincipal, - PeerRoundTrip, PeerSigner, PeerVerifier, VerifiedPeerRequest, -}; -use tokio_util::sync::CancellationToken; - -pub(super) fn now_ms() -> i64 { - i64::try_from( - SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap() -} - -pub(super) fn identity(phase: u8, index: usize, step: u8) -> MutationIdentity { - let mut bytes = [0; 16]; - bytes[0] = phase; - bytes[1] = step; - bytes[8..].copy_from_slice(&(index as u64).to_be_bytes()); - let now = now_ms(); - MutationIdentity { - request_id: RequestId::from_bytes(bytes), - issued_at_ms: now, - expires_at_ms: now + 300_000, - } -} - -pub(super) fn item_id(index: usize) -> [u8; 16] { - let mut id = [0; 16]; - id[8..].copy_from_slice(&(index as u64).to_be_bytes()); - id -} - -pub(super) fn install_sql_tables(tx: &crab_ltx::rusqlite::Transaction<'_>) -> Result<()> { - tx.execute_batch( - "CREATE TABLE orders(id INTEGER PRIMARY KEY, total_cents INTEGER NOT NULL); \ - CREATE TABLE invoice_receipts(schedule_id BLOB NOT NULL, occurrence INTEGER NOT NULL, payload BLOB NOT NULL, PRIMARY KEY(schedule_id, occurrence));", - )?; - Ok(()) -} - -pub(super) type Schema = for<'a> fn(&crab_ltx::rusqlite::Transaction<'a>) -> Result<()>; - -pub(super) fn perf_cells() -> [(NamespaceId, CatalogRole, &'static str, u8, Schema); 7] { - [ - ( - SQL_NAMESPACE, - CatalogRole::Sql, - SQL_MODULE, - 40, - install_sql_tables, - ), - ( - KV_NAMESPACE, - CatalogRole::Kv, - KV_MODULE, - 41, - install_kv_schema, - ), - ( - BLOB_NAMESPACE, - CatalogRole::Blob, - BLOB_MODULE, - 42, - install_blob_schema, - ), - ( - QUEUE_NAMESPACE, - CatalogRole::Queue, - QUEUE_MODULE, - 43, - install_queue_schema, - ), - ( - DEAD_LETTER_NAMESPACE, - CatalogRole::Queue, - DEAD_LETTER_MODULE, - 44, - install_queue_schema, - ), - ( - CRON_NAMESPACE, - CatalogRole::Cron, - CRON_MODULE, - 45, - install_cron_schema, - ), - ( - WORKFLOW_NAMESPACE, - CatalogRole::Workflow, - WORKFLOW_MODULE, - 46, - install_workflow_schema, - ), - ] -} - -pub(super) fn owner_routes( - tenant: TenantId, - application: ApplicationId, - owners: [SocketAddr; 3], -) -> HashMap { - perf_cells() - .into_iter() - .enumerate() - .map(|(index, (namespace, _, _, _, _))| { - let target = - CellTarget::new(tenant, application, namespace, &partition_for_shard(0)).unwrap(); - (target.cell_id(), owners[index % owners.len()]) - }) - .collect() -} - -pub(super) struct PerfFixture { - _directory: tempfile::TempDir, - pub(super) nodes: Vec, - pub(super) layout: Option, - leases: Vec, - pub(super) round_trip: Option>, - pub(super) owned_handles: Vec>, - pub(super) durability: Vec>, - pub(super) typed: ApplicationHandle, - pub(super) registry: Arc, - pub(super) client: CellClient, - pub(super) sql_target: CellTarget, - pub(super) cron_target: CellTarget, - peer: EffectPeerClient, - servers: Vec>, -} - -#[derive(Default)] -pub(super) struct DurabilityRecorder(Mutex>); - -impl DurabilityRecorder { - pub(super) fn object_waits(&self) -> Vec { - self.0 - .lock() - .unwrap() - .iter() - .filter_map(|(source, waited)| (*source == DurabilitySource::Object).then_some(*waited)) - .collect() - } -} - -impl CellTelemetry for DurabilityRecorder { - fn durability_proof(&self, source: DurabilitySource, waited: Duration) { - self.0.lock().unwrap().push((source, waited)); - } -} - -impl PerfFixture { - pub(super) async fn start(nodes: usize) -> Self { - Self::start_with_successor(nodes, None).await - } - - pub(super) async fn start_with_store( - nodes: usize, - store: Store, - root: object_store::path::Path, - ) -> Self { - Self::start_configured(nodes, None, store, root).await - } - - pub(super) async fn start_with_successor( - nodes: usize, - successor: Option>, - ) -> Self { - Self::start_configured( - nodes, - successor, - Store::new(Arc::new(InMemory::new())), - object_store::path::Path::from("reference-performance"), - ) - .await - } - - pub(super) async fn start_configured( - nodes: usize, - successor: Option>, - store: Store, - root: object_store::path::Path, - ) -> Self { - assert!(nodes == 1 || nodes == 3); - assert!(successor.is_none() || nodes == 3); - let application = Arc::new(compiled()); - let registry = application.registry(); - let tenant = TenantId::from_bytes([81; 16]); - let application_id = ApplicationId::from_bytes([82; 16]); - let layout = CellStorageLayout::new(store.clone(), root, *application_id.as_bytes()); - let directory = tempfile::TempDir::new().unwrap(); - let mut leases = Vec::new(); - let mut durability = Vec::new(); - let hosts = (0..nodes) - .map(|node| { - let node_application = if node == 0 { - successor.as_ref().unwrap_or(&application) - } else { - &application - }; - let host = CellNodeBuilder::new(Arc::clone(node_application)) - .with_runtime(SqlWorkerPool::new(4, 32).unwrap(), 64 * 1024 * 1024) - .with_session(node_session(node)) - .with_replica_host(reference_host()) - .build() - .unwrap(); - let recorder = Arc::new(DurabilityRecorder::default()); - host.install_telemetry(recorder.clone()).unwrap(); - durability.push(recorder); - host.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - host.install_node_lease(lease.clone()).unwrap(); - leases.push(lease); - host - }) - .collect::>(); - let cells = perf_cells(); - let mut handles = Vec::new(); - let mut owned = vec![Vec::new(); nodes]; - for (cell_index, (namespace, role, module, incarnation, schema)) in - cells.into_iter().enumerate() - { - let node = cell_index % nodes; - let handle = bootstrap_reference_cell( - &hosts[node].runtime(), - ®istry, - &layout, - &directory, - tenant, - application_id, - node_session(node), - namespace, - role, - module, - incarnation, - schema, - ) - .await - .unwrap(); - owned[node].push(handle.clone()); - handles.push(handle); - } - let sql_handle = handles[0].clone(); - let signer = Arc::new(PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - )); - let principal = PeerPrincipal { - issuer: "reference-performance".into(), - subject: "fleet-driver".into(), - actions: vec![ - "cell.read".into(), - "cell.write".into(), - "reference.cron.deliver".into(), - ], - }; - let verifier = Arc::new(PeerVerifier::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - signer.verifying_key(), - )); - let (client, peer, servers, round_trip) = if nodes == 1 { - let client = CellClient::local_many(Arc::clone(®istry), handles).unwrap(); - let dispatcher = Arc::new(PeerDispatcher::new( - Arc::clone(®istry), - Arc::new(SqlResolver { - target: CellTarget::new( - tenant, - application_id, - SQL_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(), - handle: sql_handle, - }), - Arc::new(CronAuthorizer), - )); - let peer = EffectPeerClient::new( - signer, - principal, - Arc::new(Loopback { - verifier, - dispatcher, - }), - ); - (client, peer, Vec::new(), None) - } else { - // A mixed release fleet must dispatch with each host's executable - // registry; sharing the driver's registry would mask rollout bugs. - let dispatchers = hosts - .iter() - .zip(owned.iter()) - .map(|(host, handles)| (host.application().registry(), handles.clone())) - .collect(); - let (round_trip, servers) = start_peer_servers(verifier, dispatchers).await; - let client = CellClient::peer( - Arc::clone(®istry), - Arc::clone(&signer), - principal.clone(), - Arc::clone(&round_trip), - ); - let peer = EffectPeerClient::new(signer, principal, Arc::clone(&round_trip)); - (client, peer, servers, Some(round_trip)) - }; - let binding_node = usize::from(successor.is_some()); - let typed = hosts[binding_node] - .application_handle::(client.clone(), tenant, application_id) - .unwrap() - .with_blob_artifact_store(BlobArtifactStore::new(store)); - let sql_target = CellTarget::new( - tenant, - application_id, - SQL_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - let cron_target = CellTarget::new( - tenant, - application_id, - CRON_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - Self { - _directory: directory, - nodes: hosts, - layout: Some(layout), - leases, - round_trip, - owned_handles: owned, - durability, - typed, - registry, - client, - sql_target, - cron_target, - peer, - servers, - } - } - - pub(super) fn cron_peer(&self) -> EffectPeerClient { - self.peer.clone() - } - - pub(super) fn lose_owner(&self, node: usize) { - self.leases[node].fence(); - self.servers[node].abort(); - } - - pub(super) fn from_processes( - directory: tempfile::TempDir, - store: Store, - owners: [SocketAddr; 3], - balancer: Option, - ) -> Self { - let application = Arc::new(compiled()); - let registry = application.registry(); - let tenant = TenantId::from_bytes([81; 16]); - let application_id = ApplicationId::from_bytes([82; 16]); - let round_trip = if let Some(address) = balancer { - balancer_round_trip(address) - } else { - peer_round_trip(owner_routes(tenant, application_id, owners)) - }; - let signer = Arc::new(PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - )); - let principal = PeerPrincipal { - issuer: "reference-performance".into(), - subject: "fleet-driver".into(), - actions: vec![ - "cell.read".into(), - "cell.write".into(), - "reference.cron.deliver".into(), - ], - }; - let client = CellClient::peer( - Arc::clone(®istry), - Arc::clone(&signer), - principal.clone(), - Arc::clone(&round_trip), - ); - let peer = EffectPeerClient::new(signer, principal, round_trip); - let typed = ApplicationHandle::new(client.clone(), application, tenant, application_id) - .unwrap() - .with_blob_artifact_store(BlobArtifactStore::new(store)); - let sql_target = CellTarget::new( - tenant, - application_id, - SQL_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - let cron_target = CellTarget::new( - tenant, - application_id, - CRON_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - Self { - _directory: directory, - nodes: Vec::new(), - layout: None, - leases: Vec::new(), - round_trip: None, - owned_handles: Vec::new(), - durability: Vec::new(), - typed, - registry, - client, - sql_target, - cron_target, - peer, - servers: Vec::new(), - } - } - - pub(super) async fn shutdown(&self) { - for server in &self.servers { - server.abort(); - } - for (index, node) in self.nodes.iter().enumerate() { - match node.shutdown().await { - Ok(()) => {} - // The simulated lost owner closes locally but cannot release - // authority after its lease is fenced. - Err(Error::Fenced) if self.leases[index].check().is_err() => {} - Err(error) => panic!("CellNode shutdown failed: {error}"), - } - } - } -} - -pub(super) fn node_session(node: usize) -> crab_cell_runtime::SessionId { - crab_cell_runtime::SessionId::from_bytes([24 + node as u8; 16]) -} - -struct SqlResolver { - target: CellTarget, - handle: CellHandle, -} - -impl PeerCellResolver for SqlResolver { - fn resolve( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>> { - let allowed = target == self.target; - let handle = self.handle.clone(); - Box::pin(async move { - if allowed { - Ok(handle) - } else { - Err(Error::CellNotActive) - } - }) - } -} - -struct CronAuthorizer; - -impl PeerAuthorizer for CronAuthorizer { - fn authorize(&self, request: &VerifiedPeerRequest) -> Result<()> { - if request.permits("reference.cron.deliver") { - Ok(()) - } else { - Err(Error::PeerAuthorization("missing cron delivery action")) - } - } -} - -struct Loopback { - verifier: Arc, - dispatcher: Arc, -} - -impl PeerRoundTrip for Loopback { - fn send( - &self, - target: CellTarget, - request: Vec, - _remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let verifier = Arc::clone(&self.verifier); - let dispatcher = Arc::clone(&self.dispatcher); - Box::pin(async move { - let verified = verifier.verify(&request, now_ms())?; - if verified.target() != &target { - return Err(Error::Peer("round trip target changed")); - } - dispatcher.dispatch_bytes(&verified, now_ms()).await - }) - } -} - -// One explicit provider path for the public-host and process qualification lanes. -pub(super) fn rustfs_store() -> Store { - let required = |name: &str| std::env::var(name).unwrap_or_else(|_| panic!("missing {name}")); - crab_storage::build_explicit_store( - &required("CRAB_CELL_TEST_BUCKET"), - crab_storage::ObjectStoreCredentials::Aws { - access_key_id: required("AWS_ACCESS_KEY_ID"), - secret_access_key: required("AWS_SECRET_ACCESS_KEY"), - session_token: None, - region: "us-east-1".into(), - }, - Some(&required("CRAB_CELL_TEST_ENDPOINT")), - true, - ) - .unwrap() -} diff --git a/crates/crab-cell-app/tests/reference_application/primitives.rs b/crates/crab-cell-app/tests/reference_application/primitives.rs deleted file mode 100644 index 2d8c5ff43..000000000 --- a/crates/crab-cell-app/tests/reference_application/primitives.rs +++ /dev/null @@ -1,765 +0,0 @@ -//! Every reference primitive driven through one local router. - -use crate::*; - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn typed_application_executes_every_primitive_through_a_local_router() { - let application = Arc::new(compiled()); - let tenant = TenantId::from_bytes([31; 16]); - let application_id = ApplicationId::from_bytes([32; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new( - store.clone(), - object_store::path::Path::from("reference-primitive-qualification"), - *application_id.as_bytes(), - ); - let directory = tempfile::TempDir::new().unwrap(); - let session = crab_cell_runtime::SessionId::from_bytes([24; 16]); - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(4, 32).unwrap(), - 64 * 1024 * 1024, - session, - reference_host(), - ) - .unwrap(); - let registry = application.registry(); - let handles = vec![ - bootstrap_reference_cell( - &runtime, - ®istry, - &layout, - &directory, - tenant, - application_id, - session, - SQL_NAMESPACE, - CatalogRole::Sql, - SQL_MODULE, - 40, - |_| Ok(()), - ) - .await - .unwrap(), - bootstrap_reference_cell( - &runtime, - ®istry, - &layout, - &directory, - tenant, - application_id, - session, - KV_NAMESPACE, - CatalogRole::Kv, - KV_MODULE, - 41, - install_kv_schema, - ) - .await - .unwrap(), - bootstrap_reference_cell( - &runtime, - ®istry, - &layout, - &directory, - tenant, - application_id, - session, - BLOB_NAMESPACE, - CatalogRole::Blob, - BLOB_MODULE, - 42, - install_blob_schema, - ) - .await - .unwrap(), - bootstrap_reference_cell( - &runtime, - ®istry, - &layout, - &directory, - tenant, - application_id, - session, - QUEUE_NAMESPACE, - CatalogRole::Queue, - QUEUE_MODULE, - 43, - install_queue_schema, - ) - .await - .unwrap(), - bootstrap_reference_cell( - &runtime, - ®istry, - &layout, - &directory, - tenant, - application_id, - session, - DEAD_LETTER_NAMESPACE, - CatalogRole::Queue, - DEAD_LETTER_MODULE, - 44, - install_queue_schema, - ) - .await - .unwrap(), - bootstrap_reference_cell( - &runtime, - ®istry, - &layout, - &directory, - tenant, - application_id, - session, - CRON_NAMESPACE, - CatalogRole::Cron, - CRON_MODULE, - 45, - install_cron_schema, - ) - .await - .unwrap(), - bootstrap_reference_cell( - &runtime, - ®istry, - &layout, - &directory, - tenant, - application_id, - session, - WORKFLOW_NAMESPACE, - CatalogRole::Workflow, - WORKFLOW_MODULE, - 46, - install_workflow_schema, - ) - .await - .unwrap(), - ]; - let client = CellClient::local_many(registry.clone(), handles).unwrap(); - let typed = ApplicationHandle::::new( - client, - Arc::clone(&application), - tenant, - application_id, - ) - .unwrap() - .with_blob_artifact_store(BlobArtifactStore::new(store.clone())); - let now_ms = i64::try_from( - std::time::SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap(); - - let sql_target = CellTarget::new( - tenant, - application_id, - SQL_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - let queue_target = CellTarget::new( - tenant, - application_id, - QUEUE_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - let wrong_command = typed - .command::>( - &sql_target, - reference_identity(49, now_ms), - KvAtomicRequest { - scope: b"wrong-module".to_vec(), - checks: Vec::new(), - mutations: Vec::new(), - }, - ) - .await; - assert!(matches!( - wrong_command, - Err(InvocationError::NotStarted(Error::Registry( - "namespace module differs from capability" - ))) - )); - let wrong_prepared = typed - .prepare_command::>( - &sql_target, - reference_identity(50, now_ms), - KvAtomicRequest { - scope: b"wrong-module".to_vec(), - checks: Vec::new(), - mutations: Vec::new(), - }, - ) - .await; - assert!(matches!( - wrong_prepared, - Err(InvocationError::NotStarted(Error::Registry( - "namespace module differs from capability" - ))) - )); - let wrong_query = typed - .query::>( - &sql_target, - None, - KvGetRequest { - scope: b"wrong-module".to_vec(), - key: b"key".to_vec(), - }, - ) - .await; - assert!(matches!( - wrong_query, - Err(InvocationError::NotStarted(Error::Registry( - "namespace module differs from capability" - ))) - )); - assert!(typed.sql::(queue_target).is_err()); - assert!(typed.kv::(SQL_NAMESPACE).is_err()); - assert!( - typed - .effects::(sql_target.clone()) - .is_err() - ); - let foreign_target = CellTarget::new( - TenantId::from_bytes([99; 16]), - application_id, - SQL_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - assert!(typed.sql::(foreign_target).is_err()); - let invalid_partition = CellTarget::new( - tenant, - application_id, - SQL_NAMESPACE, - &partition_for_shard(1), - ) - .unwrap(); - assert!(typed.sql::(invalid_partition).is_err()); - let kv_target = CellTarget::new( - tenant, - application_id, - KV_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - assert!(typed.effects::(kv_target).is_err()); - let sql = typed.sql::(sql_target.clone()).unwrap(); - let sql_result = sql - .batch( - reference_identity(50, now_ms), - SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT ?1".into(), - parameters: vec![SqlValue::Integer(11)], - }], - }, - ) - .await - .unwrap(); - assert_eq!(sql_result.output[0].rows.len(), 1); - - let kv = typed.kv::(KV_NAMESPACE).unwrap(); - kv.atomic( - reference_identity(51, now_ms), - KvAtomicRequest { - scope: b"qualification".to_vec(), - checks: Vec::new(), - mutations: vec![KvMutation::Put { - key: b"key".to_vec(), - value: b"value".to_vec(), - expires_at_ms: None, - }], - }, - ) - .await - .unwrap(); - let kv_value = kv - .get(b"qualification".to_vec(), b"key".to_vec(), None) - .await - .unwrap(); - assert_eq!(kv_value.output.unwrap().value, b"value"); - - let blob = typed.blob::().unwrap(); - let blob_key = b"qualification/blob".to_vec(); - let upload_id = [52; 16]; - blob.mutate( - reference_identity(52, now_ms), - BlobMutation::Begin { - key: blob_key.clone(), - upload_id, - condition: BlobCondition::Missing, - content_type: None, - metadata: Vec::new(), - expires_at_ms: now_ms + 60_000, - }, - ) - .await - .unwrap(); - blob.mutate( - reference_identity(53, now_ms), - BlobMutation::PutPart { - key: blob_key.clone(), - upload_id, - part_number: 1, - payload: b"blob-value".to_vec(), - }, - ) - .await - .unwrap(); - let completed = blob - .mutate( - reference_identity(54, now_ms), - BlobMutation::Complete { - key: blob_key.clone(), - upload_id, - part_count: 1, - }, - ) - .await - .unwrap(); - assert!(matches!( - completed.output, - BlobMutationOutcome::Committed { .. } - )); - let blob_read = blob - .query( - BlobQuery::Read { - key: blob_key, - offset: 0, - limit: 128, - }, - None, - ) - .await - .unwrap(); - assert!(matches!(blob_read.output, BlobQueryResult::Read(Some(_)))); - - let queue = typed.queue::().unwrap(); - let sent = queue - .send( - reference_identity(55, now_ms), - QueueSendRequest { - producer_id: [55; 16], - payload: b"queue-value".to_vec(), - available_at_ms: now_ms, - }, - ) - .await - .unwrap(); - let claimed = queue - .claim( - reference_identity(56, now_ms), - 0, - QueueClaimRequest { - limit: 1, - lease_ms: 5_000, - }, - ) - .await - .unwrap(); - assert_eq!(claimed.output.len(), 1); - assert!( - typed - .with_read_policy(crab_cell_runtime::client::ReadPolicy::Replica) - .queue::() - .unwrap() - .validate_claim(0, claimed.output.clone(), Some(claimed.receipt)) - .await - .unwrap() - .output - ); - let message = &claimed.output[0]; - let acked = queue - .ack( - reference_identity(57, now_ms), - 0, - message.message_id, - message.token, - ) - .await - .unwrap(); - assert!(matches!(acked.output, QueueLeaseOutcome::Applied { .. })); - assert!(matches!( - sent.output, - crab_cell_runtime::primitives::queue::QueueSendOutcome::Sent { .. } - )); - - let cron = typed.cron::().unwrap(); - let schedule_id = [58; 16]; - cron.mutate( - reference_identity(58, now_ms), - CronMutation::Upsert { - schedule_id, - target_index: 0, - target_partition: partition_for_shard(0).to_vec(), - payload: b"cron-value".to_vec(), - interval_ms: 1_000, - next_due_ms: now_ms + 1_000, - }, - ) - .await - .unwrap(); - assert!(matches!( - cron.get(schedule_id, None).await.unwrap().output, - CronQueryResult::Get(Some(_)) - )); - - let workflow = typed.workflow::().unwrap(); - let activity_workflow_id = b"activity-run".to_vec(); - let activity_started = workflow - .start( - reference_identity(59, now_ms), - activity_workflow_id.clone(), - b"activity".to_vec(), - ) - .await - .unwrap(); - let activity_run_id = match activity_started.output { - crab_cell_runtime::primitives::workflow::WorkflowOutcome::Applied { run_id, .. } => run_id, - outcome => panic!("unexpected activity start outcome: {outcome:?}"), - }; - let activity = crab_cell_runtime::primitives::workflow::ActivitySupervisor::new( - typed - .with_read_policy(crab_cell_runtime::client::ReadPolicy::Replica) - .activities::() - .unwrap(), - 5_000, - ) - .unwrap(); - assert!(matches!( - activity.run_once(0, None).await.unwrap(), - ActivityRunOutcome::Completed { .. } - )); - let activity_state = workflow - .state(activity_workflow_id, None) - .await - .unwrap() - .output - .unwrap(); - assert_eq!(activity_state.run_id, activity_run_id); - assert_eq!(activity_state.status, WorkflowStatus::Completed); - - let effect_workflow_id = b"effect-run".to_vec(); - workflow - .start( - reference_identity(60, now_ms), - effect_workflow_id, - b"effect".to_vec(), - ) - .await - .unwrap(); - let effects = typed - .with_read_policy(crab_cell_runtime::client::ReadPolicy::Replica) - .effects::( - CellTarget::new( - tenant, - application_id, - WORKFLOW_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(), - ) - .unwrap(); - let claims = effects - .claim( - reference_identity(61, now_ms), - EffectClaimRequest { - limit: 1, - lease_ms: 5_000, - }, - ) - .await - .unwrap(); - assert_eq!(claims.output.len(), 1); - assert!( - effects - .validate(claims.output.clone(), claims.receipt) - .await - .unwrap() - .output - ); - let effect = &claims.output[0]; - let acknowledged = effects - .ack( - reference_identity(62, now_ms), - effect.clone(), - b"effect-result".to_vec(), - ) - .await - .unwrap(); - assert_eq!(acknowledged.output, EffectLeaseOutcome::Delivered); - - let profile = QualificationProfile::new("typed-smoke".into(), 1, 1, 1, 5_000).unwrap(); - let workload = QualificationWorkload::generate_with_size(&profile, 91, 1, 16, 1).unwrap(); - let mut executor = TypedQualificationExecutor { - handle: typed.clone(), - tenant, - application: application_id, - now_ms, - }; - let summary = workload.run(&mut executor).await.unwrap(); - assert_eq!(summary.operations(), 16); - assert!( - summary - .primitive_counts() - .iter() - .all(|counts| counts.attempted() > 0 && counts.verified() > 0) - ); - assert!( - summary - .metrics() - .unwrap() - .iter() - .any(|metric| { metric.name() == "p99_latency_ms" && metric.unit() == "ms" }) - ); - - // Drop the first owner and its local SQLite sources. The next owner must - // recover every declared namespace from the published roots before the - // same typed application handle is allowed to continue. - drop(activity); - drop(effects); - drop(workflow); - drop(cron); - drop(queue); - drop(blob); - drop(kv); - drop(sql); - drop(typed); - // The workload executor owns the last cloned local transport. Release it - // before deleting the source files so the crashed owner cannot continue a - // background compaction against the torn-down SQLite paths. - drop(executor); - drop(runtime); - drop(directory); - - let restored_directory = tempfile::TempDir::new().unwrap(); - let takeover_session = crab_cell_runtime::SessionId::from_bytes([70; 16]); - let takeover_runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(4, 32).unwrap(), - 64 * 1024 * 1024, - takeover_session, - reference_host(), - ) - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new(layout.clone(), tenant); - seed_reference_session(&layout, session).await; - let mut restored_handles = Vec::new(); - for (namespace, module, incarnation_byte) in [ - (SQL_NAMESPACE, SQL_MODULE, 40_u8), - (KV_NAMESPACE, KV_MODULE, 41_u8), - (BLOB_NAMESPACE, BLOB_MODULE, 42_u8), - (QUEUE_NAMESPACE, QUEUE_MODULE, 43_u8), - (DEAD_LETTER_NAMESPACE, DEAD_LETTER_MODULE, 44_u8), - (CRON_NAMESPACE, CRON_MODULE, 45_u8), - (WORKFLOW_NAMESPACE, WORKFLOW_MODULE, 46_u8), - ] { - let target = - CellTarget::new(tenant, application_id, namespace, &partition_for_shard(0)).unwrap(); - let observed = authority.load(target.cell_id()).await.unwrap().unwrap(); - let restored = takeover_runtime - .takeover_restored( - catalog.lookup(target.cell_id()).await.unwrap().unwrap(), - CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *IncarnationId::from_bytes([incarnation_byte; 16]).as_bytes(), - Limits::default(), - ) - .unwrap(), - authority.clone(), - observed, - fence_reference_session(&layout, session, takeover_session) - .await - .direct_takeover() - .unwrap(), - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - layout.clone(), - Limits::default(), - ), - restored_directory - .path() - .join(format!("{module}-takeover.sqlite")), - Owner { - session: takeover_session, - endpoint: "https://reference-takeover.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!(restored.cell_id(), target.cell_id()); - restored_handles.push(restored); - } - let restored_client = CellClient::local_many(registry.clone(), restored_handles).unwrap(); - let restored = ApplicationHandle::::new( - restored_client, - application, - tenant, - application_id, - ) - .unwrap() - .with_blob_artifact_store(BlobArtifactStore::new(store)); - restored - .sql::(sql_target.clone()) - .unwrap() - .query( - None, - SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT 11".into(), - parameters: Vec::new(), - }], - }, - ) - .await - .unwrap(); - assert_eq!( - restored - .kv::(KV_NAMESPACE) - .unwrap() - .get(b"qualification".to_vec(), b"key".to_vec(), None) - .await - .unwrap() - .output - .unwrap() - .value, - b"value" - ); - restored - .kv::(KV_NAMESPACE) - .unwrap() - .atomic( - reference_identity(100, now_ms), - KvAtomicRequest { - scope: b"qualification".to_vec(), - checks: Vec::new(), - mutations: vec![KvMutation::Put { - key: b"after-recovery".to_vec(), - value: b"ok".to_vec(), - expires_at_ms: None, - }], - }, - ) - .await - .unwrap(); - let recovered_blob = restored - .blob::() - .unwrap() - .query( - BlobQuery::Read { - key: b"qualification/blob".to_vec(), - offset: 0, - limit: 128, - }, - None, - ) - .await - .unwrap(); - match recovered_blob.output { - BlobQueryResult::Read(Some(read)) => assert_eq!(read.bytes, b"blob-value"), - other => panic!("unexpected recovered blob: {other:?}"), - } - let restored_queue = restored.queue::().unwrap(); - restored_queue - .send( - reference_identity(101, now_ms), - QueueSendRequest { - producer_id: [101; 16], - payload: b"after-recovery".to_vec(), - available_at_ms: now_ms, - }, - ) - .await - .unwrap(); - assert!(restored_queue.info(0, None).await.unwrap().output.ready >= 1); - restored - .cron::() - .unwrap() - .get([58; 16], None) - .await - .unwrap(); - let restored_workflow = restored.workflow::().unwrap(); - assert_eq!( - restored_workflow - .state(b"activity-run".to_vec(), None) - .await - .unwrap() - .output - .unwrap() - .status, - WorkflowStatus::Completed - ); - let activity_workflow_id = b"activity-after-recovery".to_vec(); - restored_workflow - .start( - reference_identity(102, now_ms), - activity_workflow_id.clone(), - b"activity".to_vec(), - ) - .await - .unwrap(); - let restored_activity = crab_cell_runtime::primitives::workflow::ActivitySupervisor::new( - restored.activities::().unwrap(), - 5_000, - ) - .unwrap(); - assert!(matches!( - restored_activity.run_once(0, None).await.unwrap(), - ActivityRunOutcome::Completed { .. } - )); - assert_eq!( - restored_workflow - .state(activity_workflow_id, None) - .await - .unwrap() - .output - .unwrap() - .status, - WorkflowStatus::Completed - ); - restored_workflow - .start( - reference_identity(103, now_ms), - b"effect-after-recovery".to_vec(), - b"effect".to_vec(), - ) - .await - .unwrap(); - let restored_effects = restored - .effects::( - CellTarget::new( - tenant, - application_id, - WORKFLOW_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(), - ) - .unwrap(); - let restored_claims = restored_effects - .claim( - reference_identity(104, now_ms), - EffectClaimRequest { - limit: 1, - lease_ms: 5_000, - }, - ) - .await - .unwrap(); - assert_eq!(restored_claims.output.len(), 1); - restored_effects - .ack( - reference_identity(105, now_ms), - restored_claims.output[0].clone(), - b"recovered-effect".to_vec(), - ) - .await - .unwrap(); - takeover_runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-app/tests/reference_application/process_node.rs b/crates/crab-cell-app/tests/reference_application/process_node.rs deleted file mode 100644 index 55933e017..000000000 --- a/crates/crab-cell-app/tests/reference_application/process_node.rs +++ /dev/null @@ -1,189 +0,0 @@ -//! Public host and authoritative session lifecycle for the process fixture. - -use super::performance_fixture::{DurabilityRecorder, node_session, now_ms}; -use crate::*; -use crab_cell_host::{CellNode, CellNodeBuilder}; -use crab_cell_runtime::node::lease::NodeLeaseGuard; -use std::time::Duration; -use tokio_util::sync::CancellationToken; - -pub(super) fn directory(layout: &CellStorageLayout, registry: &Registry) -> NodeDirectory { - NodeDirectory::new( - layout.clone(), - Digest::from_bytes([90; 32]), - Digest::from_bytes([91; 32]), - registry.release_digest(), - ) -} - -pub(super) async fn start( - node: usize, - application: Arc, - layout: &CellStorageLayout, - root: &std::path::Path, - endpoint: String, -) -> ( - CellNode, - Arc, - crab_cell_host::read_replicas::ReadReplicaManager, -) { - let registry = application.registry(); - // Keep writer admission at 32 Cells while charging both the old and new - // immutable snapshots during a refresh under the same node ledger. - let pool = SqlWorkerPool::new(4, 32) - .unwrap() - .with_native_memory_limit(32 << 20) - .unwrap(); - let host = CellNodeBuilder::new(application) - .with_runtime(pool, 64 * 1024 * 1024) - .with_session(node_session(node)) - .with_replica_host(reference_host()) - .build() - .unwrap(); - let durability = Arc::new(DurabilityRecorder::default()); - host.install_telemetry(durability.clone()).unwrap(); - let directory = directory(layout, ®istry); - let signer = SigningKey::from_bytes(&[93; 32]); - let advertisement = move |now: i64, progress| { - NodeAdvertisement::sign( - NodeId::from_bytes(*node_session(node).as_bytes()), - node_session(node), - endpoint.clone(), - Digest::from_bytes([90; 32]), - Digest::from_bytes([94; 32]), - Digest::from_bytes([91; 32]), - registry.release_digest(), - &signer, - progress, - now, - now + 15_000, - registry.module_digests(), - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 64 * 1024 * 1024, - free_disk_bytes: 1 << 30, - job_credits: 32, - ..NodeCapacity::default() - }, - ) - }; - let now = now_ms(); - let mut observed = directory - .create(advertisement(now, 1).unwrap(), now) - .await - .unwrap(); - let lease = NodeLeaseGuard::new(now_ms(), observed.advertisement().expires_at_ms()).unwrap(); - let shutdown = CancellationToken::new(); - let tasks = host - .install_task_group(CancellationToken::new(), shutdown.clone()) - .unwrap(); - host.install_node_lease_for_startup(lease.clone()).unwrap(); - let (renewed, first_renewal) = tokio::sync::oneshot::channel(); - tasks - .spawn_lease_maintenance(async move { - let mut renewed = Some(renewed); - let result: Result<()> = async { - loop { - tokio::select! { - () = shutdown.cancelled() => break, - () = tokio::time::sleep(Duration::from_secs(5)) => {} - } - let now = now_ms(); - let next = advertisement(now, observed.advertisement().progress() + 1)?; - // Finish the conditional write before observing shutdown so - // withdrawal always uses the latest acknowledged generation. - observed = tokio::time::timeout( - lease.remaining(), - directory.refresh(&observed, next, now), - ) - .await - .map_err(|_| Error::Deadline)??; - lease.renew(now_ms(), observed.advertisement().expires_at_ms())?; - if let Some(renewed) = renewed.take() { - let _ = renewed.send(()); - } - } - tokio::time::timeout( - Duration::from_secs(20), - directory.withdraw(&observed, now_ms()), - ) - .await - .map_err(|_| Error::Deadline)??; - println!( - "PERF node_{node}_session_withdrawn: generation={}", - observed.advertisement().generation() - ); - Ok(()) - } - .await; - lease.fence(); - result - }) - .unwrap(); - // Even a short smoke must exercise provider-backed renewal before it can - // report readiness; successful drain then proves withdrawal of that version. - first_renewal.await.unwrap(); - let readers = host - .install_read_replicas( - layout.clone(), - self::directory(layout, &host.application().registry()), - root.join("readers"), - Limits::default(), - ) - .unwrap(); - host.install_read_replica_recruitment( - crab_cell_runtime::cell::application::ApplicationIdentity::new( - TenantId::from_bytes([81; 16]), - ApplicationId::from_bytes([82; 16]), - ), - crab_cell_runtime::peer::ReplicaPeerClient::new( - host.application().registry(), - Arc::new(crab_cell_runtime::peer::PeerSigner::new( - node_session(node), - host.application().registry().release_digest(), - SigningKey::from_bytes(&[93; 32]), - )), - crab_cell_runtime::peer::PeerPrincipal { - issuer: "reference-runtime".into(), - subject: format!("node-{node}"), - actions: vec!["cell.replica.activate".into()], - }, - Arc::new(EnrolledReplicaTransport), - ), - ) - .unwrap(); - host.start().unwrap(); - (host, durability, readers) -} - -pub(super) struct EnrolledReplicaTransport; - -impl crab_cell_runtime::peer::PeerRoundTrip for EnrolledReplicaTransport { - fn send( - &self, - _: CellTarget, - _: Vec, - _: u32, - ) -> Pin>> + Send + 'static>> { - Box::pin(async { Err(Error::Peer("reader activation requires an enrolled node")) }) - } - - fn send_to_node( - &self, - _: CellTarget, - node: NodeAdvertisement, - request: Vec, - remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let endpoint = node.endpoint().to_owned(); - Box::pin(async move { - let address = endpoint - .strip_prefix("https://") - .ok_or(Error::Peer("reader endpoint is invalid"))? - .parse() - .map_err(|_| Error::Peer("reader endpoint has no socket address"))?; - super::fleet::send_tcp(address, request, remaining_ms).await - }) - } -} diff --git a/crates/crab-cell-app/tests/reference_application/process_performance.rs b/crates/crab-cell-app/tests/reference_application/process_performance.rs deleted file mode 100644 index 6e433ccd6..000000000 --- a/crates/crab-cell-app/tests/reference_application/process_performance.rs +++ /dev/null @@ -1,508 +0,0 @@ -use super::fleet::{ - GatewayStats, start_balancer, start_bound_peer_server, start_gateway_peer_server, -}; -use super::performance::run_reference_primitive_performance; -use super::performance_fixture::{ - PerfFixture, node_session, owner_routes, perf_cells, rustfs_store, -}; -use crate::*; -use std::{ - env, - net::SocketAddr, - path::Path, - process::{Child, Command, Stdio}, - time::Duration, -}; - -use crab_cell_runtime::peer::{PeerReplicaResolver, PeerSigner, PeerVerifier}; -use tokio::net::TcpListener; - -const ROLE_ENV: &str = "CRAB_CELL_PERF_PROCESS_NODE"; -const ROOT_ENV: &str = "CRAB_CELL_PERF_PROCESS_ROOT"; -const SYNC_ENV: &str = "CRAB_CELL_PERF_PROCESS_SYNC"; -const GATEWAY_ENV: &str = "CRAB_CELL_PERF_PROCESS_GATEWAY"; - -pub(super) struct ChildGuard(pub(super) Child); - -impl Drop for ChildGuard { - fn drop(&mut self) { - let _ = self.0.kill(); - let _ = self.0.wait(); - } -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "independent node role for native and Compose qualification"] -async fn fleet_process_role() { - let node = env::var(ROLE_ENV).unwrap(); - let node: usize = node.parse().unwrap(); - // Twenty live nodes plus one killed boot; replacement never reuses a session. - assert!(node <= 20); - let root = env::var(ROOT_ENV).unwrap(); - let sync = env::var(SYNC_ENV).unwrap(); - let store = rustfs_store(); - let application = Arc::new(compiled()); - let registry = application.registry(); - let tenant = TenantId::from_bytes([81; 16]); - let application_id = ApplicationId::from_bytes([82; 16]); - let layout = CellStorageLayout::new( - store, - object_store::path::Path::from(root), - *application_id.as_bytes(), - ); - // Only control markers are shared; SQLite, WAL and cache files belong to - // this process/container and cannot be used by another owner. - let directory = tempfile::TempDir::new().unwrap(); - let started = std::time::Instant::now(); - let bind = env::var("CRAB_CELL_PERF_PROCESS_BIND").unwrap_or_else(|_| "127.0.0.1:0".into()); - let listener = TcpListener::bind(&bind).await.unwrap(); - let advertised = match env::var("CRAB_CELL_PERF_PROCESS_ADVERTISE") { - Ok(endpoint) => tokio::net::lookup_host(endpoint) - .await - .unwrap() - .next() - .unwrap(), - Err(_) => listener.local_addr().unwrap(), - }; - let (host, durability, readers) = super::process_node::start( - node, - Arc::clone(&application), - &layout, - directory.path(), - format!("https://{advertised}"), - ) - .await; - let runtime = host.runtime(); - let reader = Arc::new(readers); - let read_target = CellTarget::new( - tenant, - application_id, - SQL_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - let mut handles = Vec::new(); - for (index, (namespace, role, module, incarnation, schema)) in - perf_cells().into_iter().enumerate() - { - if index % 3 != node { - continue; - } - handles.push( - bootstrap_reference_cell( - &runtime, - ®istry, - &layout, - &directory, - tenant, - application_id, - node_session(node), - namespace, - role, - module, - incarnation, - schema, - ) - .await - .unwrap(), - ); - } - let signer = PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - signer.verifying_key(), - )); - let marker = Path::new(&sync).join(format!("node-{node}.ready")); - let gateway = env::var_os(GATEWAY_ENV).is_some(); - let stats = Arc::new(GatewayStats::default()); - let server = if gateway { - publish_address(&marker, advertised); - let mut owners = Vec::new(); - for owner in 0..3 { - let ready = Path::new(&sync).join(format!("node-{owner}.ready")); - let deadline = tokio::time::Instant::now() + Duration::from_secs(120); - while !ready.exists() { - assert!( - tokio::time::Instant::now() < deadline, - "fleet routing timed out" - ); - tokio::time::sleep(Duration::from_millis(50)).await; - } - owners.push(std::fs::read_to_string(ready).unwrap().parse().unwrap()); - } - let routes = owner_routes(tenant, application_id, [owners[0], owners[1], owners[2]]); - let routes = Arc::new(std::sync::RwLock::new(routes)); - let server = start_gateway_peer_server( - listener, - ®istry, - verifier, - handles, - routes, - Arc::clone(&stats), - Some(( - reader.clone(), - super::process_node::directory(&layout, ®istry), - )), - ); - std::fs::write(Path::new(&sync).join(format!("node-{node}.serving")), []).unwrap(); - server - } else { - let server = start_bound_peer_server( - listener, - ®istry, - verifier, - handles, - Some(( - reader.clone(), - super::process_node::directory(&layout, ®istry), - )), - ); - publish_address(&marker, advertised); - server - }; - let stop = Path::new(&sync).join("stop"); - let activated = Path::new(&sync).join(format!("node-{node}-readers.ready")); - let evict = Path::new(&sync).join("readers.evicted"); - let evicted = Path::new(&sync).join(format!("node-{node}-readers.evicted")); - let minimum = Path::new(&sync).join("readers.minimum"); - let refreshed = Path::new(&sync).join(format!("node-{node}-readers.refreshed")); - while !stop.exists() { - if !activated.exists() - && (node == 0 - || reader - .status(read_target.clone()) - .await - .is_ok_and(|(_, ready)| ready)) - { - // Readiness observes authenticated owner recruitment; no fixture hint is sent. - std::fs::write(&activated, []).unwrap(); - } - if node != 0 && minimum.exists() && !refreshed.exists() { - let sequence: u64 = std::fs::read_to_string(&minimum).unwrap().parse().unwrap(); - let (receipt, ready) = reader.status(read_target.clone()).await.unwrap(); - if ready && receipt.commit_sequence >= sequence { - // This only observes the supervisor; no refresh hint is sent. - std::fs::write(&refreshed, []).unwrap(); - } - } - if evict.exists() - && !evicted.exists() - && matches!( - reader.resolve(read_target.clone()).await, - Err(Error::ReplicaUnavailable) - ) - { - // Observe the host supervisor's eviction; the fixture does not remove the view. - std::fs::write(&evicted, []).unwrap(); - } - tokio::time::sleep(Duration::from_millis(50)).await; - } - server.abort(); - println!( - "PERF node_{node}: active_cells={} retained_bytes={} local_disk_reserved_bytes={}", - host.stats().active_cells(), - host.stats().retained_bytes(), - host.stats().local_disk_reserved_bytes() - ); - host.shutdown().await.unwrap(); - assert!(matches!( - reader.activate(read_target.clone(), node_session(0)).await, - Err(Error::RuntimeClosed) - )); - assert!(matches!( - reader.resolve(read_target).await, - Err(Error::RuntimeClosed) - )); - println!("PERF node_{node}_reader_drained: activation_closed=1 resolver_closed=1"); - let mut waits = durability.object_waits(); - assert!( - node >= 3 || !waits.is_empty(), - "node {node} did not prove an object-backed mutation" - ); - if !waits.is_empty() { - super::performance::report_samples( - &format!("node_{node}_object_proof_wait"), - &mut waits, - started.elapsed(), - ); - } - if gateway { - let (local, forwarded) = stats.counts(); - std::fs::write( - Path::new(&sync).join(format!("node-{node}.counts")), - format!("{local} {forwarded}"), - ) - .unwrap(); - } - std::fs::write(Path::new(&sync).join(format!("node-{node}.done")), []).unwrap(); -} - -pub(super) fn publish_address(marker: &Path, address: SocketAddr) { - let temporary = marker.with_extension("tmp"); - std::fs::write(&temporary, address.to_string()).unwrap(); - std::fs::rename(temporary, marker).unwrap(); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 8)] -#[ignore = "manual three-process end-to-end performance run"] -async fn reference_three_process_fleet_end_to_end_performance() { - run_three_process_fleet(false).await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 8)] -#[ignore = "manual load-balanced three-process end-to-end performance run"] -async fn reference_balanced_three_process_fleet_end_to_end_performance() { - run_three_process_fleet(true).await; -} - -async fn run_three_process_fleet(balanced: bool) { - let directory = tempfile::TempDir::new().unwrap(); - let root = format!( - "{}/process-{}-{}", - env::var("CRAB_CELL_TEST_PREFIX").unwrap(), - std::process::id(), - directory.path().file_name().unwrap().to_str().unwrap() - ); - let store = rustfs_store(); - println!("PERF backend=rustfs object_prefix={root}"); - let binary = env::current_exe().unwrap(); - // Libtest omits the crate name; preserve the remaining module path when - // this suite moves so a child cannot silently select zero tests. - let module = module_path!().split_once("::").unwrap().1; - let child_test = format!("{module}::fleet_process_role"); - let mut children = Vec::new(); - let mut owners = Vec::new(); - for node in 0..3 { - let mut command = Command::new(&binary); - command - .args(["--exact", &child_test, "--ignored", "--nocapture"]) - .env(ROLE_ENV, node.to_string()) - .env(ROOT_ENV, &root) - .env(SYNC_ENV, directory.path()) - .stdout(Stdio::inherit()) - .stderr(Stdio::inherit()); - if balanced { - command.env(GATEWAY_ENV, "1"); - } - let child = command.spawn().unwrap(); - children.push(ChildGuard(child)); - let marker = directory.path().join(format!("node-{node}.ready")); - let deadline = tokio::time::Instant::now() + Duration::from_secs(120); - loop { - if marker.exists() { - owners.push( - std::fs::read_to_string(&marker) - .unwrap() - .parse::() - .unwrap(), - ); - break; - } - assert!( - children[node].0.try_wait().unwrap().is_none(), - "fleet node {node} exited before readiness" - ); - assert!( - tokio::time::Instant::now() < deadline, - "fleet node {node} timed out during bootstrap" - ); - tokio::time::sleep(Duration::from_millis(50)).await; - } - } - if balanced { - for node in 0..3 { - let serving = directory.path().join(format!("node-{node}.serving")); - let deadline = tokio::time::Instant::now() + Duration::from_secs(120); - while !serving.exists() { - assert!( - tokio::time::Instant::now() < deadline, - "fleet gateway timed out" - ); - tokio::time::sleep(Duration::from_millis(50)).await; - } - } - } - let stop = directory.path().join("stop"); - let sync_dir = directory.path().to_path_buf(); - let balancer = if balanced { - Some(start_balancer([owners[0], owners[1], owners[2]]).await) - } else { - None - }; - let fixture = PerfFixture::from_processes( - directory, - store, - [owners[0], owners[1], owners[2]], - balancer.as_ref().map(|(address, _, _)| *address), - ); - let label = if balanced { - "fleet_balanced_three_process_mixed" - } else { - "fleet_three_process_mixed" - }; - run_reference_primitive_performance(&fixture, true, label).await; - generated_action(&fixture).await; - super::process_replica::verify( - &fixture, - &sync_dir, - &root, - [owners[0], owners[1], owners[2]], - ) - .await; - if let Some((_, server, stats)) = balancer { - server.abort(); - let counts = stats.counts(); - println!( - "PERF balancer_entries: node_0={} node_1={} node_2={}", - counts[0], counts[1], counts[2] - ); - assert!(counts.into_iter().all(|count| count > 0)); - } - std::fs::write(stop, []).unwrap(); - for (node, child) in children.iter_mut().enumerate() { - let deadline = tokio::time::Instant::now() + Duration::from_secs(120); - loop { - if let Some(status) = child.0.try_wait().unwrap() { - assert!( - status.success(), - "fleet node {node} shutdown failed: {status}" - ); - break; - } - assert!( - tokio::time::Instant::now() < deadline, - "fleet node {node} shutdown timed out" - ); - tokio::time::sleep(Duration::from_millis(50)).await; - } - } - if balanced { - for node in 0..3 { - let counts = - std::fs::read_to_string(sync_dir.join(format!("node-{node}.counts"))).unwrap(); - let (local, forwarded) = counts.trim().split_once(' ').unwrap(); - let local: usize = local.parse().unwrap(); - let forwarded: usize = forwarded.parse().unwrap(); - println!("PERF gateway_node_{node}: local={local} forwarded={forwarded}"); - assert!( - local > 0 && forwarded > 0, - "node {node} missed an ingress path" - ); - } - } - drop(children); - drop(fixture); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "Compose driver for three constrained public CellNode processes and RustFS"] -async fn reference_compose_fleet_end_to_end_performance() { - let sync = env::var(SYNC_ENV).unwrap(); - let sync = Path::new(&sync); - let mut owners = Vec::new(); - for node in 0..3 { - wait_for_marker(&sync.join(format!("node-{node}.serving"))).await; - owners.push( - std::fs::read_to_string(sync.join(format!("node-{node}.ready"))) - .unwrap() - .parse() - .unwrap(), - ); - } - let owners = [owners[0], owners[1], owners[2]]; - let (balancer, server, stats) = start_balancer(owners).await; - let fixture = PerfFixture::from_processes( - tempfile::TempDir::new().unwrap(), - rustfs_store(), - owners, - Some(balancer), - ); - run_reference_primitive_performance(&fixture, true, "compose_three_node_mixed").await; - generated_action(&fixture).await; - super::process_replica::verify(&fixture, sync, &env::var(ROOT_ENV).unwrap(), owners).await; - server.abort(); - let counts = stats.counts(); - assert!(counts.iter().all(|count| *count > 0)); - assert!(counts.iter().max().unwrap() - counts.iter().min().unwrap() <= 1); - println!("PERF balancer_entries={counts:?}"); - std::fs::write(sync.join("stop"), []).unwrap(); - for node in 0..3 { - wait_for_marker(&sync.join(format!("node-{node}.done"))).await; - let counts = std::fs::read_to_string(sync.join(format!("node-{node}.counts"))).unwrap(); - let (local, forwarded) = counts.trim().split_once(' ').unwrap(); - assert!(local.parse::().unwrap() > 0 && forwarded.parse::().unwrap() > 0); - println!("PERF gateway_node_{node}: local={local} forwarded={forwarded}"); - } -} - -pub(super) async fn wait_for_marker(marker: &Path) { - let deadline = tokio::time::Instant::now() + Duration::from_secs(120); - while !marker.exists() { - assert!( - tokio::time::Instant::now() < deadline, - "node marker timed out: {marker:?}" - ); - tokio::time::sleep(Duration::from_millis(50)).await; - } -} - -async fn generated_action(fixture: &PerfFixture) { - let client = ReferenceClient::new(fixture.typed.clone()).unwrap(); - let order = client.orders(&OrderId(b"process-proof".to_vec())).unwrap(); - let before = order.receipt_count(None, ()).await.unwrap().output; - let identity = super::performance_fixture::identity(103, 0, 0); - let input = CronInvocation { - schedule_id: [103; 16], - generation: 1, - occurrence: 1, - scheduled_at_ms: super::performance_fixture::now_ms(), - payload: b"process-proof".to_vec(), - }; - let prepared = order - .prepare_receive_cron(identity, input.clone()) - .await - .unwrap(); - let committed = prepared.execute().await.unwrap(); - let duplicate = order.receive_cron(identity, input).await.unwrap(); - assert_eq!(duplicate.receipt, committed.receipt); - assert_eq!( - order - .receipt_count(Some(committed.receipt), ()) - .await - .unwrap() - .output, - before + 1 - ); - println!("PERF generated_action: committed=1 duplicate_deliveries=1 visible_effects=1"); -} - -pub(super) struct Controller<'a> { - sync: &'a Path, - sequence: usize, -} - -impl<'a> Controller<'a> { - pub(super) fn new(sync: &'a Path) -> Self { - Self { sync, sequence: 0 } - } - - pub(super) async fn command(&mut self, action: &str, count: usize) -> Vec { - let path = self.sync.join(format!("fleet-{}.request", self.sequence)); - let temporary = path.with_extension("tmp"); - std::fs::write(&temporary, format!("{action} {count}")).unwrap(); - std::fs::rename(temporary, &path).unwrap(); - let done = path.with_extension("done"); - wait_for_marker(&done).await; - self.sequence += 1; - std::fs::read_to_string(done) - .unwrap() - .split_whitespace() - .map(|node| node.parse().unwrap()) - .collect() - } -} diff --git a/crates/crab-cell-app/tests/reference_application/process_recruitment.rs b/crates/crab-cell-app/tests/reference_application/process_recruitment.rs deleted file mode 100644 index d38f88e86..000000000 --- a/crates/crab-cell-app/tests/reference_application/process_recruitment.rs +++ /dev/null @@ -1,324 +0,0 @@ -//! Signed owner recruitment and replacement after a selected process is killed. - -use super::performance_fixture::{PerfFixture, identity, node_session, now_ms, rustfs_store}; -use super::process_node::{EnrolledReplicaTransport, directory}; -use super::process_performance::{ChildGuard, wait_for_marker}; -use crate::*; -use crab_cell_runtime::{ - client::{ReadPolicy, Receipt, ReplicaReadRouter}, - peer::{PeerPrincipal, PeerRoundTrip, PeerSigner, ReplicaPeerClient, decode_peer_reply, wire}, - read_policy::ReadPolicyStore, -}; -use std::{ - collections::{HashMap, HashSet}, - env, - net::SocketAddr, - path::Path, - process::{Command, Stdio}, - sync::Mutex, - time::Duration, -}; - -pub(super) struct ObservedReads( - pub(super) Arc>>, -); - -impl PeerRoundTrip for ObservedReads { - fn send( - &self, - _: CellTarget, - _: Vec, - _: u32, - ) -> Pin>> + Send + 'static>> { - Box::pin(async { - Err(Error::Peer( - "replacement proof requires selected-node routing", - )) - }) - } - - fn send_to_node( - &self, - target: CellTarget, - node: NodeAdvertisement, - request: Vec, - remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let observed = Arc::clone(&self.0); - Box::pin(async move { - let session = node.session(); - let reply = EnrolledReplicaTransport - .send_to_node(target, node, request, remaining_ms) - .await?; - if matches!(decode_peer_reply(&reply)?.outcome, Some(wire::peer_reply::Outcome::Read(read)) - if matches!(read.result, Some(wire::read_reply::Result::CommandOutput(_)))) - { - *observed.lock().unwrap().entry(session).or_default() += 1; - } - Ok(reply) - }) - } -} - -async fn spawn(node: usize, root: &str, sync: &Path) -> (ChildGuard, SocketAddr) { - let child = Command::new(env::current_exe().unwrap()) - .args([ - "--exact", - "reference_application::process_performance::fleet_process_role", - "--ignored", - "--nocapture", - ]) - .env("CRAB_CELL_PERF_PROCESS_NODE", node.to_string()) - .env("CRAB_CELL_PERF_PROCESS_ROOT", root) - .env("CRAB_CELL_PERF_PROCESS_SYNC", sync) - .env_remove("CRAB_CELL_PERF_PROCESS_GATEWAY") - .env_remove("CRAB_CELL_PERF_PROCESS_BIND") - .env_remove("CRAB_CELL_PERF_PROCESS_ADVERTISE") - .stdout(Stdio::inherit()) - .stderr(Stdio::inherit()) - .spawn() - .unwrap(); - let child = ChildGuard(child); - let marker = sync.join(format!("node-{node}.ready")); - wait_for_marker(&marker).await; - let address = std::fs::read_to_string(marker).unwrap().parse().unwrap(); - (child, address) -} - -pub(super) async fn ready_readers( - router: &ReplicaReadRouter, - peer: &ReplicaPeerClient, - target: &CellTarget, - minimum: Receipt, - desired: usize, - excluded: Option, -) -> HashSet { - tokio::time::timeout(Duration::from_secs(60), async { - loop { - let (expected, selected) = router.selected(target).await.unwrap(); - let mut ready = HashSet::new(); - for node in selected { - let session = node.session(); - if Some(session) == excluded { - continue; - } - if let Ok((receipt, true)) = peer.status(target, node, expected).await - && receipt.commit_sequence >= minimum.commit_sequence - { - ready.insert(session); - } - } - if ready.len() == desired { - return ready; - } - tokio::time::sleep(Duration::from_millis(100)).await; - } - }) - .await - .expect("owner did not automatically recruit the desired current readers") -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "live RustFS three-to-five-process recruitment and abrupt reader loss"] -async fn owner_replaces_killed_reader_through_public_hosts() { - let sync = tempfile::TempDir::new().unwrap(); - let root = format!( - "{}/reader-replacement-{}-{}", - env::var("CRAB_CELL_TEST_PREFIX").unwrap(), - std::process::id(), - sync.path().file_name().unwrap().to_str().unwrap() - ); - let mut children = Vec::new(); - let mut endpoints = Vec::new(); - for node in 0..3 { - let (child, address) = spawn(node, &root, sync.path()).await; - children.push(child); - endpoints.push(address); - } - let fixture = PerfFixture::from_processes( - tempfile::TempDir::new().unwrap(), - rustfs_store(), - [endpoints[0], endpoints[1], endpoints[2]], - None, - ); - super::performance::run_reference_primitive_performance( - &fixture, - true, - "reader_recruitment_initial", - ) - .await; - let target = &fixture.sql_target; - let layout = CellStorageLayout::new( - rustfs_store(), - object_store::path::Path::from(root.clone()), - *target.application().as_bytes(), - ); - let authority = CellAuthority::new(layout.clone()); - let before = authority.load(target.cell_id()).await.unwrap().unwrap(); - let owner = ReferenceClient::new(fixture.typed.clone()).unwrap(); - let order = owner - .orders(&OrderId(b"reader-replacement".to_vec())) - .unwrap(); - let written = order - .receive_cron( - identity(105, 0, 0), - CronInvocation { - schedule_id: [105; 16], - generation: 1, - occurrence: 1, - scheduled_at_ms: now_ms(), - payload: b"survive-reader-loss".to_vec(), - }, - ) - .await - .unwrap(); - let expected_output = order - .receipt_count(Some(written.receipt), ()) - .await - .unwrap(); - let observed = Arc::new(Mutex::new(HashMap::new())); - let peer = ReplicaPeerClient::new( - Arc::clone(&fixture.registry), - Arc::new(PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - fixture.registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - )), - PeerPrincipal { - issuer: "reference-performance".into(), - subject: "reader-replacement".into(), - actions: vec!["cell.read".into(), "cell.replica.status".into()], - }, - Arc::new(ObservedReads(Arc::clone(&observed))), - ); - let router = ReplicaReadRouter::new(authority.clone(), directory(&layout, &fixture.registry)); - let client = fixture - .client - .with_read_replicas(router.clone(), peer.clone(), None) - .unwrap(); - let handle = ApplicationHandle::new( - client, - Arc::new(compiled()), - target.tenant(), - target.application(), - ) - .unwrap(); - let generated = ReferenceClient::new(handle.with_read_policy(ReadPolicy::Replica)).unwrap(); - let reader = generated - .orders(&OrderId(b"reader-replacement".to_vec())) - .unwrap(); - ReadPolicyStore::new(layout.clone()) - .create(target.cell_id(), before.value().incarnation, 2) - .await - .unwrap(); - let initial = ready_readers(&router, &peer, target, written.receipt, 2, None).await; - assert_eq!(initial, HashSet::from([node_session(1), node_session(2)])); - let (expected, selected) = router.selected(target).await.unwrap(); - assert!( - matches!( - peer.activate( - target, - &directory(&layout, &fixture.registry), - selected[0].clone(), - expected, - ) - .await, - Err(Error::PeerAuthorization(_)) - ), - "read/status capability admitted an activation" - ); - - for _ in 0..8 { - assert_eq!( - reader - .receipt_count(Some(written.receipt), ()) - .await - .unwrap(), - expected_output - ); - } - assert_eq!( - observed - .lock() - .unwrap() - .keys() - .copied() - .collect::>(), - initial - ); - - for node in 3..5 { - let (child, _) = spawn(node, &root, sync.path()).await; - children.push(child); - } - assert_eq!( - directory(&layout, &fixture.registry) - .live(now_ms(), 10) - .await - .unwrap() - .len(), - 5 - ); - let selected = ready_readers(&router, &peer, target, written.receipt, 2, None).await; - let lost = *selected - .iter() - .min_by_key(|session| *session.as_bytes()) - .unwrap(); - let lost_node = (1..5).find(|node| node_session(*node) == lost).unwrap(); - children[lost_node].0.kill().unwrap(); - let exit = children[lost_node].0.wait().unwrap(); - assert!(!exit.success(), "reader fault did not kill the process"); - let started = std::time::Instant::now(); - let replacement = ready_readers(&router, &peer, target, written.receipt, 2, Some(lost)).await; - let recovery_ms = started.elapsed().as_millis(); - assert!( - replacement - .iter() - .any(|session| !selected.contains(session)) - ); - observed.lock().unwrap().clear(); - for _ in 0..12 { - assert_eq!( - reader - .receipt_count(Some(written.receipt), ()) - .await - .unwrap(), - expected_output - ); - } - assert_eq!( - observed - .lock() - .unwrap() - .keys() - .copied() - .collect::>(), - replacement - ); - let after = authority.load(target.cell_id()).await.unwrap().unwrap(); - assert_eq!(after.value().owner, before.value().owner); - assert_eq!(after.value().epoch, before.value().epoch); - assert_eq!(after.value().incarnation, written.receipt.incarnation); - println!( - "PERF reader_replacement: initial_nodes=3 expanded_nodes=5 killed_node={lost_node} ready_readers=2 owner_unchanged=1 exact_queries=12 recovery_ms={}", - recovery_ms - ); - std::fs::write(sync.path().join("stop"), []).unwrap(); - for (node, child) in children.iter_mut().enumerate() { - if node == lost_node { - continue; - } - tokio::time::timeout(Duration::from_secs(60), async { - loop { - if let Some(status) = child.0.try_wait().unwrap() { - assert!(status.success(), "node {node} failed drain: {status}"); - break; - } - tokio::time::sleep(Duration::from_millis(50)).await; - } - }) - .await - .expect("surviving node did not finish drain"); - assert!(sync.path().join(format!("node-{node}.done")).exists()); - } -} diff --git a/crates/crab-cell-app/tests/reference_application/process_replica.rs b/crates/crab-cell-app/tests/reference_application/process_replica.rs deleted file mode 100644 index 635e4f17b..000000000 --- a/crates/crab-cell-app/tests/reference_application/process_replica.rs +++ /dev/null @@ -1,197 +0,0 @@ -//! Generated snapshot queries across independent public host processes. - -use super::performance_fixture::{PerfFixture, identity, node_session, now_ms, rustfs_store}; -use super::process_node; -use crate::*; -use crab_cell_runtime::client::{ReadPolicy, ReplicaReadRouter}; -use crab_cell_runtime::peer::{ - PeerPrincipal, PeerRoundTrip, PeerSigner, ReplicaPeerClient, decode_peer_reply, wire, -}; -use crab_cell_runtime::read_policy::ReadPolicyStore; -use std::{ - net::SocketAddr, - path::Path, - sync::atomic::{AtomicUsize, Ordering}, - time::Duration, -}; - -struct ReaderTransport { - nodes: [SocketAddr; 3], - successes: Arc<[AtomicUsize; 3]>, -} - -impl PeerRoundTrip for ReaderTransport { - fn send( - &self, - _target: CellTarget, - _request: Vec, - _remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - Box::pin(async { Err(Error::Peer("replica proof requires a selected node")) }) - } - - fn send_to_node( - &self, - _target: CellTarget, - node: NodeAdvertisement, - request: Vec, - remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let index = (1..3).find(|index| node.session() == node_session(*index)); - let nodes = self.nodes; - let successes = Arc::clone(&self.successes); - Box::pin(async move { - let index = index.ok_or(Error::Peer( - "replica routing selected the owner or an unknown node", - ))?; - assert_eq!(node.node().as_bytes(), node_session(index).as_bytes()); - let reply = super::fleet::send_tcp(nodes[index], request, remaining_ms).await?; - if matches!(decode_peer_reply(&reply)?.outcome, Some(wire::peer_reply::Outcome::Read(read)) - if matches!(read.result, Some(wire::read_reply::Result::CommandOutput(_)))) - { - successes[index].fetch_add(1, Ordering::Relaxed); - } - Ok(reply) - }) - } -} - -async fn wait_for_readers(sync: &Path) { - for node in 0..3 { - super::process_performance::wait_for_marker( - &sync.join(format!("node-{node}-readers.ready")), - ) - .await; - } -} - -pub(super) async fn verify(fixture: &PerfFixture, sync: &Path, root: &str, nodes: [SocketAddr; 3]) { - let target = &fixture.sql_target; - let layout = CellStorageLayout::new( - rustfs_store(), - object_store::path::Path::from(root), - *target.application().as_bytes(), - ); - let authority = CellAuthority::new(layout.clone()); - let control = authority.load(target.cell_id()).await.unwrap().unwrap(); - assert_eq!( - control.value().owner.as_ref().unwrap().session, - node_session(0) - ); - let successes = Arc::new(std::array::from_fn(|_| AtomicUsize::new(0))); - let transport = Arc::new(ReaderTransport { - nodes, - successes: Arc::clone(&successes), - }); - let peer = ReplicaPeerClient::new( - Arc::clone(&fixture.registry), - Arc::new(PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - fixture.registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - )), - PeerPrincipal { - issuer: "reference-performance".into(), - subject: "replica-driver".into(), - actions: vec!["cell.read".into()], - }, - transport, - ); - let client = fixture - .client - .with_read_replicas( - ReplicaReadRouter::new( - authority, - process_node::directory(&layout, &fixture.registry), - ), - peer, - None, - ) - .unwrap(); - let typed = ApplicationHandle::new( - client, - Arc::new(compiled()), - target.tenant(), - target.application(), - ) - .unwrap(); - let generated = ReferenceClient::new(typed.with_read_policy(ReadPolicy::Replica)).unwrap(); - let order = generated - .orders(&OrderId(b"process-replica-proof".to_vec())) - .unwrap(); - assert!(matches!( - order.receipt_count(None, ()).await, - Err(InvocationError::NotStarted(Error::ReplicaUnavailable)) - )); - ReadPolicyStore::new(layout.clone()) - .create(target.cell_id(), control.value().incarnation, 2) - .await - .unwrap(); - wait_for_readers(sync).await; - let before = order.receipt_count(None, ()).await.unwrap(); - for _ in 0..5 { - assert_eq!( - order.receipt_count(Some(before.receipt), ()).await.unwrap(), - before - ); - } - let identity = identity(104, 0, 0); - let input = CronInvocation { - schedule_id: [104; 16], - generation: 1, - occurrence: 1, - scheduled_at_ms: now_ms(), - payload: b"replica-process-proof".to_vec(), - }; - let committed = order.receive_cron(identity, input.clone()).await.unwrap(); - let duplicate = order.receive_cron(identity, input).await.unwrap(); - assert_eq!(duplicate.receipt, committed.receipt); - assert!(committed.receipt.commit_sequence > before.receipt.commit_sequence); - // The host must discover the new root without a fixture-issued refresh hint. - std::fs::write( - sync.join("readers.minimum"), - committed.receipt.commit_sequence.to_string(), - ) - .unwrap(); - tokio::time::timeout(Duration::from_secs(30), async { - for node in 1..3 { - super::process_performance::wait_for_marker( - &sync.join(format!("node-{node}-readers.refreshed")), - ) - .await; - } - }) - .await - .expect("automatic reader refresh timed out"); - for _ in 0..6 { - let observed = order - .receipt_count(Some(committed.receipt), ()) - .await - .unwrap(); - assert_eq!(observed.output, before.output + 1); - assert_eq!(observed.receipt, committed.receipt); - } - let policy = ReadPolicyStore::new(layout); - let current = policy.load(target.cell_id()).await.unwrap().unwrap(); - policy.update(¤t, 0).await.unwrap(); - std::fs::write(sync.join("readers.evicted"), []).unwrap(); - for node in 0..3 { - super::process_performance::wait_for_marker( - &sync.join(format!("node-{node}-readers.evicted")), - ) - .await; - } - assert!(matches!( - order.receipt_count(None, ()).await, - Err(InvocationError::NotStarted(Error::ReplicaUnavailable)) - )); - let counts = successes - .each_ref() - .map(|count| count.load(Ordering::Relaxed)); - assert_eq!(counts[0], 0); - assert_eq!(counts[1] + counts[2], 12); - assert!(counts[1].abs_diff(counts[2]) <= 1); - println!( - "PERF generated_replica_reads: successful_by_node={counts:?} unavailable_before_open=1 automatic_recruitment=2 automatic_refresh=2 evicted_readers=2 duplicate_effects=0" - ); -} diff --git a/crates/crab-cell-app/tests/reference_application/process_scaling.rs b/crates/crab-cell-app/tests/reference_application/process_scaling.rs deleted file mode 100644 index c78306c28..000000000 --- a/crates/crab-cell-app/tests/reference_application/process_scaling.rs +++ /dev/null @@ -1,594 +0,0 @@ -//! Constrained reader scaling; the external controller owns container faults. - -use super::fleet::start_balancer; -use super::performance::{report_samples, run_reference_primitive_performance}; -use super::performance_fixture::{PerfFixture, identity, node_session, now_ms, rustfs_store}; -use super::process_node::directory; -use super::process_performance::{Controller, wait_for_marker}; -use super::process_recruitment::{ObservedReads, ready_readers}; -use crate::*; -use crab_cell_runtime::{ - client::{Observed, ReadPolicy, ReplicaReadRouter}, - peer::{PeerPrincipal, PeerSigner, ReplicaPeerClient}, - read_policy::ReadPolicyStore, -}; -use std::{ - collections::{BTreeMap, HashMap}, - env, - fs::File, - io::{BufWriter, Write}, - net::SocketAddr, - path::Path, - sync::Mutex, - time::{Duration, Instant}, -}; - -struct LoadWindow<'a> { - label: &'a str, - request_domain: u8, - started: Instant, -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "Compose controller required: 3/5/10/20 constrained nodes and reader SIGKILL"] -async fn reference_compose_reader_scaling() { - let sync = env::var("CRAB_CELL_PERF_PROCESS_SYNC").unwrap(); - let sync = Path::new(&sync); - let mut controller = Controller::new(sync); - let application = Arc::new(compiled()); - let registry = application.registry(); - let layout = CellStorageLayout::new( - rustfs_store(), - object_store::path::Path::from(env::var("CRAB_CELL_PERF_PROCESS_ROOT").unwrap()), - *ApplicationId::from_bytes([82; 16]).as_bytes(), - ); - let authority = CellAuthority::new(layout.clone()); - let membership = directory(&layout, ®istry); - let router = ReplicaReadRouter::new(authority.clone(), membership.clone()); - let observed = Arc::new(Mutex::new(HashMap::new())); - let peer = ReplicaPeerClient::new( - registry.clone(), - Arc::new(PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - )), - PeerPrincipal { - issuer: "reference-performance".into(), - subject: "reader-scaling".into(), - actions: vec!["cell.read".into(), "cell.replica.status".into()], - }, - Arc::new(ObservedReads(observed.clone())), - ); - let policy = ReadPolicyStore::new(layout); - let mut original_owner = None; - let mut survivors = Vec::new(); - for count in [3, 5, 10, 20] { - survivors = controller.command("scale", count).await; - assert_eq!(survivors.len(), count); - let mut endpoints = Vec::new(); - for node in &survivors { - wait_for_marker(&sync.join(format!("node-{node}.serving"))).await; - endpoints.push( - std::fs::read_to_string(sync.join(format!("node-{node}.ready"))) - .unwrap() - .parse::() - .unwrap(), - ); - } - assert_eq!(membership.live(now_ms(), 32).await.unwrap().len(), count); - let owners = [endpoints[0], endpoints[1], endpoints[2]]; - let (balancer, server, ingress) = start_balancer(endpoints).await; - let fixture = PerfFixture::from_processes( - tempfile::TempDir::new().unwrap(), - rustfs_store(), - owners, - Some(balancer), - ); - if count == 3 { - run_reference_primitive_performance(&fixture, true, "reader_scaling_initial").await; - } - let target = &fixture.sql_target; - let owner = ReferenceClient::new(fixture.typed.clone()).unwrap(); - let owner = owner.orders(&OrderId(b"compose-scaling".to_vec())).unwrap(); - let client = fixture - .client - .clone() - .with_read_replicas(router.clone(), peer.clone(), None) - .unwrap(); - let handle = ApplicationHandle::new( - client, - application.clone(), - target.tenant(), - target.application(), - ) - .unwrap(); - let generated = ReferenceClient::new(handle.with_read_policy(ReadPolicy::Replica)).unwrap(); - let reader = generated - .orders(&OrderId(b"compose-scaling".to_vec())) - .unwrap(); - let control = authority.load(target.cell_id()).await.unwrap().unwrap(); - let ownership = ( - control.value().owner.clone(), - control.value().epoch, - control.value().incarnation, - ); - assert_eq!(original_owner.get_or_insert(ownership.clone()), &ownership); - // One spare at five nodes lets expiry recruit a replacement without - // a fixture activation or a new process masking the failure. - let desired = if count == 5 { 3 } else { count - 1 }; - match policy.load(target.cell_id()).await.unwrap() { - Some(current) => { - policy.update(¤t, desired as u16).await.unwrap(); - } - None => { - policy - .create( - target.cell_id(), - control.value().incarnation, - desired as u16, - ) - .await - .unwrap(); - } - } - let mut selected = Default::default(); - let mut expected = owner.receipt_count(None, ()).await.unwrap(); - for round in 0..2 { - let input = CronInvocation { - schedule_id: [106; 16], - generation: 1, - occurrence: (count * 2 + round) as u64, - scheduled_at_ms: now_ms(), - payload: b"compose-reader-scaling".to_vec(), - }; - let identity = identity(106, count * 2 + round, 0); - let committed = owner.receive_cron(identity, input.clone()).await.unwrap(); - let started = Instant::now(); - let duplicate = owner.receive_cron(identity, input).await.unwrap(); - assert_eq!(duplicate.receipt, committed.receipt); - let output = owner - .receipt_count(Some(committed.receipt), ()) - .await - .unwrap(); - assert_eq!(output.output, expected.output + 1); - expected = output; - selected = - ready_readers(&router, &peer, target, committed.receipt, desired, None).await; - println!( - "PERF reader_readiness: nodes={count} readers={desired} round={round} elapsed_ms={}", - started.elapsed().as_millis() - ); - } - observed.lock().unwrap().clear(); - let started = Instant::now(); - let mut samples = Vec::new(); - for _ in 0..desired * 30 { - let call = Instant::now(); - assert_eq!( - reader - .receipt_count(Some(expected.receipt), ()) - .await - .unwrap(), - expected - ); - samples.push(call.elapsed()); - } - report_samples( - &format!("replica_reads_{count}_nodes"), - &mut samples, - started.elapsed(), - ); - let reads = observed.lock().unwrap().clone(); - assert_eq!( - reads - .keys() - .copied() - .collect::>(), - selected - ); - assert!(reads.values().all(|count| *count == 30)); - assert!(!reads.contains_key(&node_session(0))); - let by_node = survivors - .iter() - .map(|node| (*node, reads.get(&node_session(*node)).copied().unwrap_or(0))) - .collect::>(); - // Exercise every ingress independently of selected replica placement. - for _ in 0..count * 2 { - assert_eq!( - owner - .receipt_count(Some(expected.receipt), ()) - .await - .unwrap(), - expected - ); - } - let entries = ingress.counts(); - assert!(entries.iter().all(|count| *count > 0)); - assert!(entries.iter().max().unwrap() - entries.iter().min().unwrap() <= 1); - println!( - "PERF reader_scale: nodes={count} readers={desired} by_node={by_node:?} ingress={entries:?} owner_unchanged=1" - ); - observed.lock().unwrap().clear(); - let window = LoadWindow { - label: "mixed", - request_domain: 107, - started: Instant::now(), - }; - expected = mixed_load(&owner, &reader, sync, count, expected, window).await; - let reads = observed.lock().unwrap().clone(); - assert_eq!( - reads - .keys() - .copied() - .collect::>(), - selected - ); - assert!(!reads.contains_key(&node_session(0))); - let entries = ingress - .counts() - .iter() - .zip(entries) - .map(|(after, before)| after - before) - .collect::>(); - assert!(entries.iter().all(|count| *count > 0)); - assert!(entries.iter().max().unwrap() - entries.iter().min().unwrap() <= 1); - println!( - "PERF mixed_reader_distribution: nodes={count} by_session={reads:?} writer_ingress={entries:?}" - ); - ready_readers(&router, &peer, target, expected.receipt, desired, None).await; - if count == 5 { - let lost = *survivors - .iter() - .find(|node| **node >= 3 && selected.contains(&node_session(**node))) - .unwrap(); - // Keep owner ingress on the original three nodes while faulting a - // reader. Gateway withdrawal is a separate product fleet invariant. - let (address, fault_server, _) = start_balancer(owners).await; - let fault_fixture = PerfFixture::from_processes( - tempfile::TempDir::new().unwrap(), - rustfs_store(), - owners, - Some(address), - ); - let fault_client = ReferenceClient::new(fault_fixture.typed.clone()).unwrap(); - let fault_owner = fault_client - .orders(&OrderId(b"compose-scaling".to_vec())) - .unwrap(); - observed.lock().unwrap().clear(); - let window = LoadWindow { - label: "reader_loss", - request_domain: 108, - started: Instant::now(), - }; - let started = window.started; - let minimum = expected.receipt; - let fault = async { - tokio::time::sleep_until((started + Duration::from_secs(10)).into()).await; - assert!( - observed - .lock() - .unwrap() - .get(&node_session(lost)) - .copied() - .unwrap_or(0) - > 0 - ); - let requested_us = started.elapsed().as_micros(); - survivors = controller.command("kill", lost).await; - assert!(!survivors.contains(&lost)); - let killed_us = started.elapsed().as_micros(); - let replacement = ready_readers( - &router, - &peer, - target, - minimum, - desired, - Some(node_session(lost)), - ) - .await; - let ready_us = started.elapsed().as_micros(); - let added = replacement - .difference(&selected) - .copied() - .collect::>(); - assert!(!added.is_empty()); - loop { - // Only load lanes issue queries here. Readiness alone cannot - // prove that a newly recruited reader served live traffic. - if added.iter().all(|session| { - observed.lock().unwrap().get(session).copied().unwrap_or(0) > 0 - }) { - break; - } - assert!( - started.elapsed() < Duration::from_secs(50), - "replacement did not serve during the load window" - ); - tokio::time::sleep(Duration::from_millis(50)).await; - } - let served_us = started.elapsed().as_micros(); - assert!( - served_us < 50_000_000, - "replacement left no post-recovery load interval" - ); - std::fs::write(sync.join("reader-loss.tsv"), format!( - "killed_node\trequested_us\tkilled_us\tready_us\tserved_us\n{lost}\t{requested_us}\t{killed_us}\t{ready_us}\t{served_us}\n" - )).unwrap(); - ( - replacement, - killed_us - requested_us, - served_us - requested_us, - ) - }; - let load = mixed_load(&fault_owner, &reader, sync, count, expected, window); - let (output, (replacement, fault_command_us, recovery_us)) = tokio::join!(load, fault); - expected = output; - fault_server.abort(); - assert_eq!( - ready_readers( - &router, - &peer, - target, - expected.receipt, - desired, - Some(node_session(lost)) - ) - .await, - replacement - ); - observed.lock().unwrap().clear(); - for _ in 0..12 { - assert_eq!( - reader - .receipt_count(Some(expected.receipt), ()) - .await - .unwrap(), - expected - ); - } - assert_eq!( - observed - .lock() - .unwrap() - .keys() - .copied() - .collect::>(), - replacement - ); - println!( - "PERF constrained_reader_replacement: nodes_before=5 nodes_after=4 killed_node={lost} ready_readers={desired} exact_queries=12 fault_command_us={fault_command_us} recovery_us={recovery_us} during_arrivals=1" - ); - } - let after = authority.load(target.cell_id()).await.unwrap().unwrap(); - assert_eq!( - ( - after.value().owner.clone(), - after.value().epoch, - after.value().incarnation - ), - ownership - ); - if count == 20 { - let current = policy.load(target.cell_id()).await.unwrap().unwrap(); - policy.update(¤t, 0).await.unwrap(); - std::fs::write(sync.join("readers.evicted"), []).unwrap(); - for node in &survivors { - wait_for_marker(&sync.join(format!("node-{node}-readers.evicted"))).await; - } - assert!(matches!( - reader.receipt_count(None, ()).await, - Err(InvocationError::NotStarted(Error::ReplicaUnavailable)) - )); - } - server.abort(); - } - std::fs::write(sync.join("stop"), []).unwrap(); - for node in survivors { - wait_for_marker(&sync.join(format!("node-{node}.done"))).await; - } -} - -async fn mixed_load( - owner: &ReferenceOrderCell, - reader: &ReferenceOrderCell, - sync: &Path, - nodes: usize, - baseline: Observed, - window: LoadWindow<'_>, -) -> Observed { - let LoadWindow { - label, - request_domain, - started, - } = window; - let window = Duration::from_secs(60); - let latest = Mutex::new(baseline.clone()); - let writes = async { - let mut raw = - BufWriter::new(File::create(sync.join(format!("{label}-{nodes}-writes.tsv"))).unwrap()); - writeln!( - raw, - "arrival\tscheduled_us\tstarted_us\telapsed_us\toutcome\tsequence\tcount" - ) - .unwrap(); - writeln!( - raw, - "baseline\t0\t0\t0\tbaseline\t{}\t{}", - baseline.receipt.commit_sequence, baseline.output - ) - .unwrap(); - let mut acknowledged = - BTreeMap::from([(baseline.receipt.commit_sequence, baseline.output)]); - let mut samples = Vec::new(); - let mut missed = 0; - for arrival in 0..300 { - let offset = Duration::from_millis(arrival * 200); - tokio::time::sleep_until((started + offset).into()).await; - let dispatched = started.elapsed(); - // A slow writer must not lower the advertised rate or catch up in - // a burst. Missing arrivals remain part of the capacity result. - if dispatched.saturating_sub(offset) >= Duration::from_millis(200) { - missed += 1; - writeln!( - raw, - "{arrival}\t{}\t{}\t0\tscheduler_late\t0\t0", - offset.as_micros(), - dispatched.as_micros() - ) - .unwrap(); - continue; - } - let input = CronInvocation { - schedule_id: [request_domain; 16], - generation: 1, - occurrence: (nodes as u64 * 300) + arrival, - scheduled_at_ms: now_ms(), - payload: b"sustained-replica-refresh".to_vec(), - }; - let call = Instant::now(); - let committed = owner - .receive_cron( - identity(request_domain, nodes * 300 + arrival as usize, 0), - input, - ) - .await - .unwrap(); - let elapsed = call.elapsed(); - let count = baseline.output + samples.len() as u64 + 1; - assert_eq!(committed.receipt.cell, baseline.receipt.cell); - assert_eq!(committed.receipt.incarnation, baseline.receipt.incarnation); - assert!(committed.receipt.commit_sequence > *acknowledged.last_key_value().unwrap().0); - acknowledged.insert(committed.receipt.commit_sequence, count); - *latest.lock().unwrap() = Observed { - output: count, - receipt: committed.receipt, - }; - writeln!( - raw, - "{arrival}\t{}\t{}\t{}\tcommitted\t{}\t{count}", - offset.as_micros(), - dispatched.as_micros(), - elapsed.as_micros(), - committed.receipt.commit_sequence - ) - .unwrap(); - raw.flush().unwrap(); - samples.push(elapsed); - } - raw.flush().unwrap(); - assert!(!samples.is_empty()); - println!( - "PERF {label}_writes: nodes={nodes} planned=300 committed={} missed={missed} arrival_seconds=60", - samples.len() - ); - report_samples( - &format!("{label}_writes_{nodes}_nodes"), - &mut samples, - started.elapsed().max(window), - ); - acknowledged - }; - let reads = futures_util::future::join_all((0..8).map(|lane| { - let latest = &latest; - async move { - let mut raw = BufWriter::new( - File::create(sync.join(format!("{label}-{nodes}-reader-{lane}.tsv"))).unwrap(), - ); - writeln!( - raw, - "started_us\telapsed_us\tminimum_sequence\tlatest_count\toutcome\tsequence\tcount" - ) - .unwrap(); - let mut samples = Vec::new(); - let mut behind = 0; - while started.elapsed() < window { - // Four lanes request the last acknowledged receipt; four - // permit older snapshots and report their observed lag. - let known = latest.lock().unwrap().clone(); - let minimum = (lane % 2 == 0).then_some(known.receipt); - let floor = minimum.map_or(0, |receipt| receipt.commit_sequence); - let dispatched = started.elapsed(); - let call = Instant::now(); - match reader.receipt_count(minimum, ()).await { - Ok(observed) => { - let elapsed = call.elapsed(); - assert_eq!(observed.receipt.cell, baseline.receipt.cell); - assert_eq!(observed.receipt.incarnation, baseline.receipt.incarnation); - assert!(observed.receipt.commit_sequence >= floor); - writeln!( - raw, - "{}\t{}\t{floor}\t{}\tok\t{}\t{}", - dispatched.as_micros(), - elapsed.as_micros(), - known.output, - observed.receipt.commit_sequence, - observed.output - ) - .unwrap(); - samples.push((observed, elapsed, known.output)); - assert!( - samples.len() <= 100_000, - "reader evidence exceeded its memory bound" - ); - } - Err(InvocationError::NotStarted(Error::ReplicaBehind { .. })) - if minimum.is_some() => - { - behind += 1; - writeln!( - raw, - "{}\t{}\t{floor}\t{}\tbehind\t0\t0", - dispatched.as_micros(), - call.elapsed().as_micros(), - known.output - ) - .unwrap(); - } - Err(error) => panic!("mixed replica read failed: {error}"), - } - } - raw.flush().unwrap(); - assert!(!samples.is_empty(), "reader lane made no progress"); - (samples, behind) - } - })); - let (acknowledged, readers) = tokio::join!(writes, reads); - let mut latencies = Vec::new(); - let mut lag = 0; - let mut behind = 0; - for (samples, rejected) in readers { - behind += rejected; - for (observed, elapsed, known_count) in samples { - // A read can finish before its overlapping write is acknowledged. - // Join after both lanes finish, using exact commit positions. - let (_, count) = acknowledged - .range(..=observed.receipt.commit_sequence) - .next_back() - .unwrap(); - assert_eq!( - observed.output, *count, - "snapshot value disagrees with its receipt" - ); - lag = lag.max(known_count.saturating_sub(observed.output)); - latencies.push(elapsed); - } - } - println!( - "PERF {label}_reads: nodes={nodes} clients=8 behind={behind} max_acknowledged_count_lag={lag} window_seconds=60" - ); - report_samples( - &format!("{label}_replica_reads_{nodes}_nodes"), - &mut latencies, - started.elapsed(), - ); - let expected = latest.into_inner().unwrap(); - assert_eq!( - owner - .receipt_count(Some(expected.receipt), ()) - .await - .unwrap(), - expected - ); - expected -} diff --git a/crates/crab-cell-app/tests/reference_application/public_host.rs b/crates/crab-cell-app/tests/reference_application/public_host.rs deleted file mode 100644 index 739b26dbe..000000000 --- a/crates/crab-cell-app/tests/reference_application/public_host.rs +++ /dev/null @@ -1,459 +0,0 @@ -//! Three public `CellNode` hosts exercising the reference application's client contract. - -use super::fleet::{GatewayStats, peer_round_trip, start_gateway_peer_server, start_peer_server}; -use super::performance::report_samples; -use super::performance_fixture::{PerfFixture, identity, node_session, now_ms}; -use crate::*; -use std::collections::HashMap; -use std::sync::atomic::{AtomicBool, Ordering}; -use std::time::Instant; - -use crab_cell_runtime::cell::catalog::CellCatalog; -use crab_cell_runtime::cell::executor::Resolution; -use crab_cell_runtime::peer::{PeerPrincipal, PeerRoundTrip, PeerSigner, PeerVerifier}; -use crab_cell_runtime::recovery::manifest::RecoveryManifestStore; -use object_store::path::Path; -use tokio::net::TcpListener; - -struct LoseMutationReply { - inner: Arc, - verifier: Arc, - dropped: AtomicBool, -} - -impl PeerRoundTrip for LoseMutationReply { - fn send( - &self, - target: CellTarget, - request: Vec, - remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let inner = Arc::clone(&self.inner); - let is_mutation = self - .verifier - .verify(&request, now_ms()) - .unwrap() - .operation_tag() - == 10; - let drop_reply = is_mutation && !self.dropped.swap(true, Ordering::SeqCst); - Box::pin(async move { - let reply = inner.send(target, request, remaining_ms).await?; - if drop_reply { - return Err(Error::PeerTransportUnknown { - context: "test mutation reply lost after commit", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::ConnectionReset, - "reply dropped", - )), - }); - } - Ok(reply) - }) - } -} - -fn invocation(occurrence: u64) -> CronInvocation { - CronInvocation { - schedule_id: [72; 16], - generation: 1, - occurrence, - scheduled_at_ms: now_ms(), - payload: b"public-host".to_vec(), - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn due_workflow_activity_survives_a_clock_rollback() { - use crab_cell_runtime::codec::{BoundedEncoder, WireValue}; - use crab_cell_runtime::primitives::workflow::WorkflowStart; - use crab_cell_runtime::registry::CommandInvocation; - - for nodes in [1, 3] { - let fixture = PerfFixture::start(nodes).await; - let handle = fixture - .owned_handles - .iter() - .flatten() - .find(|handle| handle.catalog().entry().namespace() == WORKFLOW_NAMESPACE) - .unwrap(); - let target = CellTarget::new( - fixture.sql_target.tenant(), - fixture.sql_target.application(), - WORKFLOW_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - let identity = identity(8, 0, 0); - let mut encoder = BoundedEncoder::new(1024).unwrap(); - WorkflowStart { - workflow_id: b"clock-rollback".to_vec(), - request_id: identity.request_id, - event: b"activity".to_vec(), - } - .encode(&mut encoder) - .unwrap(); - let input = encoder.finish(); - let registry = Arc::clone(&fixture.registry); - // Publish at a later clock sample, then let the ordinary application - // supervisor use the earlier wall clock without changing global time. - let future = now_ms() + 10_000; - handle - .execute( - identity, - Digest::from_bytes([89; 32]), - future, - input.len(), - 1024, - move |tx| { - registry.execute_command( - tx, - CommandInvocation { - module: WORKFLOW_MODULE, - operation_id: ReferenceWorkflow::START_COMMAND_ID, - codec_version: 1, - schema: 1, - target, - sequence: 1, - now_ms: future, - input: &input, - }, - ) - }, - ) - .await - .unwrap(); - let supervisor = crab_cell_runtime::primitives::workflow::ActivitySupervisor::new( - fixture.typed.activities::().unwrap(), - 5_000, - ) - .unwrap(); - let outcome = supervisor.run_once(0, None).await.unwrap(); - assert!( - matches!(outcome, ActivityRunOutcome::Completed { .. }), - "{nodes} nodes: {outcome:?}" - ); - fixture.shutdown().await; - } -} - -fn signed_client( - fixture: &PerfFixture, - round_trip: Arc, - binding_node: usize, -) -> ReferenceClient { - let signer = Arc::new(PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - fixture.registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - )); - let principal = PeerPrincipal { - issuer: "reference-performance".into(), - subject: "fleet-driver".into(), - actions: vec!["cell.read".into(), "cell.write".into()], - }; - let client = CellClient::peer(Arc::clone(&fixture.registry), signer, principal, round_trip); - let handle = fixture.nodes[binding_node] - .application_handle::( - client, - fixture.sql_target.tenant(), - fixture.sql_target.application(), - ) - .unwrap(); - ReferenceClient::new(handle).unwrap() -} - -mod replicas; -mod rollout; - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn three_node_host_resolves_ambiguous_result_and_deduplicates_delivery() { - let fixture = PerfFixture::start(3).await; - let round_trip = fixture.round_trip.as_ref().unwrap(); - let signer = PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - fixture.registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - fixture.registry.release_digest(), - signer.verifying_key(), - )); - let client = signed_client( - &fixture, - Arc::new(LoseMutationReply { - inner: Arc::clone(round_trip), - verifier, - dropped: AtomicBool::new(false), - }), - 0, - ); - let order = client.orders(&OrderId(b"same-order".to_vec())).unwrap(); - let identity = reference_identity(73, now_ms()); - let input = invocation(1); - let prepared = order - .prepare_receive_cron(identity, input.clone()) - .await - .unwrap(); - let pending = match prepared.execute().await { - Err(InvocationError::Pending(pending)) => pending, - other => panic!("expected pending after reply loss, got {other:?}"), - }; - assert!(matches!( - client.resolve(&pending).await.unwrap(), - Resolution::Committed(_) - )); - - let ordinary = ReferenceClient::new(fixture.typed.clone()).unwrap(); - let order = ordinary.orders(&OrderId(b"same-order".to_vec())).unwrap(); - order.receive_cron(identity, input).await.unwrap(); - assert_eq!(order.receipt_count(None, ()).await.unwrap().output, 1); - fixture.shutdown().await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn three_node_host_recovers_published_state_after_owner_loss() { - let fixture = PerfFixture::start(3).await; - let original = ReferenceClient::new(fixture.typed.clone()).unwrap(); - let order = original - .orders(&OrderId(b"surviving-order".to_vec())) - .unwrap(); - order - .receive_cron(reference_identity(74, now_ms()), invocation(1)) - .await - .unwrap(); - assert_eq!(order.receipt_count(None, ()).await.unwrap().output, 1); - - let (recovered, _recovered_directory) = recover_sql_on_second_node(&fixture).await; - let old_handle = fixture.owned_handles[0] - .iter() - .find(|handle| handle.cell_id() == fixture.sql_target.cell_id()) - .unwrap() - .clone(); - let stale = fixture.nodes[0] - .application_handle::( - CellClient::local(Arc::clone(&fixture.registry), old_handle), - fixture.sql_target.tenant(), - fixture.sql_target.application(), - ) - .unwrap(); - let stale = ReferenceClient::new(stale).unwrap(); - assert!(matches!( - stale - .orders(&OrderId(b"stale-owner".to_vec())) - .unwrap() - .receipt_count(None, ()) - .await, - Err(InvocationError::NotStarted(Error::Fenced)) - )); - let order = recovered - .orders(&OrderId(b"surviving-order".to_vec())) - .unwrap(); - assert_eq!(order.receipt_count(None, ()).await.unwrap().output, 1); - order - .receive_cron(reference_identity(75, now_ms()), invocation(2)) - .await - .unwrap(); - assert_eq!(order.receipt_count(None, ()).await.unwrap().output, 2); - fixture.shutdown().await; -} - -async fn recover_sql_on_second_node(fixture: &PerfFixture) -> (ReferenceClient, tempfile::TempDir) { - fixture.lose_owner(0); - let layout = fixture.layout.as_ref().unwrap(); - let target = &fixture.sql_target; - let authority = CellAuthority::new(layout.clone()); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let observed = authority.load(target.cell_id()).await.unwrap().unwrap(); - seed_reference_session(layout, node_session(0)).await; - let fence = fence_reference_session(layout, node_session(0), node_session(1)) - .await - .direct_takeover() - .unwrap(); - let limits = reference_limits(target.namespace()).unwrap(); - let recovered_directory = tempfile::TempDir::new().unwrap(); - let recovered = fixture.nodes[1] - .runtime() - .takeover_restored( - catalog.lookup(target.cell_id()).await.unwrap().unwrap(), - CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *IncarnationId::from_bytes([40; 16]).as_bytes(), - limits, - ) - .unwrap(), - authority, - observed, - fence, - RecoveryManifestStore::new(layout.clone(), limits), - recovered_directory.path().join("recovered-sql.sqlite"), - Owner { - session: node_session(1), - endpoint: "https://reference-recovered.internal:8081".into(), - }, - ) - .await - .unwrap(); - let client = CellClient::local(Arc::clone(&fixture.registry), recovered); - let handle = fixture.nodes[1] - .application_handle::(client, target.tenant(), target.application()) - .unwrap(); - let recovered = ReferenceClient::new(handle).unwrap(); - (recovered, recovered_directory) -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "manual action-level latency and recovery qualification"] -async fn reference_public_host_action_performance() { - run_public_host_action_performance(PerfFixture::start(3).await).await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "manual RustFS action-level latency and recovery qualification"] -async fn reference_public_host_rustfs_action_performance() { - let required = |name: &str| std::env::var(name).unwrap_or_else(|_| panic!("missing {name}")); - let store = super::performance_fixture::rustfs_store(); - let run_id = std::time::SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_nanos(); - let root = Path::from(format!( - "{}/public-host-{}-{run_id}", - required("CRAB_CELL_TEST_PREFIX"), - std::process::id() - )); - println!("PERF backend=rustfs object_prefix={root}"); - run_public_host_action_performance(PerfFixture::start_with_store(3, store, root).await).await; -} - -async fn run_public_host_action_performance(fixture: PerfFixture) { - let iterations = std::env::var("CRAB_CELL_PERF_ITERATIONS") - .ok() - .map(|value| value.parse::().unwrap()) - .unwrap_or(30); - assert!((1..=1_000).contains(&iterations)); - let sql_handle = fixture.owned_handles[0] - .iter() - .find(|handle| handle.cell_id() == fixture.sql_target.cell_id()) - .unwrap() - .clone(); - let local_handle = fixture.nodes[0] - .application_handle::( - CellClient::local(Arc::clone(&fixture.registry), sql_handle.clone()), - fixture.sql_target.tenant(), - fixture.sql_target.application(), - ) - .unwrap(); - let local = ReferenceClient::new(local_handle).unwrap(); - let signer = PeerSigner::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - fixture.registry.release_digest(), - SigningKey::from_bytes(&[78; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - crab_cell_runtime::SessionId::from_bytes([77; 16]), - fixture.registry.release_digest(), - signer.verifying_key(), - )); - let (owner_address, owner_server) = start_peer_server( - &fixture.registry, - Arc::clone(&verifier), - vec![sql_handle], - None, - ) - .await; - let gateway_listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); - let gateway_address = gateway_listener.local_addr().unwrap(); - let gateway_stats = Arc::new(GatewayStats::default()); - let routes = Arc::new(std::sync::RwLock::new(HashMap::from([( - fixture.sql_target.cell_id(), - owner_address, - )]))); - let gateway_server = start_gateway_peer_server( - gateway_listener, - &fixture.registry, - verifier, - fixture.owned_handles[1].clone(), - routes, - Arc::clone(&gateway_stats), - None, - ); - let forwarded = signed_client( - &fixture, - peer_round_trip(HashMap::from([( - fixture.sql_target.cell_id(), - gateway_address, - )])), - 1, - ); - for (lane, client, phase) in [("local", &local, 101_u8), ("forwarded", &forwarded, 102)] { - let order = client - .orders(&OrderId(b"performance-order".to_vec())) - .unwrap(); - let proof_baseline = fixture.durability[0].object_waits().len(); - let mut actions = Vec::with_capacity(iterations); - let mut durable_acks = Vec::with_capacity(iterations); - let lane_started = Instant::now(); - for index in 0..iterations { - let started = Instant::now(); - let occurrence = (phase as u64) * 10_000 + index as u64; - let prepared = order - .prepare_receive_cron(identity(phase, index, 0), invocation(occurrence)) - .await - .unwrap(); - let ack_started = Instant::now(); - let committed = prepared.execute().await.unwrap(); - durable_acks.push(ack_started.elapsed()); - let observed = order - .receipt_count(Some(committed.receipt), ()) - .await - .unwrap(); - assert_eq!( - observed.output, - if lane == "local" { - index + 1 - } else { - iterations + index + 1 - } as u64 - ); - actions.push(started.elapsed()); - } - let elapsed = lane_started.elapsed(); - report_samples(&format!("{lane}_verified_action"), &mut actions, elapsed); - report_samples( - &format!("{lane}_execute_to_durable_ack"), - &mut durable_acks, - elapsed, - ); - let mut proof_waits = fixture.durability[0].object_waits(); - let mut proof_waits = proof_waits.split_off(proof_baseline); - assert_eq!(proof_waits.len(), iterations); - report_samples( - &format!("{lane}_object_durability_proof_wait"), - &mut proof_waits, - elapsed, - ); - } - let (_, forwarded_count) = gateway_stats.counts(); - assert!(forwarded_count >= iterations * 3); - - let recovery_started = Instant::now(); - let (recovered, _recovered_directory) = recover_sql_on_second_node(&fixture).await; - let observed = recovered - .orders(&OrderId(b"performance-order".to_vec())) - .unwrap() - .receipt_count(None, ()) - .await - .unwrap(); - assert_eq!(observed.output, (iterations * 2) as u64); - let recovery_elapsed = recovery_started.elapsed(); - println!( - "PERF owner_loss_to_first_verified_read: duration_ms={:.3}", - recovery_elapsed.as_secs_f64() * 1_000.0 - ); - - gateway_server.abort(); - owner_server.abort(); - fixture.shutdown().await; -} diff --git a/crates/crab-cell-app/tests/reference_application/public_host/replicas.rs b/crates/crab-cell-app/tests/reference_application/public_host/replicas.rs deleted file mode 100644 index 5be6a74de..000000000 --- a/crates/crab-cell-app/tests/reference_application/public_host/replicas.rs +++ /dev/null @@ -1,365 +0,0 @@ -//! Public host publication hints and reader shutdown. - -use super::*; -use crate::reference_application::{performance_fixture, process_node}; -use futures_util::FutureExt; - -struct ReaderHints { - sent: tokio::sync::mpsc::UnboundedSender<( - crab_cell_runtime::CellId, - crab_cell_runtime::SessionId, - )>, - stalled: Option, - pending: Arc, -} - -struct PendingHint(Arc); - -impl Drop for PendingHint { - fn drop(&mut self) { - self.0.fetch_sub(1, std::sync::atomic::Ordering::Relaxed); - } -} - -impl PeerRoundTrip for ReaderHints { - fn send( - &self, - _: CellTarget, - _: Vec, - _: u32, - ) -> Pin>> + Send + 'static>> { - Box::pin(async { Err(Error::Peer("reader hint requires a selected node")) }) - } - - fn send_to_node( - &self, - target: CellTarget, - node: NodeAdvertisement, - _: Vec, - _: u32, - ) -> Pin>> + Send + 'static>> { - let session = node.session(); - let _ = self.sent.send((target.cell_id(), session)); - if Some(session) == self.stalled { - self.pending - .fetch_add(1, std::sync::atomic::Ordering::Relaxed); - let pending = PendingHint(self.pending.clone()); - return Box::pin(async move { - let _pending = pending; - std::future::pending().await - }); - } - Box::pin(async move { - use crab_cell_runtime::peer::{encode_peer_reply, wire}; - encode_peer_reply(&wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - receipt: Some(wire::Receipt { - cell_id: target.cell_id().as_bytes().to_vec(), - incarnation: vec![40; 16], - commit_sequence: 1, - }), - result: Some(wire::read_reply::Result::ReplicaReady(true)), - })), - }) - }) - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn published_command_wakes_reader_recruitment_before_periodic_scan() { - publication_hints(1, false).await; -} - -#[tokio::test(flavor = "multi_thread")] -async fn stalled_reader_does_not_delay_healthy_reader_publication_hints() { - publication_hints(2, true).await; -} - -#[tokio::test(flavor = "multi_thread")] -async fn publication_hints_reach_readers_beyond_the_activation_concurrency() { - publication_hints(20, true).await; -} - -async fn publication_hints(readers: usize, stalled_reader: bool) { - use crab_cell_runtime::{ - cell::application::ApplicationIdentity, node::lease::NodeLeaseGuard, - peer::ReplicaPeerClient, read_policy::ReadPolicyStore, - }; - use std::time::Duration; - - let application = Arc::new(compiled()); - let registry = application.registry(); - let tenant = TenantId::from_bytes([81; 16]); - let app = ApplicationId::from_bytes([82; 16]); - let root = tempfile::TempDir::new().unwrap(); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("publication-hints"), - *app.as_bytes(), - ); - let directory = process_node::directory(&layout, ®istry); - let now = now_ms(); - let count = readers + 1; - for node in 0..count { - directory - .create( - NodeAdvertisement::sign( - NodeId::from_bytes(*node_session(node).as_bytes()), - node_session(node), - format!("https://node-{node}.internal:8081"), - Digest::from_bytes([90; 32]), - Digest::from_bytes([94; 32]), - Digest::from_bytes([91; 32]), - registry.release_digest(), - &SigningKey::from_bytes(&[93; 32]), - 1, - now, - now + 15_000, - registry.module_digests(), - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 32 << 20, - free_disk_bytes: 64 << 20, - job_credits: 4, - ..Default::default() - }, - ) - .unwrap(), - now, - ) - .await - .unwrap(); - } - let node = crab_cell_host::CellNodeBuilder::new(application) - .with_runtime(SqlWorkerPool::new(2, 8).unwrap(), 64 << 20) - .with_session(node_session(0)) - .with_replica_host(reference_host()) - .build() - .unwrap(); - node.install_task_group( - tokio_util::sync::CancellationToken::new(), - tokio_util::sync::CancellationToken::new(), - ) - .unwrap(); - node.install_node_lease_for_startup(NodeLeaseGuard::new(now, now + 15_000).unwrap()) - .unwrap(); - let handle = bootstrap_reference_cell( - &node.runtime(), - ®istry, - &layout, - &root, - tenant, - app, - node_session(0), - SQL_NAMESPACE, - CatalogRole::Sql, - SQL_MODULE, - 40, - performance_fixture::install_sql_tables, - ) - .await - .unwrap(); - let target = CellTarget::new(tenant, app, SQL_NAMESPACE, &partition_for_shard(0)).unwrap(); - ReadPolicyStore::new(layout.clone()) - .create( - target.cell_id(), - IncarnationId::from_bytes([40; 16]), - count as u16 - 1, - ) - .await - .unwrap(); - node.install_read_replicas( - layout, - directory, - root.path().join("readers"), - Limits::default(), - ) - .unwrap(); - let (sent, mut hints) = tokio::sync::mpsc::unbounded_channel(); - let pending = Arc::new(std::sync::atomic::AtomicUsize::new(0)); - node.install_read_replica_recruitment( - ApplicationIdentity::new(tenant, app), - ReplicaPeerClient::new( - registry.clone(), - Arc::new(PeerSigner::new( - node_session(0), - registry.release_digest(), - SigningKey::from_bytes(&[93; 32]), - )), - PeerPrincipal { - issuer: "reference-runtime".into(), - subject: "owner".into(), - actions: vec!["cell.replica.activate".into()], - }, - Arc::new(ReaderHints { - sent, - stalled: stalled_reader.then_some(node_session(2)), - pending: pending.clone(), - }), - ), - ) - .unwrap(); - node.start().unwrap(); - // Always drain the host before propagating a failed assertion. A regression - // must not leave the intentionally stalled transport alive in the suite. - let observed = std::panic::AssertUnwindSafe(async { - // Consume the immediate periodic pass before publishing. The next tick is - // five seconds away, so only a publication wake-up can satisfy this bound. - for _ in 1..count { - tokio::time::timeout(Duration::from_secs(2), hints.recv()) - .await - .unwrap() - .unwrap(); - } - let client = CellClient::local(registry, handle); - let typed = node - .application_handle::(client, tenant, app) - .unwrap(); - let generated = ReferenceClient::new(typed).unwrap(); - let healthy = (1..count) - .filter(|node| !stalled_reader || *node != 2) - .map(node_session) - .collect::>(); - for occurrence in 1..=3 { - generated - .orders(&OrderId(b"publication-hints".to_vec())) - .unwrap() - .receive_cron(identity(108, occurrence, 0), invocation(occurrence as u64)) - .await - .unwrap(); - let notified = tokio::time::timeout(Duration::from_secs(2), async { - let mut received = std::collections::HashSet::new(); - for _ in 0..healthy.len() { - let (cell, session) = hints.recv().await.unwrap(); - assert_eq!(cell, target.cell_id()); - assert!( - received.insert(session), - "duplicate activation in one publication pass" - ); - } - received - }) - .await; - assert_eq!( - notified.unwrap(), - healthy, - "publication was blocked or repeated a pending activation" - ); - assert_eq!( - pending.load(std::sync::atomic::Ordering::Relaxed), - usize::from(stalled_reader) - ); - } - }) - .catch_unwind() - .await; - tokio::time::timeout(Duration::from_secs(2), node.shutdown()) - .await - .unwrap() - .unwrap(); - assert_eq!(pending.load(std::sync::atomic::Ordering::Relaxed), 0); - if let Err(failure) = observed { - std::panic::resume_unwind(failure); - } -} - -#[tokio::test] -async fn public_host_drain_cancels_reader_activation_waiting_on_storage() { - use object_store::throttle::{ThrottleConfig, ThrottledStore}; - use std::{task::Poll, time::Duration}; - - let application = Arc::new(compiled()); - let layout = CellStorageLayout::new( - Store::new(Arc::new(ThrottledStore::new( - InMemory::new(), - ThrottleConfig { - wait_get_per_call: Duration::from_secs(3600), - ..Default::default() - }, - ))), - Path::from("blocked-reader"), - [82; 16], - ); - let root = tempfile::TempDir::new().unwrap(); - let node = crab_cell_host::CellNodeBuilder::new(Arc::clone(&application)) - .with_runtime(SqlWorkerPool::new(1, 32).unwrap(), 64 << 20) - .with_replica_host(reference_host()) - .with_session(node_session(1)) - .build() - .unwrap(); - node.install_task_group( - tokio_util::sync::CancellationToken::new(), - tokio_util::sync::CancellationToken::new(), - ) - .unwrap(); - let manager = node - .install_read_replicas( - layout.clone(), - process_node::directory(&layout, &application.registry()), - root.path().to_owned(), - Limits::default(), - ) - .unwrap(); - let target = CellTarget::new( - TenantId::from_bytes([81; 16]), - ApplicationId::from_bytes([82; 16]), - SQL_NAMESPACE, - &partition_for_shard(0), - ) - .unwrap(); - let recruiter = node - .install_read_replica_recruitment( - crab_cell_runtime::cell::application::ApplicationIdentity::new( - target.tenant(), - target.application(), - ), - crab_cell_runtime::peer::ReplicaPeerClient::new( - application.registry(), - Arc::new(PeerSigner::new( - node_session(1), - application.registry().release_digest(), - SigningKey::from_bytes(&[93; 32]), - )), - crab_cell_runtime::peer::PeerPrincipal { - issuer: "reference-runtime".into(), - subject: "drain".into(), - actions: vec!["cell.replica.activate".into()], - }, - Arc::new(process_node::EnrolledReplicaTransport), - ), - ) - .unwrap(); - let recruitment = recruiter.reconcile(target.clone()); - tokio::pin!(recruitment); - std::future::poll_fn(|cx| { - assert!(recruitment.as_mut().poll(cx).is_pending()); - Poll::Ready(()) - }) - .await; - let activation = manager.activate(target.clone(), node_session(0)); - tokio::pin!(activation); - // Poll into the provider delay while activation owns its lane; no sleep - // or scheduler timing assumption is needed to put shutdown behind it. - std::future::poll_fn(|cx| { - assert!(activation.as_mut().poll(cx).is_pending()); - Poll::Ready(()) - }) - .await; - let (activated, recruited, drained) = tokio::time::timeout(Duration::from_secs(1), async { - tokio::join!(&mut activation, &mut recruitment, node.shutdown()) - }) - .await - .expect("reader drain waited for stalled storage"); - assert!(matches!(activated, Err(Error::RuntimeClosed))); - drained.unwrap(); - assert!(matches!(recruited, Err(Error::RuntimeClosed))); - assert!(matches!( - recruiter.reconcile_active().await, - Err(Error::RuntimeClosed) - )); - assert!(matches!( - manager.activate(target, node_session(0)).await, - Err(Error::RuntimeClosed) - )); -} diff --git a/crates/crab-cell-app/tests/reference_application/public_host/rollout.rs b/crates/crab-cell-app/tests/reference_application/public_host/rollout.rs deleted file mode 100644 index 56eb45c40..000000000 --- a/crates/crab-cell-app/tests/reference_application/public_host/rollout.rs +++ /dev/null @@ -1,346 +0,0 @@ -//! Additive application code, retained clients, and exact-root recovery. - -use super::*; -use crate::reference_application::performance_fixture::rustfs_store; -use crab_cell_host::CellNodeBuilder; -use crab_cell_runtime::node::lease::NodeLeaseGuard; -use crab_cell_runtime::registry::{Query, QueryContext, RetainedCodeDescriptor}; -use tokio_util::sync::CancellationToken; - -struct SuccessorSql; - -impl CellModule for SuccessorSql { - const NAME: &'static str = SQL_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: OnceLock = OnceLock::new(); - DESCRIPTOR.get_or_init(|| { - let predecessor = ReferenceSql.descriptor(); - let mut queries = predecessor.queries.to_vec(); - queries.push(operation(ReceiptPayload::ID)); - ModuleDescriptor { - source_digest: Digest::from_bytes(*blake3::hash(b"reference-sql-v2").as_bytes()), - retained_codes: Box::leak(Box::new([RetainedCodeDescriptor { - code: compiled().registry().module_code(SQL_MODULE).unwrap(), - schema_min: 1, - schema_max: 1, - }])), - queries: Box::leak(queries.into_boxed_slice()), - ..*predecessor - } - }) - } - - fn register(self, registry: &mut RegistryBuilder) -> Result<()> { - ReferenceSql.register(registry)?; - registry.bind_query::() - } -} - -struct ReceiptPayload; - -impl Query for ReceiptPayload { - const MODULE: &'static str = SQL_MODULE; - const ID: u32 = 8; - const CODEC_VERSION: u32 = 1; - type Input = u64; - type Output = Vec; - - fn execute(context: &mut QueryContext<'_>, occurrence: u64) -> Result> { - let occurrence = i64::try_from(occurrence) - .map_err(|_| Error::Command("receipt occurrence exceeds SQL integer range"))?; - let results = context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT payload FROM invoice_receipts WHERE occurrence = ?1".into(), - parameters: vec![SqlValue::Integer(occurrence)], - }], - })?; - match results[0].rows.first().and_then(|row| row.first()) { - Some(SqlValue::Blob(payload)) => Ok(payload.clone()), - _ => Err(Error::Command("receipt payload is unavailable")), - } - } -} - -struct SuccessorApplication; - -impl CellApplication for SuccessorApplication { - const NAME: &'static str = ReferenceApplication::NAME; - - fn register(builder: &mut ApplicationBuilder) -> Result<()> { - ReferenceApplication::register_with_sql(builder, SuccessorSql) - } -} - -crab_cell_app::cell_client! { - struct SuccessorClient (SuccessorApplication) { - fn orders(scope: &OrderId) -> SuccessorOrder { - namespace: SQL_NAMESPACE, - module: SQL_MODULE, - commands: { fn receive_cron, prepare_receive_cron: ReferenceCronReceiver = 6; }, - queries: { - fn receipt_count: ReferenceReceiptCount = 7; - fn receipt_payload: ReceiptPayload = 8; - } - } - } -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn three_node_host_additive_code_rollout_preserves_acknowledged_state() { - verify_rollout( - Store::new(Arc::new(InMemory::new())), - Path::from("additive-rollout"), - ) - .await; -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 4)] -#[ignore = "manual additive release qualification with real RustFS"] -async fn three_node_host_rustfs_additive_code_rollout() { - let root = std::env::var("CRAB_CELL_PERF_PROCESS_ROOT").unwrap(); - verify_rollout(rustfs_store(), Path::from(root)).await; -} - -async fn verify_rollout(store: Store, root: Path) { - let predecessor = compiled(); - let successor = Arc::new( - SuccessorApplication::compile(BuildDescriptor { - source_revision: "reference-additive-successor".into(), - cargo_lock_digest: Digest::from_bytes([42; 32]), - }) - .unwrap(), - ); - let registry = successor.registry(); - let old_code = predecessor.registry().module_code(SQL_MODULE).unwrap(); - let new_code = registry.module_code(SQL_MODULE).unwrap(); - assert_ne!(old_code, new_code); - registry - .verify_rolling_from(predecessor.registry().release_bytes()) - .unwrap(); - assert!( - predecessor - .registry() - .verify_rolling_from(registry.release_bytes()) - .is_err() - ); - - let fixture = PerfFixture::start_configured(3, Some(Arc::clone(&successor)), store, root).await; - let target = &fixture.sql_target; - let owner = fixture.owned_handles[0] - .iter() - .find(|handle| handle.cell_id() == target.cell_id()) - .unwrap() - .clone(); - assert_eq!(owner.code(), old_code); - let local = fixture.nodes[0] - .application_handle::( - CellClient::local(Arc::clone(®istry), owner.clone()), - target.tenant(), - target.application(), - ) - .unwrap(); - let local = SuccessorClient::new(local).unwrap(); - - // The old release enters the successor's actual dispatcher over TCP. Its - // retained code is executable until the operator publishes the new code. - let old = ReferenceClient::new(fixture.typed.clone()).unwrap(); - let key = OrderId(b"additive-rollout-order".to_vec()); - let old_order = old.orders(&key).unwrap(); - let local_order = local.orders(&key).unwrap(); - let original_identity = reference_identity(76, now_ms()); - let original_input = invocation(1); - let (original, added) = tokio::join!( - old_order.receive_cron(original_identity, original_input.clone()), - local_order.receive_cron(reference_identity(77, now_ms()), invocation(2)), - ); - let original = original.unwrap(); - added.unwrap(); - assert_eq!(old_order.receipt_count(None, ()).await.unwrap().output, 2); - assert_eq!( - local_order.receipt_payload(None, 1).await.unwrap().output, - original_input.payload - ); - let prepared_before_cutover = local_order - .prepare_receive_cron(reference_identity(78, now_ms()), invocation(3)) - .await - .unwrap(); - - let plan = registry - .next_migration(SQL_NAMESPACE, owner.code(), owner.schema()) - .unwrap() - .unwrap(); - let migrated = owner.migrate(plan, now_ms()).await.unwrap(); - assert_eq!(migrated.handle.code(), new_code); - assert!(matches!( - prepared_before_cutover.execute().await, - Err(InvocationError::NotStarted(Error::CellDraining)) - )); - let layout = fixture.layout.as_ref().unwrap(); - let authority = CellAuthority::new(layout.clone()); - let published = authority.load(target.cell_id()).await.unwrap().unwrap(); - assert_eq!(published.value().code, new_code); - assert_eq!( - published.value().root.as_ref().unwrap().commit_sequence, - migrated.outcome.commit_sequence - ); - let retired_client = fixture.nodes[1] - .application_handle::( - CellClient::local(predecessor.registry(), migrated.handle.clone()), - target.tenant(), - target.application(), - ) - .unwrap(); - let retired_client = ReferenceClient::new(retired_client).unwrap(); - assert!(matches!( - retired_client - .orders(&key) - .unwrap() - .receipt_count(None, ()) - .await, - Err(InvocationError::NotStarted(Error::Command( - "Cell code does not match operation module" - ))) - )); - - // Rebind the peer listener to the published capability and independently - // authenticate the upgraded client's release. No stale handle is reused. - let signer = Arc::new(PeerSigner::new( - node_session(2), - registry.release_digest(), - SigningKey::from_bytes(&[79; 32]), - )); - let verifier = Arc::new(PeerVerifier::new( - node_session(2), - registry.release_digest(), - signer.verifying_key(), - )); - let (address, server) = - start_peer_server(®istry, verifier, vec![migrated.handle.clone()], None).await; - let peer = CellClient::peer( - Arc::clone(®istry), - signer, - PeerPrincipal { - issuer: "reference-rollout".into(), - subject: "upgraded-client".into(), - actions: vec!["cell.read".into(), "cell.write".into()], - }, - peer_round_trip(HashMap::from([(target.cell_id(), address)])), - ); - let upgraded = fixture.nodes[0] - .application_handle::(peer, target.tenant(), target.application()) - .unwrap(); - let upgraded = SuccessorClient::new(upgraded).unwrap(); - let order = upgraded.orders(&key).unwrap(); - let replay = order - .receive_cron(original_identity, original_input.clone()) - .await - .unwrap(); - assert_eq!(replay.receipt, original.receipt); - order - .receive_cron(reference_identity(79, now_ms()), invocation(3)) - .await - .unwrap(); - assert_eq!(order.receipt_count(None, ()).await.unwrap().output, 3); - assert_eq!( - order - .receipt_payload(Some(replay.receipt), 1) - .await - .unwrap() - .output, - original_input.payload - ); - - // Retire an old host, then recover on a fresh successor host with no shared - // SQLite directory. The object root must carry both code and dedup state. - fixture.nodes[1].shutdown().await.unwrap(); - server.abort(); - migrated.handle.drain().await.unwrap(); - let replacement = CellNodeBuilder::new(successor) - .with_runtime(SqlWorkerPool::new(2, 8).unwrap(), 64 << 20) - .with_session(node_session(3)) - .with_replica_host(reference_host()) - .build() - .unwrap(); - replacement - .install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - replacement - .install_node_lease(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - let files = tempfile::TempDir::new().unwrap(); - let observed = authority.load(target.cell_id()).await.unwrap().unwrap(); - let restored = replacement - .runtime() - .acquire_idle_restored( - CellCatalog::new(layout.clone(), target.tenant()) - .lookup(target.cell_id()) - .await - .unwrap() - .unwrap(), - CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *IncarnationId::from_bytes([40; 16]).as_bytes(), - reference_limits(SQL_NAMESPACE).unwrap(), - ) - .unwrap(), - authority, - observed, - files.path().join("restored.sqlite"), - Owner { - session: node_session(3), - endpoint: "https://reference-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!(restored.code(), new_code); - let recovered = replacement - .application_handle::( - CellClient::local(registry, restored), - target.tenant(), - target.application(), - ) - .unwrap(); - let recovered = SuccessorClient::new(recovered).unwrap(); - let recovered_order = recovered.orders(&key).unwrap(); - let replay = recovered_order - .receive_cron(original_identity, original_input.clone()) - .await - .unwrap(); - assert_eq!(replay.receipt, original.receipt); - assert_eq!( - recovered_order - .receipt_count(None, ()) - .await - .unwrap() - .output, - 3 - ); - assert_eq!( - recovered_order - .receipt_payload(None, 1) - .await - .unwrap() - .output, - original_input.payload - ); - recovered_order - .receive_cron(reference_identity(80, now_ms()), invocation(4)) - .await - .unwrap(); - assert_eq!( - recovered_order - .receipt_count(None, ()) - .await - .unwrap() - .output, - 4 - ); - replacement.shutdown().await.unwrap(); - fixture.shutdown().await; - println!( - "ROLLOUT additive_code: old_code={old_code:?} new_code={new_code:?} exact_receipts=4 duplicate_replays=2 recovered=true" - ); -} diff --git a/crates/crab-cell-app/tests/support/generated_compile_setup.rs b/crates/crab-cell-app/tests/support/generated_compile_setup.rs deleted file mode 100644 index 6477060b4..000000000 --- a/crates/crab-cell-app/tests/support/generated_compile_setup.rs +++ /dev/null @@ -1,49 +0,0 @@ -use crab_cell_app::{ApplicationBuilder, CellApplication, CellKey, cell_client}; -use crab_cell_runtime::identity::NamespaceId; -use crab_cell_runtime::registry::{Command, CommandContext, CommandResult}; - -struct App; - -struct EntityKey([u8; 16]); - -impl CellKey for EntityKey { - fn canonical_bytes(&self) -> &[u8] { - &self.0 - } -} - -impl CellApplication for App { - const NAME: &'static str = "compile-proof"; - - fn register(_builder: &mut ApplicationBuilder) -> crab_cell_runtime::Result<()> { - Ok(()) - } -} - -struct Set; - -impl Command for Set { - const MODULE: &'static str = "entity"; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - type Input = (); - type Output = (); - - fn execute( - _context: &mut CommandContext<'_, '_>, - _input: Self::Input, - ) -> crab_cell_runtime::Result> { - Ok(CommandResult::Success(())) - } -} - -cell_client! { - struct Client (App) { - fn entity(scope: &EntityKey) -> Entity { - namespace: NamespaceId::from_bytes([1; 16]), - module: "entity", - commands: { fn set, prepare_set: Set = 1; }, - queries: { } - } - } -} diff --git a/crates/crab-cell-host/AGENTS.md b/crates/crab-cell-host/AGENTS.md deleted file mode 100644 index aebd867d7..000000000 --- a/crates/crab-cell-host/AGENTS.md +++ /dev/null @@ -1,18 +0,0 @@ -# AGENTS.md - -Scoped rules for `crates/crab-cell-host/`. Root and `crates/` guidance apply. - -- This is a provider-neutral lifecycle facade. HTTP, auth, provider - construction, and user authorization stay in product crates. -- One `CellNode` owns one `CellRuntime`; do not add a second scheduler, - authority, publisher, or durability path here. -- Builder validation must fail before starting the runtime. Shutdown and drain - must await the runtime and be safe to call once. -- A node keeps one drain lane: concurrent scale-down and shutdown callers queue - on it, and a caller's deadline bounds the releases it starts rather than the - wait for the lane. Fleet-level pacing stays with the planner's movement - budget, so do not add a second per-node rate limit here. -- Layout: the integration suite is `tests/node.rs` (with `tests/node/`). The - crate holds no in-src tests, so it has no `tests-allow-list.txt`. -- Run `python3 crab/scripts/check-cell-ltx-layout.py` after layout changes. - diff --git a/crates/crab-cell-host/CLAUDE.md b/crates/crab-cell-host/CLAUDE.md deleted file mode 120000 index 47dc3e3d8..000000000 --- a/crates/crab-cell-host/CLAUDE.md +++ /dev/null @@ -1 +0,0 @@ -AGENTS.md \ No newline at end of file diff --git a/crates/crab-cell-host/Cargo.toml b/crates/crab-cell-host/Cargo.toml deleted file mode 100644 index 7d3eab508..000000000 --- a/crates/crab-cell-host/Cargo.toml +++ /dev/null @@ -1,22 +0,0 @@ -[package] -name = "crab-cell-host" -version = "0.1.0" -edition.workspace = true -license.workspace = true -repository.workspace = true -rust-version.workspace = true -publish = false -description = "Provider-neutral Cell node lifecycle facade" - -[dependencies] -crab-cell-app.workspace = true -crab-cell-runtime.workspace = true -futures-util.workspace = true -tokio = { workspace = true, features = ["fs", "rt", "sync", "time"] } -tokio-util = { workspace = true, features = ["rt"] } -tracing.workspace = true -uuid.workspace = true - -[dev-dependencies] -tempfile.workspace = true -tokio = { workspace = true, features = ["macros", "rt-multi-thread"] } diff --git a/crates/crab-cell-host/README.md b/crates/crab-cell-host/README.md deleted file mode 100644 index 6c7c66be8..000000000 --- a/crates/crab-cell-host/README.md +++ /dev/null @@ -1,95 +0,0 @@ -# crab-cell-host - -Provider-neutral lifecycle boundary for a compiled Cell application. `CellNode` -owns one embedded runtime and its shared admission ledger, drives startup, -durability, scale down, and shutdown, and reports one lifecycle status. A -product server supplies providers, authentication, and network transports; it -must not construct a second runtime alongside this host. - -Shutdown first cancels admission and joins work producers. Lease maintenance -registered with `CellNodeTaskGroup::spawn_lease_maintenance` keeps renewing the -node session while the runtime drains accepted work and closes its covered -node log. Only then does the host cancel the node-shutdown token and join lease -maintenance, allowing session withdrawal. Both phases share the original task -limit and absolute shutdown deadline. Ordinary tasks stop on the work -cancellation token; lease maintenance stops on the node-shutdown token. - -Hosts admitting snapshot readers must also provision native-memory admission -on the `SqlWorkerPool` passed to `CellNodeBuilder::with_runtime`. The default -budget covers the configured writer count at 64 KiB per writer. Use -`SqlWorkerPool::with_native_memory_limit` to supply a larger explicit envelope -without increasing writer or file-descriptor capacity. Each read snapshot -currently reserves 12 MiB; refreshing can retain both old and replacement -snapshots. The reference Compose host reserves 32 MiB for native admission -and 64 MiB for retained cuts within its 1 GiB container limit. Admission -reservations are separate from measured RSS and the container memory ceiling. - -Call `CellNode::install_read_replicas` during startup after installing the task -group. It retains the shared `ReadReplicaManager`, supervises refresh and -placement eviction, and cancels activation before closing views during drain. -Pass the returned manager to the peer dispatcher as its replica resolver and -replica control implementation. -The operator supplies the application storage layout, signed directory, -private local root, and LTX limits. An authenticated owner hint calls -`activate`; it must still pass current owner and reader-selection checks. -Queries never activate or refresh a missing reader. - -The supervisor refreshes admitted views from published roots and removes views -that are no longer selected. Call `CellNode::install_read_replica_recruitment` -with the application identity and an activation-authorized `ReplicaPeerClient` -to recruit readers automatically for locally owned Cells. Recruitment scans -all compiled namespace roles and advances its cursor before I/O. The supervisor -retains at most 64 dirty Cells, one discovered candidate list, and 16 concurrent -activation hints across Cells. Discovery, queued hints, and activation share a -30-second deadline per prepared Cell. Explicit operator passes visit at most -64 Cells within 30 seconds. The five-second poll interval is not a freshness -or replacement SLO. Expired readers are -replaced through signed live membership and the same receiver admission path. -Pending activation hints recheck the selected boot at its observed lease -expiry. A renewal preserves the in-flight request; an expired or withdrawn -session releases the wait so a later pass can recruit its replacement. - -Successful object publication, activation, and schema migration also notify -recruitment through a bounded runtime channel. The host coalesces queued hints -per Cell and refreshes through the same signed peer path without waiting for -the next tick. Pending hints coalesce by Cell and node session; a stalled reader -does not prevent completed healthy readers from receiving subsequent hints. -Commands never await readers. Fleet-only acknowledgement sends -no publication hint until its root reaches object storage. Periodic scans -still repair dropped hints and reconcile membership and target changes; this -does not create a bounded-staleness guarantee. - -The task group owns recruitment alongside refresh; cancellation interrupts -provider and peer waits before drain. Explicit operator hints can use the -returned recruiter's `reconcile` method. HTTP authentication, administrative -policy, and repository scope stay in the server. The issue service and -independent reference hosts use these same implementations. - -## Module map - -| Module | Responsibility | -| --- | --- | -| `builder` | `CellNodeBuilder` validation and required-owner wiring | -| `node` | `CellNode`, its task group, lifecycle, qualification, and scale down | -| `read_replicas` | Selected immutable views, refresh, eviction, and terminal close | -| `read_replicas/recruitment` | Scoped owner recruitment, bounded fanout, and replacement | -| `durability` | Node-log durability supervision and rotation | -| `facility` | Facility registration and drained owners | -| `status` | `NodeState` and `NodeStatus` reporting | -| `tasks` | Bounded supervision for the node's facilities | - -## Tests - -`tests/node.rs` is the suite, with modules for builder validation, components, -lifecycle (including concurrent scale down), qualification, and task -supervision. The crate holds no in-src tests, so it has no -`tests-allow-list.txt`. - -```sh -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/ \ - cargo test -p crab-cell-host --locked -``` - -See `AGENTS.md` for contributor rules and -`crates/crab-cell-runtime/docs/runtime.md` for the runtime contract this facade -drives. diff --git a/crates/crab-cell-host/src/builder.rs b/crates/crab-cell-host/src/builder.rs deleted file mode 100644 index 636b4fec1..000000000 --- a/crates/crab-cell-host/src/builder.rs +++ /dev/null @@ -1,215 +0,0 @@ -//! Builder internals for the Cell node host. - -use super::*; - -/// Required inputs for one provider-neutral node. -pub struct CellNodeBuilder { - pub(super) application: Arc, - pub(super) pool: Option, - pub(super) replica_host: Option, - pub(super) session: Option, - pub(super) node_retained_bytes: Option, - pub(super) required_components: Vec<&'static str>, - pub(super) follower_store: Option<(PathBuf, ReplicaLimits, DiskBudget)>, -} - -pub(crate) struct CellNodeParts { - pub(super) application: Arc, - pub(super) pool: SqlWorkerPool, - pub(super) replica_host: ReplicaHost, - pub(super) session: SessionId, - pub(super) node_retained_bytes: usize, - pub(super) required_components: Vec<&'static str>, - pub(super) follower_store: Option<(PathBuf, ReplicaLimits, DiskBudget)>, -} - -impl CellNodeBuilder { - /// Starts a builder for one immutable compiled application. - #[must_use] - pub fn new(application: Arc) -> Self { - Self { - application, - pool: None, - replica_host: None, - session: None, - node_retained_bytes: None, - required_components: Vec::new(), - follower_store: None, - } - } - - /// Supplies the shared SQL worker pool and its runtime byte ceiling. - #[must_use] - pub fn with_runtime(mut self, pool: SqlWorkerPool, node_retained_bytes: usize) -> Self { - self.pool = Some(pool); - self.node_retained_bytes = Some(node_retained_bytes); - self - } - - /// Supplies the LTX host admission and local capacity policy. - #[must_use] - pub fn with_replica_host(mut self, host: ReplicaHost) -> Self { - self.replica_host = Some(host); - self - } - - /// Supplies the unique node session used by runtime ownership records. - #[must_use] - pub fn with_session(mut self, session: SessionId) -> Self { - self.session = Some(session); - self - } - - /// Declares the owned production components required before readiness. - pub fn with_required_owned_components( - mut self, - names: impl IntoIterator, - ) -> crab_cell_runtime::Result { - append_required_components(&mut self.required_components, names)?; - Ok(self) - } - - /// Supplies the durable follower store that the node must retain. - #[must_use] - pub fn with_follower_store( - mut self, - root: PathBuf, - limits: ReplicaLimits, - disk: DiskBudget, - ) -> Self { - self.follower_store = Some((root, limits, disk)); - self - } - - /// Validates all required inputs before starting any background runtime task. - pub fn build(self) -> crab_cell_runtime::Result { - let CellNodeParts { - application, - pool, - replica_host, - session, - node_retained_bytes, - required_components, - follower_store, - } = self.required_parts()?; - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - pool, - node_retained_bytes, - session, - replica_host, - )?; - install_application_limits(&runtime, &application)?; - let node = CellNode { - application, - runtime, - session, - state: Arc::new(Mutex::new(NodeState::Starting)), - lease_installed: AtomicBool::new(false), - shutdown_lock: Arc::new(tokio::sync::Mutex::new(())), - facilities: Arc::new(Mutex::new(Vec::new())), - required_components: Arc::new(Mutex::new(required_components)), - task_group: Arc::new(Mutex::new(None)), - }; - node.install_follower_store(follower_store)?; - Ok(node) - } - - /// Builds an unadvertised host for bounded offline maintenance. - /// - /// This path intentionally uses object-only runtime admission: the caller - /// must keep the host private and may not expose serving readiness. - pub fn build_unleased_for_maintenance(self) -> crab_cell_runtime::Result { - let CellNodeParts { - application, - pool, - replica_host, - session, - node_retained_bytes, - required_components, - follower_store, - } = self.required_parts()?; - let runtime = - CellRuntime::new_with_replica_host(pool, node_retained_bytes, session, replica_host)?; - install_application_limits(&runtime, &application)?; - let node = CellNode { - application, - runtime, - session, - state: Arc::new(Mutex::new(NodeState::Starting)), - lease_installed: AtomicBool::new(false), - shutdown_lock: Arc::new(tokio::sync::Mutex::new(())), - facilities: Arc::new(Mutex::new(Vec::new())), - required_components: Arc::new(Mutex::new(required_components)), - task_group: Arc::new(Mutex::new(None)), - }; - node.install_follower_store(follower_store)?; - Ok(node) - } - - pub(super) fn required_parts(self) -> crab_cell_runtime::Result { - let pool = self - .pool - .ok_or(Error::Control("CellNode requires a SQL worker pool"))?; - let host = self - .replica_host - .ok_or(Error::Control("CellNode requires an LTX replica host"))?; - let session = self - .session - .ok_or(Error::Control("CellNode requires a node session"))?; - if session.as_bytes().iter().all(|byte| *byte == 0) { - return Err(Error::Control("CellNode node session is zero")); - } - let node_retained_bytes = self - .node_retained_bytes - .filter(|bytes| *bytes != 0) - .ok_or(Error::Control("CellNode retained-byte ceiling is missing"))?; - Ok(CellNodeParts { - application: self.application, - pool, - replica_host: host, - session, - node_retained_bytes, - required_components: self.required_components, - follower_store: self.follower_store, - }) - } -} - -fn install_application_limits( - runtime: &CellRuntime, - application: &CompiledApplication, -) -> crab_cell_runtime::Result<()> { - runtime.install_application_limits(application.cell_types().iter().map(|cell_type| { - ( - cell_type.namespace(), - cell_type.database_limit_bytes(), - cell_type.capture_limit_bytes(), - ) - })) -} - -pub(crate) fn append_required_components( - required: &mut Vec<&'static str>, - names: impl IntoIterator, -) -> crab_cell_runtime::Result<()> { - let names = names.into_iter().collect::>(); - if names.is_empty() || names.len() > MAX_NODE_FACILITIES { - return Err(Error::Control( - "CellNode required component count is out of bounds", - )); - } - let mut seen = HashSet::with_capacity(names.len()); - if names - .iter() - .any(|name| name.is_empty() || required.contains(name) || !seen.insert(*name)) - { - return Err(Error::Control( - "CellNode required component names must be unique and non-empty", - )); - } - if required.len().saturating_add(names.len()) > MAX_NODE_FACILITIES { - return Err(Error::Capacity("CellNode required component limit reached")); - } - required.extend(names); - Ok(()) -} diff --git a/crates/crab-cell-host/src/durability.rs b/crates/crab-cell-host/src/durability.rs deleted file mode 100644 index 006fcd3ad..000000000 --- a/crates/crab-cell-host/src/durability.rs +++ /dev/null @@ -1,222 +0,0 @@ -//! Durability internals for the Cell node host. - -use super::*; - -/// Error returned by a provider-owned node facility during drain. -pub type FacilityResult = std::result::Result>; - -/// Node-log rotation events emitted by the host supervisor. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum NodeDurabilityRotation { - /// The supervisor is retiring the current node-log generation. - Started, - /// Shutdown is waiting for pending publications to settle. - Pending, - /// A rotation step failed and this rotation is being abandoned. - Failed, - /// The replacement generation is installed and serving. - Completed, -} - -/// Provider-owned enrollment adapter used by the host durability supervisor. -/// -/// The provider is responsible for authority and transport enrollment. The -/// host consumes the resulting provider-neutral configuration and is the only -/// owner that constructs and installs [`crab_cell_runtime::node::durability::NodeDurability`]. -pub trait NodeDurabilityProvider: Send + Sync + 'static { - /// Recruits one enrollment round for the replacement generation. - /// - /// `Ok(None)` means the provider is not ready and the supervisor should - /// ask again after its recruit interval. - fn recruit( - self: Arc, - limits: ReplicaLimits, - required_follower_bytes: u64, - live_node_limit: usize, - ) -> Pin>> + Send>>; - - /// Reports one rotation event to the provider; the default ignores it. - fn rotation_event(&self, _event: NodeDurabilityRotation) {} -} - -/// Fixed host-owned bounds and identity for the node-log supervisor. -#[derive(Clone, Copy, Debug)] -pub struct NodeDurabilitySupervisorConfig { - pub(super) application: ApplicationId, - pub(super) limits: ReplicaLimits, - pub(super) required_follower_bytes: u64, - pub(super) live_node_limit: usize, - pub(super) recruit_interval: std::time::Duration, - pub(super) rotation_interval: std::time::Duration, - pub(super) max_issued_frames: u64, -} - -impl NodeDurabilitySupervisorConfig { - /// Creates a bounded supervisor configuration. - pub fn new( - application: ApplicationId, - limits: ReplicaLimits, - required_follower_bytes: u64, - live_node_limit: usize, - recruit_interval: std::time::Duration, - rotation_interval: std::time::Duration, - max_issued_frames: u64, - ) -> crab_cell_runtime::Result { - if application.as_bytes().iter().all(|byte| *byte == 0) - || required_follower_bytes == 0 - || live_node_limit == 0 - || recruit_interval.is_zero() - || rotation_interval.is_zero() - || max_issued_frames == 0 - { - return Err(Error::Control( - "invalid CellNode durability supervisor configuration", - )); - } - Ok(Self { - application, - limits, - required_follower_bytes, - live_node_limit, - recruit_interval, - rotation_interval, - max_issued_frames, - }) - } -} - -pub(crate) async fn run_node_durability_supervisor

( - provider: Arc

, - runtime: CellRuntime, - configuration: NodeDurabilitySupervisorConfig, - cancellation: CancellationToken, -) -> FacilityResult -where - P: NodeDurabilityProvider, -{ - let mut recruit = tokio::time::interval(configuration.recruit_interval); - let mut rotation = tokio::time::interval(configuration.rotation_interval); - loop { - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - _ = recruit.tick(), if runtime.node_durability().is_none() => { - match provider.clone().recruit( - configuration.limits, - configuration.required_follower_bytes, - configuration.live_node_limit, - ).await { - Ok(Some(config)) => { - match config.build() { - Ok(durability) => { - if let Err(error) = runtime.install_node_durability( - configuration.application, - durability, - ) { - provider.rotation_event(NodeDurabilityRotation::Failed); - return Err(Box::new(error)); - } - } - Err(_error) => { - provider.rotation_event(NodeDurabilityRotation::Failed); - } - } - } - Ok(None) => {} - Err(_) => provider.rotation_event(NodeDurabilityRotation::Failed), - } - } - _ = rotation.tick(), if runtime.node_durability().is_some() => { - rotate_node_durability( - Arc::clone(&provider), - runtime.clone(), - configuration, - cancellation.clone(), - ).await?; - } - } - } -} - -pub(crate) async fn rotate_node_durability

( - provider: Arc

, - runtime: CellRuntime, - configuration: NodeDurabilitySupervisorConfig, - cancellation: CancellationToken, -) -> FacilityResult -where - P: NodeDurabilityProvider, -{ - let Some((application, durability)) = runtime.node_durability() else { - return Ok(()); - }; - if application != configuration.application { - return Err(Box::new(Error::Control( - "CellNode node durability application changed during rotation", - ))); - } - if !durability.needs_rotation(configuration.max_issued_frames) { - return Ok(()); - } - provider.rotation_event(NodeDurabilityRotation::Started); - loop { - match durability.shutdown().await { - Ok(()) => break, - Err(Error::PendingPublication) => { - provider.rotation_event(NodeDurabilityRotation::Pending); - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - () = tokio::time::sleep(configuration.recruit_interval) => {} - } - } - Err(error) => { - provider.rotation_event(NodeDurabilityRotation::Failed); - return Err(Box::new(error)); - } - } - } - if cancellation.is_cancelled() { - return Ok(()); - } - let replacement = loop { - match provider - .clone() - .recruit( - configuration.limits, - configuration.required_follower_bytes, - configuration.live_node_limit, - ) - .await - { - Ok(Some(config)) => match config.build() { - Ok(durability) => break durability, - Err(_) => provider.rotation_event(NodeDurabilityRotation::Failed), - }, - Ok(None) => {} - Err(_) => provider.rotation_event(NodeDurabilityRotation::Failed), - } - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - () = tokio::time::sleep(configuration.recruit_interval) => {} - } - }; - if cancellation.is_cancelled() { - replacement - .shutdown() - .await - .map_err(|error| Box::new(error) as Box)?; - return Ok(()); - } - match runtime.replace_node_durability(configuration.application, Arc::clone(&replacement)) { - Ok(_) => { - provider.rotation_event(NodeDurabilityRotation::Completed); - Ok(()) - } - Err(error) => { - provider.rotation_event(NodeDurabilityRotation::Failed); - replacement.shutdown().await.map_err(|shutdown_error| { - Box::new(shutdown_error) as Box - })?; - Err(Box::new(error)) - } - } -} diff --git a/crates/crab-cell-host/src/facility.rs b/crates/crab-cell-host/src/facility.rs deleted file mode 100644 index 23b4ecf68..000000000 --- a/crates/crab-cell-host/src/facility.rs +++ /dev/null @@ -1,55 +0,0 @@ -//! Facility internals for the Cell node host. - -use super::*; - -/// One provider-owned lifecycle component attached to a [`CellNode`]. -pub struct CellNodeFacility { - pub(super) name: &'static str, - pub(super) owner: Option>, - pub(super) drain: - Arc Pin + Send>> + Send + Sync>, -} - -impl CellNodeFacility { - /// Creates one named drain callback. The callback must be idempotent and - /// must finish promptly when the node's shutdown cancellation is observed. - pub fn new(name: &'static str, drain: F) -> crab_cell_runtime::Result - where - F: Fn() -> Fut + Send + Sync + 'static, - Fut: Future + Send + 'static, - { - if name.is_empty() { - return Err(Error::Control("CellNodeFacility name is empty")); - } - Ok(Self { - name, - owner: None, - drain: Arc::new(move || Box::pin(drain())), - }) - } - - /// Creates one named node-owned component with an idempotent drain callback. - pub fn owned( - name: &'static str, - owner: Arc, - drain: F, - ) -> crab_cell_runtime::Result - where - T: Send + Sync + 'static, - F: Fn() -> Fut + Send + Sync + 'static, - Fut: Future + Send + 'static, - { - if name.is_empty() { - return Err(Error::Control("CellNodeFacility name is empty")); - } - Ok(Self { - name, - owner: Some(owner), - drain: Arc::new(move || Box::pin(drain())), - }) - } - - pub(super) fn owner(&self) -> Option> { - self.owner.as_ref()?.clone().downcast::().ok() - } -} diff --git a/crates/crab-cell-host/src/lib.rs b/crates/crab-cell-host/src/lib.rs deleted file mode 100644 index 03115788c..000000000 --- a/crates/crab-cell-host/src/lib.rs +++ /dev/null @@ -1,72 +0,0 @@ -//! Provider-neutral lifecycle boundary for a compiled Cell application. -//! -//! `CellNode` owns the embedded runtime and its shared admission ledger. A -//! product server supplies providers, authentication and network transports; -//! it must not construct another runtime alongside this host. - -#![deny(missing_docs)] -// A panic in a filter process or FUSE path corrupts a worktree, so production -// builds deny unwrap, expect, panic, todo, and unimplemented; test builds keep -// them available. -#![cfg_attr( - not(test), - deny( - clippy::unwrap_used, - clippy::expect_used, - clippy::panic, - clippy::todo, - clippy::unimplemented - ) -)] - -pub use builder::CellNodeBuilder; -pub use durability::{ - FacilityResult, NodeDurabilityProvider, NodeDurabilityRotation, NodeDurabilitySupervisorConfig, -}; -pub use facility::CellNodeFacility; -pub use node::CellNode; -pub use status::{NodeState, NodeStatus, ScaleDownStatus}; -use std::{ - any::Any, - collections::HashSet, - future::Future, - path::PathBuf, - pin::Pin, - sync::{ - Arc, Mutex, - atomic::{AtomicBool, Ordering}, - }, - time::{Duration, Instant}, -}; -pub use tasks::CellNodeTaskGroup; -mod builder; -mod durability; -mod facility; -mod node; -pub mod read_replicas; -mod status; -mod tasks; - -use crab_cell_app::{ApplicationHandle, CellApplication, CompiledApplication}; -use crab_cell_runtime::Error; -use crab_cell_runtime::cell::actor::{CellRuntime, CellRuntimeStats}; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::client::CellClient; -use crab_cell_runtime::follower::FollowerStore; -use crab_cell_runtime::identity::{ApplicationId, CellId, SessionId, TenantId}; -use crab_cell_runtime::ltx::DiskBudget; -use crab_cell_runtime::ltx::{Host as ReplicaHost, Limits as ReplicaLimits}; -use crab_cell_runtime::node::durability::NodeDurabilityConfig; -use crab_cell_runtime::qualification::{ - QualificationOperationExecutor, QualificationRunSummary, QualificationWorkload, -}; -use tokio::task::JoinHandle; -use tokio_util::sync::CancellationToken; - -const MAX_NODE_FACILITIES: usize = 64; -const MAX_NODE_TASKS: usize = 256; - -/// Stable host-owned component name for the follower store. -pub const FOLLOWER_STORE_COMPONENT: &str = "follower-store"; -/// Stable host-owned component name for the node-log enrollment provider. -pub const NODE_DURABILITY_PROVIDER_COMPONENT: &str = "node-durability-provider"; diff --git a/crates/crab-cell-host/src/node.rs b/crates/crab-cell-host/src/node.rs deleted file mode 100644 index ff1acf3bb..000000000 --- a/crates/crab-cell-host/src/node.rs +++ /dev/null @@ -1,84 +0,0 @@ -//! Node internals for the Cell node host. - -use super::*; -use crate::builder::append_required_components; -use crate::durability::run_node_durability_supervisor; - -/// One started application host with an ordered drain/shutdown boundary. -pub struct CellNode { - pub(super) application: Arc, - pub(super) runtime: CellRuntime, - pub(super) session: SessionId, - pub(super) state: Arc>, - pub(super) lease_installed: AtomicBool, - pub(super) shutdown_lock: Arc>, - pub(super) facilities: Arc>>, - pub(super) required_components: Arc>>, - pub(super) task_group: Arc>>>, -} -impl CellNode { - /// Returns the compiled application artifact owned by this node. - #[must_use] - pub fn application(&self) -> &CompiledApplication { - &self.application - } - /// Returns the node-owned runtime for operator telemetry and admission. - #[must_use] - pub fn runtime(&self) -> CellRuntime { - self.runtime.clone() - } - /// Returns the current lifecycle state. - #[must_use] - pub fn state(&self) -> NodeState { - self.state - .lock() - .map(|state| *state) - .unwrap_or(NodeState::Stopped) - } - /// Returns whether the node has installed its lease and accepts work. - #[must_use] - pub fn is_ready(&self) -> bool { - matches!(self.state(), NodeState::Ready | NodeState::ScalingDown) - && self - .task_group - .lock() - .ok() - .and_then(|task_group| task_group.as_ref().map(|group| group.is_healthy())) - .unwrap_or(false) - } - /// Returns current shared runtime admission metrics. - #[must_use] - pub fn stats(&self) -> CellRuntimeStats { - self.runtime.stats() - } - /// Returns one coherent lifecycle and admission snapshot. - #[must_use] - pub fn status(&self) -> NodeStatus { - NodeStatus { - state: self.state(), - shutting_down: self.is_shutting_down(), - stats: self.stats(), - } - } - /// Returns whether the owned runtime has entered shutdown. - #[must_use] - pub fn is_shutting_down(&self) -> bool { - self.runtime.is_shutting_down() - } - /// Binds a product-created typed client to this application's tenant scope. - /// - /// Rejects a client built from a different compiled registry. - pub fn application_handle( - &self, - client: CellClient, - tenant: TenantId, - application: ApplicationId, - ) -> crab_cell_runtime::Result> { - ApplicationHandle::new(client, Arc::clone(&self.application), tenant, application) - } -} - -mod components; -mod lifecycle; -mod qualification; -mod scale_down; diff --git a/crates/crab-cell-host/src/node/components.rs b/crates/crab-cell-host/src/node/components.rs deleted file mode 100644 index 2d5450a65..000000000 --- a/crates/crab-cell-host/src/node/components.rs +++ /dev/null @@ -1,327 +0,0 @@ -//! Telemetry, lease, task-group, facility, and component installation. - -use super::*; - -impl CellNode { - /// Installs the product's metrics adapter before the node is advertised. - pub fn install_telemetry( - &self, - telemetry: Arc, - ) -> crab_cell_runtime::Result<()> { - self.runtime.install_telemetry(telemetry) - } - - /// Installs the authoritative node lease before readiness is exposed. - /// - /// A coordination task group must already be installed. This convenience - /// path is intentionally fail-closed so a node can never advertise while - /// its runtime-wide supervisors are unowned. - pub fn install_node_lease( - &self, - lease: crab_cell_runtime::NodeLeaseGuard, - ) -> crab_cell_runtime::Result<()> { - self.require_task_group()?; - self.install_node_lease_for_startup(lease)?; - self.start() - } - - /// Installs the lease without opening readiness to the product boundary. - /// - /// Servers use this during startup, then call [`Self::start`] only - /// after their listeners and owned facilities have been started. - pub fn install_node_lease_for_startup( - &self, - lease: crab_cell_runtime::NodeLeaseGuard, - ) -> crab_cell_runtime::Result<()> { - self.runtime.install_node_lease(lease)?; - self.lease_installed.store(true, Ordering::Release); - Ok(()) - } - - /// Declares the typed production components that must be retained before - /// readiness can open. The declaration is immutable after startup begins; - /// this keeps a product adapter from accidentally starting a node with a - /// missing owner and discovering the gap only on its first request. - pub fn require_owned_components( - &self, - names: impl IntoIterator, - ) -> crab_cell_runtime::Result<()> { - if self.state() != NodeState::Starting { - return Err(Error::CellDraining); - } - let mut required = self - .required_components - .lock() - .map_err(|_| Error::Control("CellNode required-component lock poisoned"))?; - append_required_components(&mut required, names) - } - - /// Creates and retains the bounded coordination task group for this node. - pub fn install_task_group( - &self, - cancellation: CancellationToken, - node_shutdown: CancellationToken, - ) -> crab_cell_runtime::Result> { - let task_group = Arc::new(CellNodeTaskGroup::new(cancellation, node_shutdown)); - let mut installed = self - .task_group - .lock() - .map_err(|_| Error::Control("CellNode task group lock poisoned"))?; - if installed.is_some() { - return Err(Error::Control("CellNode task group already installed")); - } - let drain_group = Arc::clone(&task_group); - let facility = CellNodeFacility::new("cell-coordination-tasks", move || { - let drain_group = Arc::clone(&drain_group); - async move { drain_group.drain().await } - })?; - self.install_facility(facility)?; - *installed = Some(Arc::clone(&task_group)); - Ok(task_group) - } - - pub(super) fn require_task_group(&self) -> crab_cell_runtime::Result<()> { - let installed = self - .task_group - .lock() - .map_err(|_| Error::Control("CellNode task group lock poisoned"))?; - if installed.is_none() { - return Err(Error::Control( - "CellNode cannot become ready before its task group is installed", - )); - } - Ok(()) - } - - /// Installs the provider enrollment adapter and moves node-log recruitment - /// and rotation into the host-owned task group. - pub fn install_node_durability_provider

( - &self, - provider: Arc

, - configuration: NodeDurabilitySupervisorConfig, - ) -> crab_cell_runtime::Result<()> - where - P: NodeDurabilityProvider, - { - let task_group = self - .task_group - .lock() - .map_err(|_| Error::Control("CellNode task group lock poisoned"))? - .clone() - .ok_or(Error::Control( - "CellNode durability provider requires an installed task group", - ))?; - self.install_owned_component(NODE_DURABILITY_PROVIDER_COMPONENT, Arc::clone(&provider))?; - let runtime = self.runtime.clone(); - let cancellation = task_group.cancellation.clone(); - let result = task_group.spawn_boxed(async move { - run_node_durability_supervisor(provider, runtime, configuration, cancellation).await - }); - if result.is_err() { - self.remove_facility(NODE_DURABILITY_PROVIDER_COMPONENT)?; - } - result - } - - /// Owns read-snapshot refresh, eviction, and terminal close for this node. - /// - /// Install during startup after the task group. The product supplies an - /// application-scoped store, signed directory, private local root, and LTX - /// bounds; it authorizes owner hints and wires the returned peer resolver. - pub fn install_read_replicas( - &self, - layout: crab_cell_runtime::ltx::CellStorageLayout, - directory: crab_cell_runtime::node::NodeDirectory, - root: PathBuf, - limits: ReplicaLimits, - ) -> crab_cell_runtime::Result { - const COMPONENT: &str = "read-replicas"; - let tasks = self - .task_group - .lock() - .map_err(|_| Error::Control("CellNode task group lock poisoned"))? - .clone() - .ok_or(Error::Control( - "CellNode read replicas require an installed task group", - ))?; - let manager = crate::read_replicas::ReadReplicaManager::new( - self.runtime.clone(), - self.application.registry(), - layout, - directory, - self.session, - root, - limits, - ); - let drained = manager.clone(); - self.install_owned_component_with_drain(COMPONENT, Arc::new(manager.clone()), move || { - let drained = drained.clone(); - async move { - drained.shutdown().await; - Ok(()) - } - })?; - let supervised = manager.clone(); - let cancellation = tasks.cancellation_token(); - if let Err(error) = tasks.spawn(async move { supervised.run(cancellation).await }) { - self.remove_facility(COMPONENT)?; - return Err(error); - } - Ok(manager) - } - - /// Supervises owner-side reader recruitment after installing read replicas. - /// - /// The product supplies application scope and an activation-authorized peer - /// client. The host retains the cursor and cancels recruitment before drain. - pub fn install_read_replica_recruitment( - &self, - identity: crab_cell_runtime::cell::application::ApplicationIdentity, - peer: crab_cell_runtime::peer::ReplicaPeerClient, - ) -> crab_cell_runtime::Result { - const COMPONENT: &str = "read-replica-recruitment"; - let readers = self - .owned_component::("read-replicas") - .ok_or(Error::Control( - "read recruitment requires installed read replicas", - ))?; - let tasks = self - .task_group - .lock() - .map_err(|_| Error::Control("CellNode task group lock poisoned"))? - .clone() - .ok_or(Error::Control( - "read recruitment requires installed task group", - ))?; - let recruiter = - crate::read_replicas::ReadReplicaRecruiter::new((*readers).clone(), identity, peer)?; - self.install_owned_component(COMPONENT, Arc::new(recruiter.clone()))?; - let supervised = recruiter.clone(); - let cancellation = tasks.cancellation_token(); - if let Err(error) = tasks.spawn(async move { supervised.run(cancellation).await }) { - self.remove_facility(COMPONENT)?; - return Err(error); - } - Ok(recruiter) - } - - pub(super) fn remove_facility(&self, name: &'static str) -> crab_cell_runtime::Result<()> { - let mut facilities = self - .facilities - .lock() - .map_err(|_| Error::Control("CellNode facility lock poisoned"))?; - let Some(index) = facilities.iter().position(|facility| facility.name == name) else { - return Err(Error::Control( - "CellNode facility rollback target is missing", - )); - }; - facilities.remove(index); - Ok(()) - } - - /// Attaches one provider-owned lifecycle component during node startup. - pub fn install_facility(&self, facility: CellNodeFacility) -> crab_cell_runtime::Result<()> { - self.install_facilities(std::iter::once(facility)) - } - - /// Atomically attaches a bounded batch of provider-owned facilities. - /// - /// All names and capacity are validated before any facility is retained, so - /// a failed composition cannot leave the node with a partial owner set. - /// Registration closes when readiness opens so the owner set cannot change - /// underneath admitted requests. - pub fn install_facilities( - &self, - facilities: impl IntoIterator, - ) -> crab_cell_runtime::Result<()> { - let mut additions = Vec::new(); - for facility in facilities { - if additions.len() >= MAX_NODE_FACILITIES { - return Err(Error::Capacity("CellNode facility limit reached")); - } - additions.push(facility); - } - let state = self - .state - .lock() - .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; - if *state != NodeState::Starting { - return Err(Error::CellDraining); - } - let mut facilities = self - .facilities - .lock() - .map_err(|_| Error::Control("CellNode facility lock poisoned"))?; - if facilities.len().saturating_add(additions.len()) > MAX_NODE_FACILITIES { - return Err(Error::Capacity("CellNode facility limit reached")); - } - let mut names = facilities - .iter() - .map(|facility| facility.name) - .collect::>(); - if additions - .iter() - .any(|facility| !names.insert(facility.name)) - { - return Err(Error::Control("CellNode facility name already installed")); - } - facilities.extend(additions); - Ok(()) - } - - /// Retains one shared composition component under the node lifecycle. - /// - /// Components are deliberately type-erased only inside the host. Callers - /// retrieve them by the same stable name and concrete type, while the - /// node remains the sole owner of the production composition boundary. - pub fn install_owned_component( - &self, - name: &'static str, - component: Arc, - ) -> crab_cell_runtime::Result<()> - where - T: Send + Sync + 'static, - { - self.install_owned_component_with_drain(name, component, || async { Ok(()) }) - } - - /// Retains one component and attaches its idempotent drain callback. - pub fn install_owned_component_with_drain( - &self, - name: &'static str, - component: Arc, - drain: F, - ) -> crab_cell_runtime::Result<()> - where - T: Send + Sync + 'static, - F: Fn() -> Fut + Send + Sync + 'static, - Fut: Future + Send + 'static, - { - self.install_facility(CellNodeFacility::owned(name, component, drain)?) - } - - pub(crate) fn install_follower_store( - &self, - configuration: Option<(PathBuf, ReplicaLimits, DiskBudget)>, - ) -> crab_cell_runtime::Result<()> { - let Some((root, limits, disk)) = configuration else { - return Ok(()); - }; - let store = FollowerStore::open(root, limits, disk)?; - self.install_owned_component(FOLLOWER_STORE_COMPONENT, Arc::new(store)) - } - - /// Looks up one node-owned component for a product adapter. - #[must_use] - pub fn owned_component(&self, name: &str) -> Option> - where - T: Send + Sync + 'static, - { - self.facilities - .lock() - .ok()? - .iter() - .find(|facility| facility.name == name) - .and_then(CellNodeFacility::owner) - } -} diff --git a/crates/crab-cell-host/src/node/lifecycle.rs b/crates/crab-cell-host/src/node/lifecycle.rs deleted file mode 100644 index 3149487e4..000000000 --- a/crates/crab-cell-host/src/node/lifecycle.rs +++ /dev/null @@ -1,199 +0,0 @@ -//! Startup, drain, and shutdown. - -use super::*; - -impl CellNode { - /// Opens readiness after all product startup probes have completed. - pub fn start(&self) -> crab_cell_runtime::Result<()> { - if !self.lease_installed.load(Ordering::Acquire) { - return Err(Error::Control( - "CellNode cannot become ready before its node lease is installed", - )); - } - self.require_task_group()?; - if !self - .task_group - .lock() - .map_err(|_| Error::Control("CellNode task group lock poisoned"))? - .as_ref() - .is_some_and(|task_group| task_group.is_healthy()) - { - return Err(Error::Control("CellNode task group is unhealthy")); - } - let mut state = self - .state - .lock() - .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; - if *state == NodeState::Starting { - self.require_components_present()?; - *state = NodeState::Ready; - return Ok(()); - } - if *state == NodeState::Ready { - return Ok(()); - } - Err(Error::Control( - "CellNode cannot become ready after shutdown", - )) - } - pub(super) fn require_components_present(&self) -> crab_cell_runtime::Result<()> { - let required = self - .required_components - .lock() - .map_err(|_| Error::Control("CellNode required-component lock poisoned"))? - .clone(); - if required.is_empty() { - return Ok(()); - } - let facilities = self - .facilities - .lock() - .map_err(|_| Error::Control("CellNode facility lock poisoned"))?; - for name in required { - if !facilities - .iter() - .any(|facility| facility.name == name && facility.owner.is_some()) - { - return Err(Error::Control("CellNode required component is missing")); - } - } - Ok(()) - } - /// Stops admission, drains the runtime, and waits for its dispatcher. - pub async fn drain(&self) -> crab_cell_runtime::Result<()> { - self.drain_until(None).await - } - /// Stops admission and completes every owned drain phase by `deadline`. - pub async fn drain_until(&self, deadline: Option) -> crab_cell_runtime::Result<()> { - let _shutdown = self.shutdown_lock.lock().await; - self.drain_until_locked(deadline).await - } - pub(super) async fn drain_until_locked( - &self, - deadline: Option, - ) -> crab_cell_runtime::Result<()> { - { - let mut state = self - .state - .lock() - .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; - if *state == NodeState::Stopped { - return Ok(()); - } - *state = NodeState::Draining; - } - let task_group = self - .task_group - .lock() - .map(|task_group| task_group.clone()) - .map_err(|_| Error::Control("CellNode task group lock poisoned")); - let facilities = self - .facilities - .lock() - .map(|facilities| { - facilities - .iter() - .rev() - .map(|facility| (facility.name, Arc::clone(&facility.drain))) - .collect::>() - }) - .map_err(|_| Error::Control("CellNode facility lock poisoned")); - let mut first_error = None; - if let Some(error) = task_group.as_ref().err().map(|error| match error { - Error::Control(message) => Error::Control(message), - _ => Error::Control("CellNode task group unavailable during drain"), - }) { - first_error = Some(error); - } - if let Ok(Some(task_group)) = task_group.as_ref() { - task_group.cancel_work(); - } - match facilities { - Err(error) if first_error.is_none() => first_error = Some(error), - Err(_) => {} - Ok(facilities) => { - for (name, drain) in facilities { - let result = if name == "cell-coordination-tasks" { - // The task group is a retained owner, so its join must share the - // node deadline; an unbounded callback could strand shutdown. - match task_group.as_ref() { - Ok(Some(task_group)) => task_group.drain_work_until(deadline).await, - _ => drain().await, - } - } else { - match deadline { - Some(deadline) => { - match tokio::time::timeout_at(deadline.into(), drain()).await { - Ok(result) => result, - Err(_) => Err(Box::new(std::io::Error::new( - std::io::ErrorKind::TimedOut, - "CellNode facility drain deadline exceeded", - )) - as Box), - } - } - None => drain().await, - } - }; - if let Err(source) = result - && first_error.is_none() - { - first_error = Some(Error::Facility { name, source }); - } - } - } - } - let runtime_result = match deadline { - Some(deadline) => { - match tokio::time::timeout_at(deadline.into(), self.runtime.shutdown()).await { - Ok(result) => result, - Err(_) => Err(Error::Control("CellNode runtime drain deadline exceeded")), - } - } - None => self.runtime.shutdown().await, - }; - if first_error.is_none() { - first_error = runtime_result.err(); - } - // Session withdrawal fences the log authority. Keep its heartbeat live - // until runtime publication and the durable log-close barrier finish. - if let Ok(Some(task_group)) = task_group - && let Err(source) = task_group.drain_until(deadline).await - && first_error.is_none() - { - first_error = Some(Error::Facility { - name: "cell-coordination-tasks", - source, - }); - } - let result = match first_error { - Some(error) => Err(error), - None => Ok(()), - }; - let result = if result.is_ok() { - match self.facilities.lock() { - Ok(mut facilities) => { - facilities.clear(); - Ok(()) - } - Err(_) => Err(Error::Control("CellNode facility lock poisoned")), - } - } else { - result - }; - if result.is_ok() - && let Ok(mut state) = self.state.lock() - { - *state = NodeState::Stopped; - } - result - } - /// Idempotent alias for graceful drain used by process shutdown hooks. - pub async fn shutdown(&self) -> crab_cell_runtime::Result<()> { - self.drain().await - } - /// Deadline-aware alias for graceful shutdown hooks. - pub async fn shutdown_until(&self, deadline: Instant) -> crab_cell_runtime::Result<()> { - self.drain_until(Some(deadline)).await - } -} diff --git a/crates/crab-cell-host/src/node/qualification.rs b/crates/crab-cell-host/src/node/qualification.rs deleted file mode 100644 index 3418e117d..000000000 --- a/crates/crab-cell-host/src/node/qualification.rs +++ /dev/null @@ -1,78 +0,0 @@ -//! Qualification runs over the node's runtime. - -use super::*; - -impl CellNode { - /// Runs a deterministic qualification workload while this node is ready. - /// - /// The typed executor remains responsible for primitive requests and - /// verification. The host only admits a run while serving and rejects a - /// result if shutdown or a supervisor failure removed readiness during the - /// run, so callers cannot retain evidence from a draining node. - pub async fn run_qualification( - &self, - workload: &QualificationWorkload, - executor: &mut E, - ) -> crab_cell_runtime::Result - where - E: QualificationOperationExecutor, - { - if !self.is_ready() { - return Err(Error::CellDraining); - } - let summary = workload.run_with_case_coverage(executor).await?; - if !self.is_ready() { - return Err(Error::CellDraining); - } - Ok(summary) - } - /// Runs an observed workload while this node is ready without asserting - /// that every scheduled lifecycle case was exercised. - /// - /// This entry point is for local wiring and smoke evidence. Its result must - /// not be promoted to a protected profile unless the resulting artifact - /// proves the profile's required case coverage independently. - pub async fn run_qualification_observed( - &self, - workload: &QualificationWorkload, - executor: &mut E, - ) -> crab_cell_runtime::Result - where - E: QualificationOperationExecutor, - { - if !self.is_ready() { - return Err(Error::CellDraining); - } - let summary = workload.run(executor).await?; - if !self.is_ready() { - return Err(Error::CellDraining); - } - Ok(summary) - } - /// Runs independent qualification operations with bounded concurrency. - /// - /// The application-specific executor must make each scheduled operation - /// independent or idempotent. Readiness is checked before admission and - /// after all in-flight work drains, so a result from a draining or failed - /// node is never accepted as qualification evidence. - pub async fn run_qualification_concurrent( - &self, - workload: &QualificationWorkload, - executor: E, - concurrency: usize, - ) -> crab_cell_runtime::Result - where - E: QualificationOperationExecutor + Clone + Send + 'static, - { - if !self.is_ready() { - return Err(Error::CellDraining); - } - let summary = workload - .run_concurrent_with_case_coverage(executor, concurrency) - .await?; - if !self.is_ready() { - return Err(Error::CellDraining); - } - Ok(summary) - } -} diff --git a/crates/crab-cell-host/src/node/scale_down.rs b/crates/crab-cell-host/src/node/scale_down.rs deleted file mode 100644 index 64adb5954..000000000 --- a/crates/crab-cell-host/src/node/scale_down.rs +++ /dev/null @@ -1,118 +0,0 @@ -//! Idle-cell movement for operator-driven scale down. - -use super::*; - -impl CellNode { - /// Lists settled local Cells as advisory candidates for the fleet planner. - pub async fn idle_transfer_candidates( - &self, - ) -> crab_cell_runtime::Result> { - if !self.is_ready() { - return Err(Error::CellDraining); - } - self.runtime.idle_transfer_candidates().await - } - /// Releases one exact settled Cell generation and waits for owner release. - pub async fn release_idle_cell( - &self, - cell: CellId, - source: SessionId, - generation: u64, - ) -> crab_cell_runtime::Result<()> { - if !self.is_ready() { - return Err(Error::CellDraining); - } - self.runtime - .release_idle_cell(cell, source, generation) - .await - } - /// Stops new Cell acquisition while retaining the lease and current owners. - pub fn begin_scale_down(&self) -> crab_cell_runtime::Result<()> { - let mut state = self - .state - .lock() - .map_err(|_| Error::Control("CellNode lifecycle lock poisoned"))?; - match *state { - NodeState::Ready => { - self.runtime.stop_acquiring()?; - *state = NodeState::ScalingDown; - Ok(()) - } - NodeState::ScalingDown => Ok(()), - _ => Err(Error::CellDraining), - } - } - /// Waits for confirmed releases. An incomplete result leaves the node - /// serving and its facilities alive for the next fleet planning window. - pub async fn drain_for_scale_down( - &self, - deadline: Instant, - ) -> crab_cell_runtime::Result { - let _shutdown = self.shutdown_lock.lock().await; - if self.state() == NodeState::Stopped { - return Ok(ScaleDownStatus { - remaining_cells: 0, - settled_candidates: 0, - released_cells: 0, - blocked_cells: 0, - }); - } - self.begin_scale_down()?; - let mut released_cells = 0_usize; - let mut blocked = HashSet::new(); - loop { - let candidates = self.runtime.idle_transfer_candidates().await?; - for (cell, generation, _, _) in candidates.iter().copied() { - if Instant::now() >= deadline { - break; - } - let result = tokio::time::timeout_at( - deadline.into(), - self.runtime - .release_idle_cell(cell, self.session, generation), - ) - .await; - match result { - Ok(Ok(())) => { - released_cells = released_cells.saturating_add(1); - blocked.remove(&cell); - } - Ok(Err(_)) => { - blocked.insert(cell); - } - Err(_) => break, - } - } - let remaining_cells = self.runtime.unreleased_cell_count().await?; - let current_candidates = self.runtime.idle_transfer_candidates().await?; - let settled_candidates = current_candidates.len(); - let candidate_ids = current_candidates - .iter() - .map(|(cell, _, _, _)| *cell) - .collect::>(); - blocked.retain(|cell| candidate_ids.contains(cell)); - let status = ScaleDownStatus { - remaining_cells, - settled_candidates, - released_cells, - blocked_cells: blocked - .len() - .saturating_add(remaining_cells.saturating_sub(settled_candidates)), - }; - if status.ready_to_stop() { - if Instant::now() >= deadline { - return Ok(status); - } - self.drain_until_locked(Some(deadline)).await?; - return Ok(status); - } - if Instant::now() >= deadline { - return Ok(status); - } - let wait = deadline - .saturating_duration_since(Instant::now()) - .min(Duration::from_secs(1)); - tokio::time::sleep(wait).await; - } - } -} diff --git a/crates/crab-cell-host/src/read_replicas.rs b/crates/crab-cell-host/src/read_replicas.rs deleted file mode 100644 index e74015fad..000000000 --- a/crates/crab-cell-host/src/read_replicas.rs +++ /dev/null @@ -1,422 +0,0 @@ -//! Node-owned read-only snapshots selected by current Cell policy. - -use std::{ - collections::HashMap, - future::Future, - path::PathBuf, - pin::Pin, - sync::Arc, - time::{Duration, SystemTime, UNIX_EPOCH}, -}; - -use crab_cell_runtime::{ - Error, Result, - cell::actor::CellRuntime, - client::{CellReadReplica, Receipt}, - control::{Control, ControlState, authority::CellAuthority}, - identity::{CellId, CellTarget, IncarnationId, SessionId}, - ltx::{CellReplica, CellStorageLayout, Limits}, - node::NodeDirectory, - peer::{PeerReplicaControl, PeerReplicaResolver}, - read_policy::{ReadPolicy, ReadPolicyStore}, - registry::Registry, -}; -use futures_util::future::BoxFuture; -use tokio::sync::{Mutex, RwLock}; -use tokio_util::sync::CancellationToken; -use uuid::Uuid; - -mod recruitment; -pub use recruitment::ReadReplicaRecruiter; - -const RECONCILE_INTERVAL: Duration = Duration::from_secs(5); -const RECONCILE_BATCH: usize = 64; -const RECONCILE_DEADLINE: Duration = Duration::from_secs(30); -const MAX_LIVE_NODES: usize = 10_000; - -/// Admitted immutable readers sharing one node runtime and current placement policy. -/// -/// Product adapters authorize activation hints and policy changes before calling -/// this manager; selection never grants write ownership. -#[derive(Clone)] -pub struct ReadReplicaManager { - runtime: CellRuntime, - registry: Arc, - layout: CellStorageLayout, - authority: CellAuthority, - directory: NodeDirectory, - policy: ReadPolicyStore, - session: SessionId, - root: PathBuf, - limits: Limits, - closed: CancellationToken, - activation: Arc>, - active: Arc>>, -} - -impl ReadReplicaManager { - /// Creates a reader manager for an existing node runtime. - /// - /// The caller must supervise `run` and invoke `shutdown` before runtime drain. - /// Prefer `CellNode::install_read_replicas` for automatic lifecycle ownership. - #[must_use] - pub fn new( - runtime: CellRuntime, - registry: Arc, - layout: CellStorageLayout, - directory: NodeDirectory, - session: SessionId, - root: PathBuf, - limits: Limits, - ) -> Self { - let authority = CellAuthority::with_telemetry(layout.clone(), runtime.telemetry_handle()); - Self { - runtime, - registry, - authority, - policy: ReadPolicyStore::new(layout.clone()), - layout, - directory, - session, - root, - limits, - closed: CancellationToken::new(), - activation: Arc::new(Mutex::new(())), - active: Arc::new(RwLock::new(HashMap::new())), - } - } - - /// Loads the current Cell incarnation and advisory desired-reader policy. - pub async fn target(&self, target: &CellTarget) -> Result<(IncarnationId, Option)> { - let control = self - .authority - .load(target.cell_id()) - .await? - .ok_or(Error::CellNotActive)?; - if control.value().state == ControlState::Tombstoned { - return Err(Error::CellNotActive); - } - let policy = self - .policy - .load(target.cell_id()) - .await? - .map(|observed| observed.value()); - Ok((control.value().incarnation, policy)) - } - - /// Conditionally updates the reader target after caller authorization. - pub async fn set_target( - &self, - target: &CellTarget, - expected_revision: u64, - desired_readers: u16, - ) -> Result> { - let control = self - .authority - .load(target.cell_id()) - .await? - .ok_or(Error::CellNotActive)?; - if control.value().state == ControlState::Tombstoned { - return Err(Error::CellNotActive); - } - let incarnation = control.value().incarnation; - let observed = self.policy.load(target.cell_id()).await?; - let updated = match observed { - None if expected_revision == 0 => { - self.policy - .create(target.cell_id(), incarnation, desired_readers) - .await? - } - Some(observed) if observed.value().revision() == expected_revision => { - if observed.value().incarnation() == incarnation { - if observed.value().desired_readers() == desired_readers { - observed - } else { - self.policy.update(&observed, desired_readers).await? - } - } else { - self.policy - .replace_incarnation(&observed, incarnation, desired_readers) - .await? - } - } - _ => return Ok(None), - }; - let current = self - .authority - .load(target.cell_id()) - .await? - .ok_or(Error::Fenced)?; - if current.value().incarnation != incarnation - || current.value().state == ControlState::Tombstoned - { - return Err(Error::Fenced); - } - Ok(Some(updated.value())) - } - - /// Admits or refreshes a selected snapshot after an authenticated owner hint. - pub async fn activate(&self, target: CellTarget, origin: SessionId) -> Result { - tokio::select! { - () = self.closed.cancelled() => Err(Error::RuntimeClosed), - result = self.activate_open(target, origin) => result, - } - } - - async fn activate_open(&self, target: CellTarget, origin: SessionId) -> Result { - let _activation = self.activation.lock().await; - self.ensure_open()?; - let cell = target.cell_id(); - let control = self - .authority - .load(cell) - .await? - .ok_or(Error::CellNotActive)?; - let control = control.value(); - let owner = control.owner.as_ref().ok_or(Error::Fenced)?; - if control.state != ControlState::Serving - || control.recovery.is_some() - || owner.session != origin - { - return Err(Error::Fenced); - } - if !self.selected(control, origin).await? { - return Err(Error::Fenced); - } - let path = self.destination(cell).await?; - let existing = { self.active.read().await.get(&cell).cloned() }; - if let Some(existing) = existing { - match existing.refresh(&path).await { - Ok(receipt) if self.still_selected(cell).await? => return Ok(receipt), - Ok(_) => { - self.remove_locked(cell).await; - return Err(Error::Fenced); - } - Err(Error::Fenced) => { - self.remove_locked(cell).await; - } - Err(error) => return Err(error), - } - } - let replica = CellReplica::new( - self.layout.clone(), - *cell.as_bytes(), - *control.incarnation.as_bytes(), - self.limits, - )?; - let reader = CellReadReplica::open( - self.runtime.clone(), - Arc::clone(&self.registry), - self.authority.clone(), - self.directory.clone(), - replica, - target, - &path, - ) - .await?; - let receipt = reader.receipt().await; - if !self.still_selected(cell).await? { - return Err(Error::Fenced); - } - self.active.write().await.insert(cell, reader); - Ok(receipt) - } - - /// Returns the selected snapshot receipt and live-owner readiness. - pub async fn status(&self, target: CellTarget) -> Result<(Receipt, bool)> { - if !self.still_selected(target.cell_id()).await? { - return Err(Error::ReplicaUnavailable); - } - self.resolve(target).await?.readiness().await - } - - async fn destination(&self, cell: CellId) -> Result { - let directory = self.root.join(format!("{cell:?}")); - tokio::fs::create_dir_all(&directory) - .await - .map_err(|source| Error::Facility { - name: "Cell read replica directory", - source: Box::new(source), - })?; - Ok(directory.join(format!("{}.sqlite", Uuid::now_v7()))) - } - - async fn selected(&self, control: &Control, origin: SessionId) -> Result { - let Some(policy) = self.policy.load(control.cell).await? else { - return Ok(false); - }; - let policy = policy.value(); - if policy.incarnation() != control.incarnation || policy.desired_readers() == 0 { - return Ok(false); - } - let candidates = self - .directory - .select_readers( - control.cell, - origin, - control.code, - usize::from(policy.desired_readers()), - now_ms()?, - MAX_LIVE_NODES, - ) - .await?; - Ok(candidates - .iter() - .any(|candidate| candidate.session() == self.session)) - } - - async fn still_selected(&self, cell: CellId) -> Result { - let Some(control) = self.authority.load(cell).await? else { - return Ok(false); - }; - let control = control.value(); - if control.state != ControlState::Serving || control.recovery.is_some() { - return Ok(false); - } - let Some(owner) = control.owner.as_ref() else { - return Ok(false); - }; - self.selected(control, owner.session).await - } - - /// Refreshes admitted snapshots and evicts readers removed from placement. - /// - /// This loop does not discover new Cells; authenticated owner hints call - /// `activate`. Cancellation interrupts provider waits and bounded batches. - pub async fn run(&self, cancellation: CancellationToken) -> Result<()> { - let mut tick = tokio::time::interval(RECONCILE_INTERVAL); - tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - let mut cursor = 0_usize; - loop { - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - () = self.closed.cancelled() => return Ok(()), - _ = tick.tick() => {} - } - if self.closed.is_cancelled() { - return Ok(()); - } - let mut readers = self.active.read().await.keys().copied().collect::>(); - readers.sort_by_key(|cell| cell.as_bytes().to_owned()); - let count = readers.len().min(RECONCILE_BATCH); - for _ in 0..count { - let index = cursor % readers.len(); - let cell = readers[index]; - // Advance before I/O so an unavailable Cell cannot starve the - // rest of the bounded batch after cancellation or timeout. - cursor = (index + 1) % readers.len(); - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - () = self.closed.cancelled() => return Ok(()), - result = tokio::time::timeout(RECONCILE_DEADLINE, self.refresh_selected(cell)) => { - match result { - Ok(Ok(())) => {}, - Ok(Err(error)) => tracing::warn!(?cell, error = %error, "read replica refresh failed"), - Err(_) => tracing::warn!(?cell, "read replica refresh deadline exceeded"), - } - } - } - } - } - } - - async fn refresh_selected(&self, cell: CellId) -> Result<()> { - // Share activation's lane so an old view cannot evict a replacement - // installed concurrently for the same Cell after an epoch change. - let _activation = self.activation.lock().await; - self.ensure_open()?; - let current = self.active.read().await.get(&cell).cloned(); - let Some(reader) = current else { - return Ok(()); - }; - if !self.still_selected(cell).await? { - self.remove_locked(cell).await; - return Ok(()); - } - let path = self.destination(cell).await?; - match reader.refresh(&path).await { - Ok(_) => Ok(()), - Err(Error::Fenced) => { - // Keep verified warm bytes after owner death. Queries still - // require a live owner; changed authority evicts the view. - if reader.readiness().await.is_err() { - self.remove_locked(cell).await; - } - Ok(()) - } - Err(error) => Err(error), - } - } - - fn ensure_open(&self) -> Result<()> { - if self.closed.is_cancelled() { - return Err(Error::RuntimeClosed); - } - Ok(()) - } - - /// Closes and removes one read view before eviction or writable activation. - pub async fn remove(&self, cell: CellId) { - let _activation = self.activation.lock().await; - self.remove_locked(cell).await; - } - - async fn remove_locked(&self, cell: CellId) { - if let Some(reader) = self.active.write().await.remove(&cell) { - reader.close(); - } - } - - /// Permanently closes activation and every retained read view. - pub async fn shutdown(&self) { - // Cancel provider waits before joining activation's lane; retained peer - // adapters must neither strand drain nor reopen snapshots afterward. - self.closed.cancel(); - let _activation = self.activation.lock().await; - for (_, reader) in self.active.write().await.drain() { - reader.close(); - } - } -} - -fn now_ms() -> Result { - let elapsed = SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_err(|_| Error::Command("system clock precedes Unix epoch"))?; - i64::try_from(elapsed.as_millis()).map_err(|_| Error::Command("system clock overflow")) -} - -impl PeerReplicaResolver for ReadReplicaManager { - fn resolve( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>> { - let manager = self.clone(); - Box::pin(async move { - manager.ensure_open()?; - manager - .active - .read() - .await - .get(&target.cell_id()) - .cloned() - .ok_or(Error::ReplicaUnavailable) - }) - } -} - -impl PeerReplicaControl for ReadReplicaManager { - fn activate( - &self, - target: CellTarget, - origin: SessionId, - ) -> BoxFuture<'static, Result> { - let manager = self.clone(); - Box::pin(async move { manager.activate(target, origin).await }) - } - - fn status(&self, target: CellTarget) -> BoxFuture<'static, Result<(Receipt, bool)>> { - let manager = self.clone(); - Box::pin(async move { manager.status(target).await }) - } -} diff --git a/crates/crab-cell-host/src/read_replicas/recruitment.rs b/crates/crab-cell-host/src/read_replicas/recruitment.rs deleted file mode 100644 index 46c3657c4..000000000 --- a/crates/crab-cell-host/src/read_replicas/recruitment.rs +++ /dev/null @@ -1,309 +0,0 @@ -//! Owner-side recruitment for every compiled application namespace. - -use super::*; -use crab_cell_runtime::node::NodeAdvertisement; -use crab_cell_runtime::{ - cell::{application::ApplicationIdentity, catalog::CatalogEntry}, - client::CellDescription, - peer::ReplicaPeerClient, -}; -use futures_util::{StreamExt, stream, stream::FuturesUnordered}; -use std::collections::{HashSet, VecDeque}; -use tokio::sync::broadcast; - -const ACTIVATION_CONCURRENCY: usize = 16; - -struct Activation { - target: CellTarget, - expected: CellDescription, - node: NodeAdvertisement, -} - -/// Recruits selected readers for the locally owned Cells of one application. -/// -/// Activation is advisory. Each receiver independently checks current owner, -/// policy, membership, and resource admission before opening a read snapshot. -#[derive(Clone)] -pub struct ReadReplicaRecruiter { - readers: ReadReplicaManager, - identity: ApplicationIdentity, - peer: ReplicaPeerClient, - cursor: Arc>, -} - -impl ReadReplicaRecruiter { - /// Binds owner recruitment to an existing manager and authenticated transport. - pub fn new( - readers: ReadReplicaManager, - identity: ApplicationIdentity, - peer: ReplicaPeerClient, - ) -> Result { - if identity.application().as_bytes() != readers.layout.application_id() { - return Err(Error::Control( - "read recruitment application does not match storage", - )); - } - Ok(Self { - readers, - identity, - peer, - cursor: Arc::new(Mutex::new(0)), - }) - } - - /// Reconciles one authorized target within a bounded, cancellable owner check. - pub async fn reconcile(&self, target: CellTarget) -> Result<()> { - tokio::select! { - () = self.readers.closed.cancelled() => Err(Error::RuntimeClosed), - result = tokio::time::timeout(RECONCILE_DEADLINE, async { - let cell = target.cell_id(); - let attempts = stream::iter(self.prepare(target).await?) - .map(|activation| self.activate(activation)) - .buffer_unordered(ACTIVATION_CONCURRENCY); - tokio::pin!(attempts); - while let Some(result) = attempts.next().await { - if let Err(error) = result { - tracing::warn!(?cell, error = %error, "read replica activation hint failed"); - } - } - Ok(()) - }) => result.map_err(|_| Error::Deadline)?, - } - } - - async fn prepare(&self, target: CellTarget) -> Result> { - self.readers.ensure_open()?; - if target.tenant() != self.identity.tenant() - || target.application() != self.identity.application() - || self - .readers - .registry - .namespace_contract(target.namespace()) - .is_none() - { - return Err(Error::PeerAuthorization( - "read recruitment target is outside the application", - )); - } - let cell = target.cell_id(); - let Some(policy) = self.readers.policy.load(cell).await? else { - return Ok(Vec::new()); - }; - let policy = policy.value(); - if policy.desired_readers() == 0 { - return Ok(Vec::new()); - } - let Some(control) = self.readers.authority.load(cell).await? else { - return Ok(Vec::new()); - }; - let control = control.value(); - if control.state != ControlState::Serving - || control.recovery.is_some() - || control - .owner - .as_ref() - .is_none_or(|owner| owner.session != self.readers.session) - || policy.incarnation() != control.incarnation - { - return Ok(Vec::new()); - } - let expected = CellDescription { - cell, - incarnation: control.incarnation, - code: control.code, - schema: control.schema, - }; - let selected = self - .readers - .directory - .select_readers( - cell, - self.readers.session, - control.code, - usize::from(policy.desired_readers()), - now_ms()?, - MAX_LIVE_NODES, - ) - .await?; - Ok(selected - .into_iter() - .map(|node| Activation { - target: target.clone(), - expected, - node, - }) - .collect()) - } - - async fn activate(&self, activation: Activation) -> Result<()> { - self.peer - .activate( - &activation.target, - &self.readers.directory, - activation.node, - activation.expected, - ) - .await?; - Ok(()) - } - - /// Runs one bounded pass over local owner Cells, retaining fair progress across retries. - pub async fn reconcile_active(&self) -> Result<()> { - self.readers.ensure_open()?; - tokio::select! { - () = self.readers.closed.cancelled() => Err(Error::RuntimeClosed), - result = tokio::time::timeout(RECONCILE_DEADLINE, self.reconcile_active_open()) => { - result.map_err(|_| Error::Deadline)? - } - } - } - - async fn reconcile_active_open(&self) -> Result<()> { - let mut entries = self.readers.runtime.active_catalog_entries().await?; - entries.sort_by_key(|entry| *entry.cell().as_bytes()); - if entries.is_empty() { - *self.cursor.lock().await = 0; - return Ok(()); - } - for _ in 0..entries.len().min(RECONCILE_BATCH) { - let entry = { - let mut cursor = self.cursor.lock().await; - let index = *cursor % entries.len(); - // Advance before I/O and release the cursor so an explicit - // operator pass cannot hold publication scheduling behind it. - *cursor = (index + 1) % entries.len(); - &entries[index] - }; - let result = match self.target(entry) { - Ok(target) => self.reconcile(target).await, - Err(error) => Err(error), - }; - if let Err(error) = result { - tracing::warn!(cell = ?entry.cell(), error = %error, "read replica recruitment failed"); - } - } - Ok(()) - } - - fn target(&self, entry: &CatalogEntry) -> Result { - let target = CellTarget::new( - self.identity.tenant(), - self.identity.application(), - entry.namespace(), - entry.partition(), - )?; - if target.cell_id() != entry.cell() { - return Err(Error::Control("reader catalog target differs")); - } - Ok(target) - } - - pub(crate) async fn run(&self, cancellation: CancellationToken) -> Result<()> { - let mut publications = self.readers.runtime.subscribe_publications(); - let mut tick = tokio::time::interval(RECONCILE_INTERVAL); - tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - let mut dirty = VecDeque::new(); - let mut queued = VecDeque::<(Activation, tokio::time::Instant)>::new(); - let mut pending = HashSet::new(); - let mut scanning = FuturesUnordered::>>>::new(); - let mut preparing = FuturesUnordered::< - BoxFuture<'_, (tokio::time::Instant, Result>)>, - >::new(); - let mut active = - FuturesUnordered::)>>::new(); - loop { - while active.len() < ACTIVATION_CONCURRENCY { - let Some((activation, deadline)) = queued.pop_front() else { - break; - }; - let key = (activation.target.cell_id(), activation.node.session()); - if tokio::time::Instant::now() >= deadline { - pending.remove(&key); - continue; - } - active.push(Box::pin(async move { - let result = tokio::time::timeout_at(deadline, self.activate(activation)) - .await - .unwrap_or(Err(Error::Deadline)); - (key, result) - })); - } - // Retain at most one discovered Cell's candidate list, plus a bounded - // dirty queue. Pending activations survive later publication hints; - // only completed (Cell, session) pairs can be scheduled again. - if queued.is_empty() - && preparing.is_empty() - && let Some(entry) = dirty.pop_front() - { - let deadline = tokio::time::Instant::now() + RECONCILE_DEADLINE; - preparing.push(Box::pin(async move { - let result = tokio::time::timeout_at(deadline, async { - self.prepare(self.target(&entry)?).await - }) - .await - .unwrap_or(Err(Error::Deadline)); - (deadline, result) - })); - } - tokio::select! { - () = cancellation.cancelled() => return Ok(()), - () = self.readers.closed.cancelled() => return Ok(()), - Some(result) = scanning.next(), if !scanning.is_empty() => { - let mut entries = result?; - entries.sort_by_key(|entry| *entry.cell().as_bytes()); - let mut cursor = self.cursor.lock().await; - for _ in 0..entries.len().min(RECONCILE_BATCH) { - let index = *cursor % entries.len(); - *cursor = (index + 1) % entries.len(); - enqueue(&mut dirty, entries[index].clone()); - } - } - Some((key, result)) = active.next(), if !active.is_empty() => { - pending.remove(&key); - if let Err(error) = result { - tracing::warn!(cell = ?key.0, session = ?key.1, error = %error, "read replica activation hint failed"); - } - } - Some((deadline, result)) = preparing.next(), if !preparing.is_empty() => { - match result { - Ok(activations) => { - for activation in activations { - let key = (activation.target.cell_id(), activation.node.session()); - if pending.insert(key) { - queued.push_back((activation, deadline)); - } - } - } - Err(error) => tracing::warn!(error = %error, "read replica discovery failed"), - } - } - _ = tick.tick() => { - // Catalog enumeration uses the runtime mailbox. Poll it with - // peer work and cancellation so a busy actor cannot strand - // healthy hints or host drain behind the scan. - if scanning.is_empty() { - scanning.push(Box::pin(self.readers.runtime.active_catalog_entries())); - } - } - published = publications.recv() => match published { - Ok(entry) => enqueue(&mut dirty, entry), - // The periodic scan repairs overflow without allowing a hot - // publisher to create an unbounded queue of retained work. - Err(broadcast::error::RecvError::Lagged(_)) => {}, - Err(broadcast::error::RecvError::Closed) => return Ok(()), - }, - } - } - } -} - -fn enqueue(dirty: &mut VecDeque, entry: CatalogEntry) { - if let Some(queued) = dirty - .iter_mut() - .find(|queued| queued.cell() == entry.cell()) - { - *queued = entry; - } else if dirty.len() < RECONCILE_BATCH { - dirty.push_back(entry); - } -} diff --git a/crates/crab-cell-host/src/status.rs b/crates/crab-cell-host/src/status.rs deleted file mode 100644 index 229f18587..000000000 --- a/crates/crab-cell-host/src/status.rs +++ /dev/null @@ -1,68 +0,0 @@ -//! Status internals for the Cell node host. - -use super::*; - -/// Node lifecycle state visible to readiness and shutdown adapters. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum NodeState { - /// The node is installing its runtime and does not serve yet. - Starting, - /// The node serves Cells and may take new ownership. - Ready, - /// Scale-down started: the node still serves but takes no new ownership. - ScalingDown, - /// The node is releasing its Cells and refuses new work. - Draining, - /// Every Cell is released and the node's facilities are stopped. - Stopped, -} - -/// Progress while a node serves the Cells that cannot yet move. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct ScaleDownStatus { - /// Cells the node still owns, whether or not they can move yet. - pub remaining_cells: usize, - /// Cells the fleet still lists as transfer candidates for this node. - pub settled_candidates: usize, - /// Cells confirmed released during this planning window. - pub released_cells: usize, - /// Cells that cannot move yet: failed release attempts plus remaining - /// Cells with no settled candidate. - pub blocked_cells: usize, -} - -impl ScaleDownStatus { - /// Reports confirmed local release; receiver service is a separate proof. - #[must_use] - pub const fn ready_to_stop(self) -> bool { - self.remaining_cells == 0 - } -} - -/// Point-in-time lifecycle and admission status for one [`CellNode`]. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct NodeStatus { - pub(super) state: NodeState, - pub(super) shutting_down: bool, - pub(super) stats: CellRuntimeStats, -} - -impl NodeStatus { - /// Returns the lifecycle state observed for this status sample. - #[must_use] - pub const fn state(self) -> NodeState { - self.state - } - - /// Returns whether runtime admission has been cancelled. - #[must_use] - pub const fn is_shutting_down(self) -> bool { - self.shutting_down - } - - /// Returns the node-wide admission metrics observed for this sample. - #[must_use] - pub const fn stats(self) -> CellRuntimeStats { - self.stats - } -} diff --git a/crates/crab-cell-host/src/tasks.rs b/crates/crab-cell-host/src/tasks.rs deleted file mode 100644 index 6d790cab2..000000000 --- a/crates/crab-cell-host/src/tasks.rs +++ /dev/null @@ -1,264 +0,0 @@ -//! Tasks internals for the Cell node host. - -use super::*; - -struct AbortOnDrop { - pub(super) handle: JoinHandle, -} - -impl AbortOnDrop { - pub(super) fn new(handle: JoinHandle) -> Self { - Self { handle } - } - - pub(super) async fn join(&mut self) -> std::result::Result { - (&mut self.handle).await - } -} - -impl Drop for AbortOnDrop { - fn drop(&mut self) { - self.handle.abort(); - } -} - -/// Bounded task supervisor owned by a [`CellNode`] facility. -pub struct CellNodeTaskGroup { - pub(super) cancellation: CancellationToken, - pub(super) node_shutdown: CancellationToken, - tasks: Mutex>, - pub(super) failed: Arc, - pub(super) draining: AtomicBool, -} - -#[derive(Clone, Copy, PartialEq, Eq)] -enum TaskPhase { - Work, - Lease, -} - -struct NodeTask { - phase: TaskPhase, - handle: JoinHandle, -} - -impl Drop for CellNodeTaskGroup { - fn drop(&mut self) { - let tasks = match self.tasks.lock() { - Ok(tasks) => tasks, - Err(poisoned) => poisoned.into_inner(), - }; - for task in tasks.iter() { - task.handle.abort(); - } - } -} - -impl CellNodeTaskGroup { - pub(super) fn cancel_work(&self) { - self.draining.store(true, Ordering::Release); - self.cancellation.cancel(); - } - - /// Creates a task group whose cancellation tokens are controlled by the product host. - #[must_use] - pub fn new(cancellation: CancellationToken, node_shutdown: CancellationToken) -> Self { - Self { - cancellation, - node_shutdown, - tasks: Mutex::new(Vec::new()), - failed: Arc::new(AtomicBool::new(false)), - draining: AtomicBool::new(false), - } - } - - /// Returns the cancellation signal that stops owned tasks during node drain. - #[must_use] - pub fn cancellation_token(&self) -> CancellationToken { - self.cancellation.clone() - } - - pub(super) fn is_healthy(&self) -> bool { - if self.failed.load(Ordering::Acquire) || self.draining.load(Ordering::Acquire) { - return false; - } - self.tasks - .lock() - .map(|tasks| tasks.iter().all(|task| !task.handle.is_finished())) - .unwrap_or(false) - } - - pub(super) fn ensure_accepting_tasks(&self) -> crab_cell_runtime::Result<()> { - if self.draining.load(Ordering::Acquire) { - return Err(Error::CellDraining); - } - Ok(()) - } - - /// Spawns one bounded node task and retains its join handle for drain. - pub fn spawn(&self, task: F) -> crab_cell_runtime::Result<()> - where - F: Future> + Send + 'static, - E: std::error::Error + Send + Sync + 'static, - { - self.spawn_task( - async move { - task.await - .map_err(|error| Box::new(error) as Box) - }, - TaskPhase::Work, - ) - } - - /// Retains lease renewal until the node has drained its runtime and closed its log. - /// - /// This task must stop on the node-shutdown token rather than the work - /// cancellation token. It shares the ordinary task limit and supervision. - pub fn spawn_lease_maintenance(&self, task: F) -> crab_cell_runtime::Result<()> - where - F: Future> + Send + 'static, - E: std::error::Error + Send + Sync + 'static, - { - self.spawn_task( - async move { - task.await - .map_err(|error| Box::new(error) as Box) - }, - TaskPhase::Lease, - ) - } - - /// Spawns one task that already uses the node's boxed facility error type. - pub fn spawn_boxed(&self, task: F) -> crab_cell_runtime::Result<()> - where - F: Future + Send + 'static, - { - self.spawn_task(task, TaskPhase::Work) - } - - fn spawn_task(&self, task: F, phase: TaskPhase) -> crab_cell_runtime::Result<()> - where - F: Future + Send + 'static, - { - self.ensure_accepting_tasks()?; - let mut tasks = self - .tasks - .lock() - .map_err(|_| Error::Control("CellNode task group lock poisoned"))?; - self.ensure_accepting_tasks()?; - if tasks.len() >= MAX_NODE_TASKS { - return Err(Error::Capacity("CellNode task limit reached")); - } - let failed = Arc::clone(&self.failed); - let handle = tokio::spawn(async move { - let mut task = AbortOnDrop::new(tokio::spawn(task)); - match task.join().await { - Ok(result) => { - if result.is_err() { - failed.store(true, Ordering::Release); - } - result - } - Err(error) => { - failed.store(true, Ordering::Release); - Err(Box::new(error) as Box) - } - } - }); - tasks.push(NodeTask { phase, handle }); - Ok(()) - } - - pub(super) async fn drain_work_until(&self, deadline: Option) -> FacilityResult { - self.cancel_work(); - self.join_until(deadline, false).await - } - - /// Cancels admission and joins tasks in reverse registration order. - pub async fn drain(&self) -> FacilityResult { - self.drain_until(None).await - } - - /// Cancels admission and joins tasks until an optional absolute deadline. - pub async fn drain_until(&self, deadline: Option) -> FacilityResult { - self.cancel_work(); - self.node_shutdown.cancel(); - self.join_until(deadline, true).await - } - - async fn join_until(&self, deadline: Option, include_lease: bool) -> FacilityResult { - let tasks = match self.tasks.lock() { - Ok(mut tasks) => { - let (joining, retained): (Vec<_>, Vec<_>) = std::mem::take(&mut *tasks) - .into_iter() - .partition(|task| include_lease || task.phase == TaskPhase::Work); - *tasks = retained; - joining.into_iter().map(|task| task.handle).collect() - } - Err(poisoned) => { - for task in poisoned.into_inner().drain(..) { - task.handle.abort(); - } - return Err(Box::new(std::io::Error::other( - "CellNode task group lock poisoned", - ))); - } - }; - let mut tasks = TaskBatch { - tasks, - abort_on_drop: true, - }; - let mut first_error = None; - let mut timed_out = false; - while let Some(index) = tasks.tasks.len().checked_sub(1) { - let result = match deadline { - Some(deadline) => { - match tokio::time::timeout_at(deadline.into(), &mut tasks.tasks[index]).await { - Ok(result) => result, - Err(_) => { - timed_out = true; - first_error.get_or_insert_with(|| { - Box::new(std::io::Error::new( - std::io::ErrorKind::TimedOut, - "CellNode task group drain deadline exceeded", - )) - as Box - }); - break; - } - } - } - None => (&mut tasks.tasks[index]).await, - }; - tasks.tasks.pop(); - match result { - Ok(Ok(())) => {} - Ok(Err(error)) if first_error.is_none() => first_error = Some(error), - Ok(Err(_)) => {} - Err(error) if first_error.is_none() => { - first_error = Some(Box::new(error) as Box) - } - Err(_) => {} - } - } - if !timed_out { - tasks.abort_on_drop = false; - } - first_error.map_or(Ok(()), Err) - } -} - -struct TaskBatch { - pub(super) tasks: Vec>, - pub(super) abort_on_drop: bool, -} - -impl Drop for TaskBatch { - fn drop(&mut self) { - if self.abort_on_drop { - for task in &self.tasks { - task.abort(); - } - } - } -} diff --git a/crates/crab-cell-host/tests/node.rs b/crates/crab-cell-host/tests/node.rs deleted file mode 100644 index 678f6f864..000000000 --- a/crates/crab-cell-host/tests/node.rs +++ /dev/null @@ -1,109 +0,0 @@ -//! Cell node host integration tests. -//! -//! The suite is one test binary. Its modules live in `tests/node/`; the shared -//! application fixture stays in the suite module, so no target needs a `#[path]` -//! attribute. - -mod node { - use std::future::Future; - use std::pin::Pin; - use std::sync::atomic::{AtomicBool, Ordering}; - use std::sync::{Arc, Mutex}; - use std::time::{Duration, Instant}; - - use crab_cell_app::CompiledApplication; - use crab_cell_host::*; - use crab_cell_runtime::Error; - use crab_cell_runtime::cell::catalog::CatalogRole; - use crab_cell_runtime::cell::worker::SqlWorkerPool; - use crab_cell_runtime::follower::FollowerStore; - use crab_cell_runtime::identity::{ApplicationId, Digest, SessionId}; - use crab_cell_runtime::ltx::{DiskBudget, Host as ReplicaHost, Limits as ReplicaLimits}; - use crab_cell_runtime::node::durability::NodeDurabilityConfig; - use crab_cell_runtime::node::lease::NodeLeaseGuard; - use crab_cell_runtime::qualification::{ - QualificationExecution, QualificationOperation, QualificationOperationExecutor, - QualificationProfile, QualificationWorkload, - }; - use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, ModuleDescriptor, NamespaceDescriptor, RegistryBuilder, - }; - use tokio_util::sync::CancellationToken; - - /// Host admission bounds mirrored from the node contract. - const MAX_NODE_TASKS: usize = 256; - const MAX_NODE_FACILITIES: usize = 64; - - struct Module; - - impl CellModule for Module { - const NAME: &'static str = "host-test"; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: ModuleDescriptor = ModuleDescriptor { - name: "host-test", - source_digest: Digest::from_bytes([1; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: &[crab_cell_runtime::registry::MigrationDescriptor { - version: 1, - sql: "-- host migration v1", - digest: Digest::from_bytes([ - 0xd7, 0x41, 0xcb, 0x18, 0xae, 0xd4, 0x80, 0xb0, 0xe1, 0x55, 0x8e, 0x34, - 0x5a, 0x6b, 0xef, 0xf5, 0xe1, 0x60, 0x80, 0x59, 0x06, 0xba, 0xfe, 0x75, - 0xff, 0x9f, 0xa0, 0x7d, 0x10, 0xe7, 0x77, 0xbf, - ]), - }], - commands: &[], - queries: &[], - workflow_definitions: &[], - activity_types: &[], - namespaces: &[NamespaceDescriptor { - id: crab_cell_runtime::NamespaceId::from_bytes([2; 16]), - name: "host-test", - role: CatalogRole::Sql, - shards: 1, - effect_targets: &[], - dead_letter: None, - }], - }; - &DESCRIPTOR - } - - fn register(self, _registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - Ok(()) - } - } - - fn application() -> Arc { - let mut builder = crab_cell_app::ApplicationBuilder::new( - "host-test", - BuildDescriptor { - source_revision: "host-test".into(), - cargo_lock_digest: Digest::from_bytes([7; 32]), - }, - ) - .unwrap(); - builder.register(Module).unwrap(); - builder - .cell_type( - crab_cell_app::CellType::new( - "host-test", - "host-test", - crab_cell_runtime::NamespaceId::from_bytes([2; 16]), - CatalogRole::Sql, - 1, - ) - .unwrap(), - ) - .unwrap(); - Arc::new(builder.finish().unwrap()) - } - - pub mod builder; - pub mod components; - pub mod lifecycle; - pub mod qualification; - pub mod tasks; -} diff --git a/crates/crab-cell-host/tests/node/builder.rs b/crates/crab-cell-host/tests/node/builder.rs deleted file mode 100644 index 72b6237d7..000000000 --- a/crates/crab-cell-host/tests/node/builder.rs +++ /dev/null @@ -1,47 +0,0 @@ -//! Cell node builder validation and configuration retention. - -use super::*; - -#[test] -fn builder_rejects_missing_owners_before_starting() { - let error = match CellNodeBuilder::new(application()).build() { - Ok(_) => panic!("missing node owners must fail closed"), - Err(error) => error, - }; - assert!(matches!(error, Error::Control(_))); -} - -#[test] -fn builder_rejects_zero_node_session_before_starting() { - let result = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([0; 16])) - .build(); - let error = match result { - Ok(_) => panic!("zero node sessions must fail closed"), - Err(error) => error, - }; - assert!(matches!(error, Error::Control(_))); -} - -#[tokio::test] -async fn builder_retains_configured_follower_store_as_an_owned_component() { - let data_dir = tempfile::tempdir().unwrap(); - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([3; 16])) - .with_follower_store( - data_dir.path().join("followers"), - ReplicaLimits::default(), - DiskBudget::new(1 << 20), - ) - .build() - .unwrap(); - - assert!( - node.owned_component::(FOLLOWER_STORE_COMPONENT) - .is_some() - ); -} diff --git a/crates/crab-cell-host/tests/node/components.rs b/crates/crab-cell-host/tests/node/components.rs deleted file mode 100644 index 242ee3c63..000000000 --- a/crates/crab-cell-host/tests/node/components.rs +++ /dev/null @@ -1,228 +0,0 @@ -//! Owned components and provider facilities: registration, readiness, drain. - -use super::*; - -#[tokio::test] -async fn node_retains_typed_components_without_duplicate_names() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([26; 16])) - .build() - .unwrap(); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - let component = Arc::new(7_u64); - let weak = Arc::downgrade(&component); - node.install_facility(CellNodeFacility::new("unowned", || async { Ok(()) }).unwrap()) - .unwrap(); - assert!( - node.install_owned_component("unowned", Arc::new(6_u64)) - .is_err() - ); - node.install_owned_component("fixture-component", Arc::clone(&component)) - .unwrap(); - drop(component); - assert_eq!( - node.owned_component::("fixture-component").as_deref(), - Some(&7) - ); - assert!( - node.install_owned_component("fixture-component", Arc::new(8_u64)) - .is_err() - ); - node.shutdown().await.unwrap(); - assert!(weak.upgrade().is_none()); -} - -#[tokio::test] -async fn facility_batch_installation_is_atomic_on_name_conflict() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([30; 16])) - .build() - .unwrap(); - node.install_owned_component("existing", Arc::new(1_u64)) - .unwrap(); - let first = CellNodeFacility::owned("first", Arc::new(2_u64), || async { Ok(()) }).unwrap(); - let duplicate = - CellNodeFacility::owned("existing", Arc::new(3_u64), || async { Ok(()) }).unwrap(); - - assert!(node.install_facilities([first, duplicate]).is_err()); - assert!(node.owned_component::("first").is_none()); - assert_eq!(node.owned_component::("existing").as_deref(), Some(&1)); - node.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn readiness_requires_declared_owned_components() { - let node = CellNodeBuilder::new(application()) - .with_required_owned_components(["catalog", "router"]) - .unwrap() - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([28; 16])) - .build() - .unwrap(); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - node.install_owned_component("catalog", Arc::new(1_u64)) - .unwrap(); - - assert!(matches!(node.start(), Err(Error::Control(_)))); - node.install_owned_component("router", Arc::new(2_u64)) - .unwrap(); - node.start().unwrap(); - assert!(node.is_ready()); - node.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn facility_registration_is_frozen_after_readiness() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([34; 16])) - .build() - .unwrap(); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - assert!(node.is_ready()); - - assert!(matches!( - node.install_facility(CellNodeFacility::new("late", || async { Ok(()) }).unwrap()), - Err(Error::CellDraining) - )); - node.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn required_component_declaration_rejects_duplicates_and_late_changes() { - assert!( - CellNodeBuilder::new(application()) - .with_required_owned_components(["catalog", "catalog"]) - .is_err() - ); - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([29; 16])) - .build() - .unwrap(); - assert!( - node.require_owned_components(["catalog", "catalog"]) - .is_err() - ); - node.require_owned_components(["catalog"]).unwrap(); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap_err(); - assert!(node.require_owned_components(["router"]).is_ok()); - node.shutdown().await.unwrap(); - assert!(node.require_owned_components(["late"]).is_err()); -} - -#[tokio::test] -async fn facilities_drain_in_reverse_registration_order() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([16; 16])) - .build() - .unwrap(); - let events = Arc::new(Mutex::new(Vec::new())); - for name in ["storage", "transport", "scheduler"] { - let events = Arc::clone(&events); - node.install_facility( - CellNodeFacility::new(name, move || { - let events = Arc::clone(&events); - async move { - events - .lock() - .map_err(|_| { - Box::new(std::io::Error::other("event lock poisoned")) - as Box - })? - .push(name); - Ok(()) - } - }) - .unwrap(), - ) - .unwrap(); - } - - node.shutdown().await.unwrap(); - assert_eq!( - *events.lock().unwrap(), - vec!["scheduler", "transport", "storage"] - ); -} - -#[tokio::test] -async fn facility_failure_is_reported_after_all_facilities_attempt_and_runtime_drains() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([17; 16])) - .build() - .unwrap(); - let completed = Arc::new(std::sync::atomic::AtomicBool::new(false)); - let completed_clone = Arc::clone(&completed); - node.install_facility( - CellNodeFacility::new("healthy", move || { - let completed = Arc::clone(&completed_clone); - async move { - completed.store(true, Ordering::Release); - Ok(()) - } - }) - .unwrap(), - ) - .unwrap(); - node.install_facility( - CellNodeFacility::new("broken", || async { - Err(Box::new(std::io::Error::other("drain failed")) - as Box) - }) - .unwrap(), - ) - .unwrap(); - - let error = node.shutdown().await.unwrap_err(); - assert!(matches!(error, Error::Facility { name: "broken", .. })); - assert!(completed.load(Ordering::Acquire)); - assert!(node.is_shutting_down()); - assert_eq!(node.state(), NodeState::Draining); -} - -#[tokio::test] -async fn facility_registration_is_bounded_and_rejected_after_drain() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([18; 16])) - .build() - .unwrap(); - assert!(CellNodeFacility::new("", || async { Ok(()) }).is_err()); - for index in 0..MAX_NODE_FACILITIES { - let name = Box::leak(format!("facility-{index}").into_boxed_str()); - node.install_facility(CellNodeFacility::new(name, || async { Ok(()) }).unwrap()) - .unwrap(); - } - assert!(matches!( - node.install_facility(CellNodeFacility::new("overflow", || async { Ok(()) }).unwrap()), - Err(Error::Capacity(_)) - )); - node.shutdown().await.unwrap(); - assert!(matches!( - node.install_facility(CellNodeFacility::new("late", || async { Ok(()) }).unwrap()), - Err(Error::CellDraining) - )); -} diff --git a/crates/crab-cell-host/tests/node/lifecycle.rs b/crates/crab-cell-host/tests/node/lifecycle.rs deleted file mode 100644 index 798cdcded..000000000 --- a/crates/crab-cell-host/tests/node/lifecycle.rs +++ /dev/null @@ -1,283 +0,0 @@ -//! Node startup, deadline, shutdown, and scale-down lifecycle. - -use super::*; - -#[tokio::test] -async fn node_shutdown_is_idempotent_and_returns_stopped_state() { - let pool = SqlWorkerPool::new(1, 1).unwrap(); - let host = ReplicaHost::default(); - let node = CellNodeBuilder::new(application()) - .with_runtime(pool, 16 * 1024 * 1024) - .with_replica_host(host) - .with_session(SessionId::from_bytes([4; 16])) - .build() - .unwrap(); - assert_eq!(node.state(), NodeState::Starting); - assert!(!node.is_ready()); - let starting = node.status(); - assert_eq!(starting.state(), NodeState::Starting); - assert!(!starting.is_shutting_down()); - assert_eq!(starting.stats(), node.stats()); - assert!(node.start().is_err()); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - assert_eq!(node.state(), NodeState::Ready); - assert!(node.is_ready()); - node.shutdown().await.unwrap(); - assert_eq!(node.state(), NodeState::Stopped); - let stopped = node.status(); - assert_eq!(stopped.state(), NodeState::Stopped); - assert!(stopped.is_shutting_down()); - node.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn startup_lease_does_not_open_readiness_before_host_startup_finishes() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([19; 16])) - .build() - .unwrap(); - node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - assert_eq!(node.state(), NodeState::Starting); - assert!(!node.is_ready()); - assert!(node.start().is_err()); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.start().unwrap(); - assert!(node.is_ready()); - node.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn node_deadline_returns_after_a_stalled_facility() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([21; 16])) - .build() - .unwrap(); - node.install_facility( - CellNodeFacility::new("stalled", || async { - std::future::pending::().await - }) - .unwrap(), - ) - .unwrap(); - - let result = node - .shutdown_until(Instant::now() + std::time::Duration::from_millis(10)) - .await; - - assert!(result.is_err()); - assert_eq!(node.state(), NodeState::Draining); -} - -#[tokio::test] -async fn node_deadline_bounds_a_stalled_coordination_task() { - for lease_maintenance in [false, true] { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([23; 16])) - .build() - .unwrap(); - let tasks = node - .install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - let stalled = async { - std::future::pending::<()>().await; - Ok::<(), Error>(()) - }; - if lease_maintenance { - tasks.spawn_lease_maintenance(stalled).unwrap(); - } else { - tasks.spawn(stalled).unwrap(); - } - - let result = tokio::time::timeout( - std::time::Duration::from_secs(1), - node.shutdown_until(Instant::now() + std::time::Duration::from_millis(10)), - ) - .await - .expect("deadline-aware shutdown must return"); - - assert!(result.is_err()); - assert_eq!(node.state(), NodeState::Draining); - } -} - -#[tokio::test] -async fn node_cancels_admission_before_draining_provider_facilities() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([27; 16])) - .build() - .unwrap(); - let cancellation = CancellationToken::new(); - node.install_task_group(cancellation.clone(), CancellationToken::new()) - .unwrap(); - let observed = Arc::new(AtomicBool::new(false)); - let facility_observed = Arc::clone(&observed); - node.install_facility( - CellNodeFacility::new("provider", move || { - let cancellation = cancellation.clone(); - let facility_observed = facility_observed.clone(); - async move { - cancellation.cancelled().await; - facility_observed.store(true, Ordering::Release); - Ok(()) - } - }) - .unwrap(), - ) - .unwrap(); - - node.shutdown_until(Instant::now() + std::time::Duration::from_secs(1)) - .await - .unwrap(); - - assert!(observed.load(Ordering::Acquire)); -} - -#[tokio::test] -async fn scale_down_stops_acquisition_without_stopping_the_host() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([97; 16])) - .build() - .unwrap(); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - node.start().unwrap(); - let status = node - .drain_for_scale_down(Instant::now() + Duration::from_secs(1)) - .await - .unwrap(); - assert!(status.ready_to_stop()); - assert_eq!(node.state(), NodeState::Stopped); - assert!(!node.is_ready()); - assert!(!node.runtime().is_acquiring()); - let retry = node - .drain_for_scale_down(Instant::now() + Duration::from_secs(1)) - .await - .unwrap(); - assert!(retry.ready_to_stop()); - node.drain().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn concurrent_scale_downs_share_one_bounded_drain_lane() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([98; 16])) - .build() - .unwrap(); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - node.start().unwrap(); - - // A fleet reconciler and a node shutdown hook can both drive scale down for - // the same node. They share one drain lane, so running them at once must - // stay bounded instead of interleaving releases or waiting on each other - // forever. - let deadline = Instant::now() + Duration::from_secs(5); - let (first, second) = tokio::time::timeout(Duration::from_secs(15), async { - tokio::join!( - node.drain_for_scale_down(deadline), - node.drain_for_scale_down(deadline) - ) - }) - .await - .expect("concurrent scale downs must not block each other"); - - let first = first.unwrap(); - let second = second.unwrap(); - assert!(first.ready_to_stop()); - assert!(second.ready_to_stop()); - assert_eq!(first.remaining_cells, 0); - assert_eq!(second.remaining_cells, 0); - assert_eq!(first.blocked_cells, 0); - assert_eq!(second.blocked_cells, 0); - assert_eq!(node.state(), NodeState::Stopped); - assert!(!node.is_ready()); - assert!(!node.runtime().is_acquiring()); - node.drain().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn concurrent_shutdown_waits_for_the_single_runtime_drain() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([5; 16])) - .build() - .unwrap(); - let first = node.shutdown(); - let second = node.shutdown(); - let (first, second) = tokio::join!(first, second); - first.unwrap(); - second.unwrap(); - assert_eq!(node.state(), NodeState::Stopped); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn three_nodes_have_independent_lifecycle_and_resource_ledgers() { - let mut nodes = Vec::new(); - for index in 0..3_u8 { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([index + 10; 16])) - .build() - .unwrap(); - assert_eq!(node.state(), NodeState::Starting); - assert!(!node.is_ready()); - nodes.push(node); - } - - for node in &nodes { - node.shutdown().await.unwrap(); - assert_eq!(node.state(), NodeState::Stopped); - assert!(node.is_shutting_down()); - } -} - -#[tokio::test] -async fn session_withdrawal_waits_for_runtime_drain() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([44; 16])) - .build() - .unwrap(); - let shutdown = CancellationToken::new(); - let tasks = node - .install_task_group(CancellationToken::new(), shutdown.clone()) - .unwrap(); - let runtime = node.runtime(); - tasks - .spawn_lease_maintenance(async move { - shutdown.cancelled().await; - if !runtime.is_shutting_down() { - return Err(Error::Control("session withdrew before runtime drain")); - } - Ok(()) - }) - .unwrap(); - - node.shutdown_until(Instant::now() + Duration::from_secs(1)) - .await - .unwrap(); -} diff --git a/crates/crab-cell-host/tests/node/qualification.rs b/crates/crab-cell-host/tests/node/qualification.rs deleted file mode 100644 index 6314d134e..000000000 --- a/crates/crab-cell-host/tests/node/qualification.rs +++ /dev/null @@ -1,105 +0,0 @@ -//! Qualification gating for a starting node. - -use super::*; - -#[derive(Clone)] -struct QualificationStub; - -impl QualificationOperationExecutor for QualificationStub { - type Future<'a> = std::future::Ready>; - - fn execute<'a>(&'a mut self, operation: QualificationOperation) -> Self::Future<'a> { - std::future::ready(Ok( - QualificationExecution::acknowledged(true).with_case(operation.case()) - )) - } -} - -#[tokio::test] -async fn qualification_rejects_a_node_before_readiness() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([32; 16])) - .build() - .unwrap(); - let workload = QualificationWorkload::generate_with_size( - &QualificationProfile::pr_contract(), - 41, - 1, - 56, - 1, - ) - .unwrap(); - let error = node - .run_qualification(&workload, &mut QualificationStub) - .await - .unwrap_err(); - assert!(matches!(error, Error::CellDraining)); -} - -#[tokio::test] -async fn concurrent_qualification_is_readiness_gated_and_bounded() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([35; 16])) - .build() - .unwrap(); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - let workload = QualificationWorkload::generate_with_size( - &QualificationProfile::pr_contract(), - 41, - 1, - 56, - 1, - ) - .unwrap(); - let summary = node - .run_qualification_concurrent(&workload, QualificationStub, 2) - .await - .unwrap(); - assert_eq!(summary.operations(), 56); - node.shutdown().await.unwrap(); -} - -struct ObservedQualificationStub; - -impl QualificationOperationExecutor for ObservedQualificationStub { - type Future<'a> = std::future::Ready>; - - fn execute<'a>(&'a mut self, _operation: QualificationOperation) -> Self::Future<'a> { - std::future::ready(Ok(QualificationExecution::acknowledged(true))) - } -} - -#[tokio::test] -async fn observed_qualification_preserves_unclaimed_case_coverage() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([36; 16])) - .build() - .unwrap(); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - let workload = QualificationWorkload::generate_with_size( - &QualificationProfile::pr_contract(), - 41, - 1, - 56, - 1, - ) - .unwrap(); - let summary = node - .run_qualification_observed(&workload, &mut ObservedQualificationStub) - .await - .unwrap(); - assert!(summary.case_coverage().iter().all(|byte| *byte == 0)); - node.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-host/tests/node/tasks.rs b/crates/crab-cell-host/tests/node/tasks.rs deleted file mode 100644 index e5485d838..000000000 --- a/crates/crab-cell-host/tests/node/tasks.rs +++ /dev/null @@ -1,341 +0,0 @@ -//! Node task groups, supervision, and readiness reporting. - -use super::*; - -struct NoopNodeDurabilityProvider; - -impl NodeDurabilityProvider for NoopNodeDurabilityProvider { - fn recruit( - self: Arc, - _limits: ReplicaLimits, - _required_follower_bytes: u64, - _live_node_limit: usize, - ) -> Pin>> + Send>> { - Box::pin(async { Ok(None) }) - } -} - -#[tokio::test] -async fn node_durability_supervisor_is_host_owned_and_joined() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([28; 16])) - .build() - .unwrap(); - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - let provider = Arc::new(NoopNodeDurabilityProvider); - node.install_node_durability_provider( - Arc::clone(&provider), - NodeDurabilitySupervisorConfig::new( - ApplicationId::from_bytes([29; 16]), - ReplicaLimits::default(), - 1, - 1, - std::time::Duration::from_millis(1), - std::time::Duration::from_millis(1), - 1, - ) - .unwrap(), - ) - .unwrap(); - assert!( - node.owned_component::(NODE_DURABILITY_PROVIDER_COMPONENT) - .is_some() - ); - node.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn failed_node_task_removes_readiness_and_is_reported_during_drain() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([30; 16])) - .build() - .unwrap(); - let task_group = node - .install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - node.start().unwrap(); - assert!(node.is_ready()); - task_group - .spawn(async { Err::<(), _>(std::io::Error::other("supervisor failed")) }) - .unwrap(); - tokio::time::sleep(std::time::Duration::from_millis(1)).await; - assert!(!node.is_ready()); - let error = node.drain().await.unwrap_err(); - assert!(matches!( - error, - Error::Facility { - name: "cell-coordination-tasks", - .. - } - )); -} - -#[tokio::test] -async fn completed_node_task_removes_readiness() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([33; 16])) - .build() - .unwrap(); - let task_group = node - .install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - node.start().unwrap(); - assert!(node.is_ready()); - - task_group.spawn(async { Ok::<(), Error>(()) }).unwrap(); - tokio::time::timeout(std::time::Duration::from_secs(1), async { - while node.is_ready() { - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - assert!(!node.is_ready()); - node.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn panicked_node_task_removes_readiness_and_is_reported_during_drain() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([31; 16])) - .build() - .unwrap(); - let task_group = node - .install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - node.start().unwrap(); - assert!(node.is_ready()); - task_group - .spawn_boxed(async { - panic!("supervisor panicked"); - }) - .unwrap(); - tokio::task::yield_now().await; - assert!(!node.is_ready()); - let error = node.drain().await.unwrap_err(); - assert!(matches!( - error, - Error::Facility { - name: "cell-coordination-tasks", - .. - } - )); -} - -#[tokio::test] -async fn readiness_requires_the_node_task_group() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([22; 16])) - .build() - .unwrap(); - node.install_node_lease_for_startup(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - - let error = node.start().unwrap_err(); - assert!(matches!(error, Error::Control(_))); - assert_eq!(node.state(), NodeState::Starting); - - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .unwrap(); - node.start().unwrap(); - assert!(node.is_ready()); - node.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn task_group_cancels_both_tokens_and_joins_tasks() { - let cancellation = CancellationToken::new(); - let node_shutdown = CancellationToken::new(); - let finished = Arc::new(AtomicBool::new(false)); - let task_cancellation = cancellation.clone(); - let task_shutdown = node_shutdown.clone(); - let task_finished = Arc::clone(&finished); - let tasks = CellNodeTaskGroup::new(cancellation.clone(), node_shutdown.clone()); - tasks - .spawn(async move { - task_cancellation.cancelled().await; - task_shutdown.cancelled().await; - task_finished.store(true, Ordering::Release); - Ok::<(), Error>(()) - }) - .unwrap(); - - tasks.drain().await.unwrap(); - - assert!(cancellation.is_cancelled()); - assert!(node_shutdown.is_cancelled()); - assert!(finished.load(Ordering::Acquire)); -} - -#[tokio::test] -async fn task_group_rejects_new_tasks_after_drain_starts() { - let cancellation = CancellationToken::new(); - let node_shutdown = CancellationToken::new(); - let task_shutdown = node_shutdown.clone(); - let tasks = CellNodeTaskGroup::new(cancellation, node_shutdown); - tasks - .spawn(async move { - task_shutdown.cancelled().await; - Ok::<(), Error>(()) - }) - .unwrap(); - tasks.drain().await.unwrap(); - - assert!(matches!( - tasks.spawn(async { Ok::<(), Error>(()) }), - Err(Error::CellDraining) - )); - assert!(matches!( - tasks.spawn_boxed(async { Ok(()) }), - Err(Error::CellDraining) - )); -} - -#[tokio::test] -async fn task_group_deadline_aborts_unfinished_tasks() { - struct DropProbe(Arc); - - impl Drop for DropProbe { - fn drop(&mut self) { - self.0.store(true, Ordering::Release); - } - } - - let tasks = CellNodeTaskGroup::new(CancellationToken::new(), CancellationToken::new()); - let dropped = Arc::new(AtomicBool::new(false)); - let task_dropped = Arc::clone(&dropped); - tasks - .spawn(async move { - let _probe = DropProbe(task_dropped); - std::future::pending::<()>().await; - Ok::<(), Error>(()) - }) - .unwrap(); - - let result = tasks - .drain_until(Some(Instant::now() + std::time::Duration::from_millis(10))) - .await; - - assert!(result.is_err()); - tokio::task::yield_now().await; - assert!(dropped.load(Ordering::Acquire)); -} - -#[tokio::test] -async fn dropping_task_group_aborts_unjoined_tasks() { - struct DropProbe(Arc); - - impl Drop for DropProbe { - fn drop(&mut self) { - self.0.store(true, Ordering::Release); - } - } - - let dropped = Arc::new(AtomicBool::new(false)); - let started = Arc::new(tokio::sync::Notify::new()); - { - let tasks = CellNodeTaskGroup::new(CancellationToken::new(), CancellationToken::new()); - let task_dropped = Arc::clone(&dropped); - let task_started = Arc::clone(&started); - tasks - .spawn(async move { - let _probe = DropProbe(task_dropped); - task_started.notify_one(); - std::future::pending::<()>().await; - Ok::<(), Error>(()) - }) - .unwrap(); - started.notified().await; - } - tokio::time::timeout(std::time::Duration::from_secs(1), async { - while !dropped.load(Ordering::Acquire) { - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - assert!(dropped.load(Ordering::Acquire)); -} - -#[tokio::test] -async fn node_owns_one_task_group_and_drains_it() { - let node = CellNodeBuilder::new(application()) - .with_runtime(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024) - .with_replica_host(ReplicaHost::default()) - .with_session(SessionId::from_bytes([20; 16])) - .build() - .unwrap(); - let cancellation = CancellationToken::new(); - let node_shutdown = CancellationToken::new(); - let tasks = node - .install_task_group(cancellation.clone(), node_shutdown.clone()) - .unwrap(); - assert!( - node.install_task_group(CancellationToken::new(), CancellationToken::new()) - .is_err() - ); - let finished = Arc::new(AtomicBool::new(false)); - let task_finished = Arc::clone(&finished); - let task_shutdown = node_shutdown.clone(); - tasks - .spawn_lease_maintenance(async move { - task_shutdown.cancelled().await; - task_finished.store(true, Ordering::Release); - Ok::<(), Error>(()) - }) - .unwrap(); - - node.shutdown().await.unwrap(); - - assert!(cancellation.is_cancelled()); - assert!(node_shutdown.is_cancelled()); - assert!(finished.load(Ordering::Acquire)); -} - -#[tokio::test] -async fn task_group_rejects_tasks_above_bound() { - let cancellation = CancellationToken::new(); - let node_shutdown = CancellationToken::new(); - let tasks = CellNodeTaskGroup::new(cancellation, node_shutdown.clone()); - for _ in 0..MAX_NODE_TASKS { - let node_shutdown = node_shutdown.clone(); - tasks - .spawn(async move { - node_shutdown.cancelled().await; - Ok::<(), Error>(()) - }) - .unwrap(); - } - let node_shutdown = node_shutdown.clone(); - let error = tasks - .spawn(async move { - node_shutdown.cancelled().await; - Ok::<(), Error>(()) - }) - .unwrap_err(); - assert!(matches!(error, Error::Capacity(_))); - - assert!(matches!( - tasks.spawn_lease_maintenance(async { Ok::<(), Error>(()) }), - Err(Error::Capacity(_)) - )); - - tasks.drain().await.unwrap(); -} diff --git a/crates/crab-cell-peer-http/Cargo.toml b/crates/crab-cell-peer-http/Cargo.toml deleted file mode 100644 index ef12c2315..000000000 --- a/crates/crab-cell-peer-http/Cargo.toml +++ /dev/null @@ -1,26 +0,0 @@ -[package] -name = "crab-cell-peer-http" -version = "0.1.0" -edition.workspace = true -license.workspace = true -repository.workspace = true -rust-version.workspace = true -publish = false -description = "Authenticated HTTP transport and pinned mTLS for Crab Cell peers" - -[dependencies] -axum = { version = "0.8", default-features = false, features = ["http1", "tokio"] } -crab-cell-runtime.workspace = true -ed25519-dalek = { workspace = true, features = ["pkcs8"] } -futures-util.workspace = true -http = "1" -reqwest = { workspace = true, features = ["rustls-tls", "stream"] } -rustls = "0.23" -rustls-pemfile = "2" -sha2 = "0.10" -thiserror.workspace = true -tokio = { workspace = true, features = ["net", "time"] } -tokio-rustls = "0.26" -tracing.workspace = true -url = "2" -x509-cert = "0.2.5" diff --git a/crates/crab-cell-peer-http/README.md b/crates/crab-cell-peer-http/README.md deleted file mode 100644 index 1fde01efa..000000000 --- a/crates/crab-cell-peer-http/README.md +++ /dev/null @@ -1,25 +0,0 @@ -# Crab Cell peer HTTP transport - -`PeerHttpRoundTrip` sends authenticated Cell peer envelopes to the current -enrolled owner. It reloads the authority and node directory on a retry, bounds -responses, distinguishes an unknown outcome from a request that never started, -and caches HTTP clients by enrolled session, certificate, and public key. - -Owner-routed requests make at most two attempts. On HTTP 429/503 with an integer -`Retry-After` header, the transport paces the retry within the original deadline, -then reloads ownership. Persistent admission rejection returns a capacity error. -A delay that leaves no request budget returns a deadline error without sleeping. -Responses without that header retain the immediate owner-refresh behavior used -by existing receivers. Lost or invalid responses remain unknown outcomes and -are not retried here. Direct-node activation remains a single attempt: its -caller owns scheduling and retries, and receives the same capacity distinction. - -The product supplies `PeerTargetScope` and `PeerHttpClientFactory`. The factory -must authenticate its local client identity and pin the remote certificate and -public key passed to it. The receiver must verify the peer envelope, enrollment, -and product authorization before dispatching through `PeerDispatcher`. - -[BeyondDB](https://github.com/crabbuild/beyonddb) now uses Cellule's peer -transport in its separate repository. `crab-http-server` retains its -repository-scoped transport, owner-description hints, and telemetry; those -product routing semantics are not enabled by this transport. diff --git a/crates/crab-cell-peer-http/src/lib.rs b/crates/crab-cell-peer-http/src/lib.rs deleted file mode 100644 index 67f38b509..000000000 --- a/crates/crab-cell-peer-http/src/lib.rs +++ /dev/null @@ -1,436 +0,0 @@ -//! Owner-resolving HTTP transport for authenticated Cell peer requests. - -mod tls; - -pub use tls::{LoadedPeerTls, PeerTlsClient, PeerTlsIdentity, PeerTlsListener, TlsError}; - -use std::{ - collections::VecDeque, - future::Future, - pin::Pin, - sync::{Arc, Mutex}, - time::{Duration, Instant, SystemTime, UNIX_EPOCH}, -}; - -use crab_cell_runtime::Error as CellError; -use crab_cell_runtime::cell::application::ApplicationIdentity; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::identity::{CellTarget, Digest, SessionId}; -use crab_cell_runtime::node::{NodeAdvertisement, NodeDirectory}; -use crab_cell_runtime::peer::{PeerRoundTrip, wire as peer_wire}; -use futures_util::StreamExt; -use http::{StatusCode, header}; - -const PEER_FORWARD_PATH: &str = "internal/cells/v1/forward"; -const MAX_PEER_CLIENTS: usize = 1_024; -/// Content type accepted by the private peer forwarding endpoint. -pub const PROTOBUF_MEDIA_TYPE: &str = "application/x-protobuf"; - -/// Builds an mTLS HTTP client pinned to one enrolled peer certificate and key. -pub trait PeerHttpClientFactory: Send + Sync + 'static { - /// Builds a client that rejects any peer other than the pinned identity. - fn client( - &self, - certificate: Digest, - public_key: [u8; 32], - ) -> crab_cell_runtime::Result; -} - -/// Restricts which Cell targets this node may forward through a peer. -pub trait PeerTargetScope: Send + Sync + 'static { - /// Rejects a target outside the product's routing scope. - fn check_target(&self, target: &CellTarget) -> crab_cell_runtime::Result<()>; -} - -impl PeerTargetScope for ApplicationIdentity { - fn check_target(&self, target: &CellTarget) -> crab_cell_runtime::Result<()> { - if target.tenant() != self.tenant() || target.application() != self.application() { - return Err(CellError::PeerAuthorization( - "Cell target is outside the routed application", - )); - } - Ok(()) - } -} - -/// Sends signed Cell requests to the current enrolled owner over HTTP. -pub struct PeerHttpRoundTrip { - scope: Arc, - authority: CellAuthority, - directory: NodeDirectory, - tls: Arc, - session: SessionId, - clients: Arc>>, -} - -impl PeerHttpRoundTrip { - /// Binds one application, authority, fleet directory, and local session. - #[must_use] - pub fn new( - scope: Arc, - authority: CellAuthority, - directory: NodeDirectory, - tls: Arc, - session: SessionId, - ) -> Self { - Self { - scope, - authority, - directory, - tls, - session, - clients: Arc::new(Mutex::new(VecDeque::new())), - } - } - - async fn send_inner( - &self, - target: CellTarget, - request: Vec, - timeout_ms: u32, - ) -> crab_cell_runtime::Result> { - if request.len() > crab_cell_runtime::peer::MAX_PEER_REQUEST_BYTES { - return Err(CellError::Peer("request exceeds peer byte limit")); - } - let started = Instant::now(); - let mut last_retry: Option<(CellError, Duration)> = None; - for _ in 0..2 { - if let Some((_, delay)) = &last_retry { - let remaining = - Duration::from_millis(u64::from(remaining_timeout(started, timeout_ms)?)); - if *delay >= remaining { - return Err(CellError::Deadline); - } - if !delay.is_zero() { - // Admission rejection has not started the operation. Pace - // its one retry, then reload ownership in case it moved. - tokio::time::sleep(*delay).await; - } - } - let remaining_ms = remaining_timeout(started, timeout_ms)?; - let owner = tokio::time::timeout( - Duration::from_millis(u64::from(remaining_ms)), - self.owner(&target), - ) - .await - .map_err(|_| CellError::Deadline)??; - let remaining_ms = remaining_timeout(started, timeout_ms)?; - match self.send_once(&owner, request.clone(), remaining_ms).await { - Ok(PeerHttpAttempt::Reply(reply)) => return Ok(reply), - Ok(PeerHttpAttempt::Retry(error, delay)) => last_retry = Some((error, delay)), - Ok(PeerHttpAttempt::Unknown(error)) => { - return Err(CellError::PeerTransportUnknown { - context: "peer HTTP response was lost or invalid", - source: Box::new(error), - }); - } - Err(error) => return Err(error), - } - } - Err(last_retry.map_or(CellError::CellNotActive, |(error, _)| error)) - } - - async fn send_to_node_inner( - &self, - target: CellTarget, - node: NodeAdvertisement, - request: Vec, - timeout_ms: u32, - ) -> crab_cell_runtime::Result> { - self.scope.check_target(&target)?; - if node.session() == self.session { - return Err(CellError::CellNotActive); - } - let now_ms = now_ms()?; - if node.expires_at_ms() <= now_ms || node.endpoint().is_empty() { - return Err(CellError::CellNotActive); - } - let owner = RemotePeer { - session: node.session(), - endpoint: url::Url::parse(node.endpoint()).map_err(peer_transport)?, - certificate: node.certificate(), - public_key: node.verifying_key()?.to_bytes(), - }; - match self.send_once(&owner, request, timeout_ms).await? { - PeerHttpAttempt::Reply(reply) => Ok(reply), - PeerHttpAttempt::Retry(error, _) => Err(error), - PeerHttpAttempt::Unknown(error) => Err(CellError::PeerTransportUnknown { - context: "peer HTTP activation response was lost or invalid", - source: Box::new(error), - }), - } - } - - async fn owner(&self, target: &CellTarget) -> crab_cell_runtime::Result { - self.scope.check_target(target)?; - let control = self - .authority - .load(target.cell_id()) - .await? - .ok_or(CellError::CellNotActive)?; - let owner = control - .value() - .owner - .as_ref() - .ok_or(CellError::CellNotActive)?; - if owner.session == self.session { - return Err(CellError::CellNotActive); - } - let now_ms = now_ms()?; - let enrolled = self - .directory - .load(owner.session, now_ms) - .await? - .ok_or(CellError::CellNotActive)?; - let advertisement = enrolled.advertisement(); - if advertisement.endpoint() != owner.endpoint { - return Err(CellError::PeerAuthorization( - "Cell owner endpoint is not enrolled", - )); - } - Ok(RemotePeer { - session: owner.session, - endpoint: url::Url::parse(advertisement.endpoint()).map_err(peer_transport)?, - certificate: advertisement.certificate(), - public_key: advertisement.verifying_key()?.to_bytes(), - }) - } - - fn client(&self, owner: &RemotePeer) -> crab_cell_runtime::Result { - { - let clients = self - .clients - .lock() - .map_err(|_| CellError::Peer("peer HTTP client cache is poisoned"))?; - if let Some(cached) = clients.iter().find(|cached| cached.matches(owner)) { - return Ok(cached.client.clone()); - } - } - - let client = self.tls.client(owner.certificate, owner.public_key)?; - let mut clients = self - .clients - .lock() - .map_err(|_| CellError::Peer("peer HTTP client cache is poisoned"))?; - if let Some(cached) = clients.iter().find(|cached| cached.matches(owner)) { - return Ok(cached.client.clone()); - } - if clients.len() == MAX_PEER_CLIENTS { - clients.pop_front(); - } - clients.push_back(CachedPeerClient { - session: owner.session, - certificate: owner.certificate, - public_key: owner.public_key, - client: client.clone(), - }); - Ok(client) - } - - async fn send_once( - &self, - owner: &RemotePeer, - request: Vec, - remaining_ms: u32, - ) -> crab_cell_runtime::Result { - let client = self.client(owner)?; - let url = owner - .endpoint - .join(PEER_FORWARD_PATH) - .map_err(peer_transport)?; - let response = match client - .post(url) - .header(header::CONTENT_TYPE, PROTOBUF_MEDIA_TYPE) - .header(header::CACHE_CONTROL, "no-store") - .timeout(Duration::from_millis(u64::from(remaining_ms))) - .body(request) - .send() - .await - { - Ok(response) => response, - Err(error) if error.is_connect() => { - return Ok(PeerHttpAttempt::Retry( - peer_transport(error), - Duration::ZERO, - )); - } - Err(error) => return Ok(PeerHttpAttempt::Unknown(peer_transport(error))), - }; - match response.status() { - StatusCode::OK => {} - StatusCode::TOO_MANY_REQUESTS | StatusCode::SERVICE_UNAVAILABLE => { - // Receivers also use a bare 503 to refresh stale ownership. - // Only an explicit delay identifies admission pressure here. - if let Some(seconds) = response - .headers() - .get(header::RETRY_AFTER) - .and_then(|value| value.to_str().ok()) - .and_then(|value| value.parse::().ok()) - { - return Ok(PeerHttpAttempt::Retry( - CellError::Capacity("peer HTTP admission"), - Duration::from_secs(seconds), - )); - } - return Ok(PeerHttpAttempt::Retry( - CellError::CellNotActive, - Duration::ZERO, - )); - } - StatusCode::UNAUTHORIZED | StatusCode::FORBIDDEN => { - return Err(CellError::PeerAuthorization( - "remote node rejected the enrolled peer", - )); - } - status if status.is_server_error() => { - return Ok(PeerHttpAttempt::Unknown(CellError::Peer( - "remote peer returned a server error", - ))); - } - _ => return Err(CellError::Peer("remote peer rejected the HTTP request")), - } - if response - .headers() - .get(header::CONTENT_TYPE) - .and_then(|value| value.to_str().ok()) - != Some(PROTOBUF_MEDIA_TYPE) - || response - .headers() - .get(header::CACHE_CONTROL) - .and_then(|value| value.to_str().ok()) - != Some("no-store") - || response.content_length().is_some_and(|length| { - length > crab_cell_runtime::peer::MAX_PEER_REQUEST_BYTES as u64 - }) - { - return Ok(PeerHttpAttempt::Unknown(CellError::Peer( - "remote peer response metadata is invalid", - ))); - } - let mut body = Vec::new(); - let mut stream = response.bytes_stream(); - while let Some(chunk) = stream.next().await { - let chunk = match chunk { - Ok(chunk) => chunk, - Err(error) => return Ok(PeerHttpAttempt::Unknown(peer_transport(error))), - }; - if body.len().saturating_add(chunk.len()) - > crab_cell_runtime::peer::MAX_PEER_REQUEST_BYTES - { - return Ok(PeerHttpAttempt::Unknown(CellError::Peer( - "remote peer response exceeds the byte limit", - ))); - } - body.extend_from_slice(&chunk); - } - let decoded = match crab_cell_runtime::peer::decode_peer_reply(&body) { - Ok(decoded) => decoded, - Err(error) => return Ok(PeerHttpAttempt::Unknown(error)), - }; - if matches!( - decoded.outcome, - Some(peer_wire::peer_reply::Outcome::Error(ref error)) - if error.code == peer_wire::error::Code::Unavailable as i32 - && error.outcome == peer_wire::error::Outcome::NotStarted as i32 - ) { - return Ok(PeerHttpAttempt::Retry( - CellError::CellNotActive, - Duration::ZERO, - )); - } - Ok(PeerHttpAttempt::Reply(body)) - } -} - -impl PeerRoundTrip for PeerHttpRoundTrip { - fn send( - &self, - target: CellTarget, - request: Vec, - remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let round_trip = self.clone(); - Box::pin(async move { round_trip.send_inner(target, request, remaining_ms).await }) - } - - fn send_to_node( - &self, - target: CellTarget, - node: NodeAdvertisement, - request: Vec, - remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let round_trip = self.clone(); - Box::pin(async move { - round_trip - .send_to_node_inner(target, node, request, remaining_ms) - .await - }) - } -} - -impl Clone for PeerHttpRoundTrip { - fn clone(&self) -> Self { - Self { - scope: Arc::clone(&self.scope), - authority: self.authority.clone(), - directory: self.directory.clone(), - tls: Arc::clone(&self.tls), - session: self.session, - clients: Arc::clone(&self.clients), - } - } -} - -struct RemotePeer { - session: SessionId, - endpoint: url::Url, - certificate: Digest, - public_key: [u8; 32], -} - -struct CachedPeerClient { - session: SessionId, - certificate: Digest, - public_key: [u8; 32], - client: reqwest::Client, -} - -impl CachedPeerClient { - fn matches(&self, owner: &RemotePeer) -> bool { - self.session == owner.session - && self.certificate == owner.certificate - && self.public_key == owner.public_key - } -} - -enum PeerHttpAttempt { - Reply(Vec), - Retry(CellError, Duration), - Unknown(CellError), -} - -fn peer_transport( - source: impl std::error::Error + Send + Sync + 'static, -) -> crab_cell_runtime::Error { - CellError::PeerTransport { - context: "peer HTTP transport failed", - source: Box::new(source), - } -} - -fn now_ms() -> crab_cell_runtime::Result { - let elapsed = SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_err(|_| CellError::Peer("system clock precedes Unix epoch"))?; - i64::try_from(elapsed.as_millis()) - .map_err(|_| CellError::Peer("system clock exceeds peer time range")) -} - -fn remaining_timeout(started: Instant, original_ms: u32) -> crab_cell_runtime::Result { - let elapsed_ms = u32::try_from(started.elapsed().as_millis()).unwrap_or(u32::MAX); - original_ms - .checked_sub(elapsed_ms) - .filter(|remaining| *remaining > 0) - .ok_or(CellError::Deadline) -} diff --git a/crates/crab-cell-peer-http/src/tls.rs b/crates/crab-cell-peer-http/src/tls.rs deleted file mode 100644 index f6bc42d7d..000000000 --- a/crates/crab-cell-peer-http/src/tls.rs +++ /dev/null @@ -1,562 +0,0 @@ -use std::{ - fs::File, - future::Future, - io::{self, BufReader}, - net::SocketAddr, - path::Path, - pin::Pin, - sync::Arc, - task::{Context, Poll}, - time::Duration, -}; - -use axum::extract::connect_info::Connected; -use crab_cell_runtime::Digest as CellDigest; -use ed25519_dalek::{SigningKey, pkcs8::DecodePrivateKey}; -use futures_util::{StreamExt, stream::FuturesUnordered}; -use rustls::{ - CertificateError, ClientConfig, DigitallySignedStruct, RootCertStore, ServerConfig, - SignatureScheme, - client::{ - WebPkiServerVerifier, - danger::{HandshakeSignatureValid, ServerCertVerified, ServerCertVerifier}, - }, - pki_types::{CertificateDer, PrivateKeyDer, ServerName, UnixTime}, - server::WebPkiClientVerifier, -}; -use sha2::{Digest, Sha256}; -use tokio::io::{AsyncRead, AsyncWrite, ReadBuf}; -use tokio_rustls::{TlsAcceptor, server::TlsStream}; -use x509_cert::{Certificate, der::Decode, spki::ObjectIdentifier}; - -use crate::PeerHttpClientFactory; - -/// An invalid pinned peer TLS identity, certificate, or listener setup. -#[derive(Debug, thiserror::Error)] -pub enum TlsError { - /// A required TLS file or identity property is invalid. - #[error("{0}")] - Config(&'static str), - /// A certificate or private-key file could not be read. - #[error("peer TLS file I/O failed")] - Io(#[from] std::io::Error), - /// Certificate validation or TLS client construction failed. - #[error("private Cell TLS setup failed: {context}")] - Setup { - context: &'static str, - #[source] - source: Box, - }, -} - -type Result = std::result::Result; - -const ED25519_OID: ObjectIdentifier = ObjectIdentifier::new_unwrap("1.3.101.112"); -const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(10); -const MAX_PENDING_HANDSHAKES: usize = 128; -const FLEET_DIGEST_DOMAIN: &[u8] = b"crab.peer-ca.v1\0"; - -/// Loaded private peer identity and its verified mTLS server configuration. -pub struct LoadedPeerTls { - config: Arc, - certificates: Vec>, - private_key: PrivateKeyDer<'static>, - roots: Arc, - server_name: ServerName<'static>, - signing_key: SigningKey, - certificate: CellDigest, - fleet: CellDigest, -} - -impl LoadedPeerTls { - /// Loads a CA-trusted Ed25519 identity without exposing key bytes. - /// - /// The leaf must be valid for client and server authentication under the - /// given CA and name; a missing, mismatched, or invalid input returns an error. - pub fn load( - certificate_path: &Path, - private_key_path: &Path, - ca_path: &Path, - server_name: &str, - ) -> Result { - install_crypto_provider(); - let certificates = load_certificates(certificate_path)?; - let private_key = load_private_key(private_key_path)?; - let signing_key = - SigningKey::from_pkcs8_der(private_key.secret_der()).map_err(|source| { - TlsError::Setup { - context: "peer private key is not Ed25519 PKCS#8", - source: Box::new(source), - } - })?; - let leaf_key = certificate_public_key(&certificates[0])?; - if signing_key.verifying_key().to_bytes() != leaf_key { - return Err(TlsError::Config( - "Cell peer certificate and private key do not match", - )); - } - - let authorities = load_certificates(ca_path)?; - let roots = Arc::new(root_store(&authorities)?); - let server_name = - ServerName::try_from(server_name.to_owned()).map_err(|source| TlsError::Setup { - context: "Cell peer TLS server name is invalid", - source: Box::new(source), - })?; - verify_own_certificate(&certificates, Arc::clone(&roots), server_name.clone())?; - let client_verifier = WebPkiClientVerifier::builder(Arc::clone(&roots)) - .build() - .map_err(|source| TlsError::Setup { - context: "Cell peer CA cannot verify clients", - source: Box::new(source), - })?; - let mut config = ServerConfig::builder() - .with_client_cert_verifier(client_verifier) - .with_single_cert(certificates.clone(), private_key.clone_key()) - .map_err(|source| TlsError::Setup { - context: "Cell peer certificate or private key is invalid", - source: Box::new(source), - })?; - config.alpn_protocols = vec![b"http/1.1".to_vec()]; - let certificate = sha256_digest(certificates[0].as_ref()); - - Ok(Self { - config: Arc::new(config), - certificates, - private_key, - roots, - server_name, - signing_key, - certificate, - fleet: fleet_digest(&authorities), - }) - } - - /// Wraps a TCP listener with mutual TLS and verified peer connect info. - pub fn listener(&self, listener: tokio::net::TcpListener) -> PeerTlsListener { - PeerTlsListener { - listener, - acceptor: TlsAcceptor::from(Arc::clone(&self.config)), - handshakes: FuturesUnordered::new(), - } - } - - /// Returns the key that signs this node's advertisement and peer requests. - pub fn signing_key(&self) -> &SigningKey { - &self.signing_key - } - - /// Returns the digest pinned by the node advertisement. - pub const fn certificate(&self) -> CellDigest { - self.certificate - } - - /// Returns the digest of the trusted CA set used to scope the fleet. - pub const fn fleet(&self) -> CellDigest { - self.fleet - } - - /// Builds the outbound identity for owner-resolving peer requests. - pub fn client_identity(&self) -> PeerTlsClient { - PeerTlsClient { - certificates: self.certificates.clone(), - private_key: self.private_key.clone_key(), - roots: Arc::clone(&self.roots), - server_name: self.server_name.clone(), - } - } -} - -/// Fleet-authenticated client identity that pins every request to one enrolled leaf. -pub struct PeerTlsClient { - certificates: Vec>, - private_key: PrivateKeyDer<'static>, - roots: Arc, - server_name: ServerName<'static>, -} - -impl Clone for PeerTlsClient { - fn clone(&self) -> Self { - Self { - certificates: self.certificates.clone(), - private_key: self.private_key.clone_key(), - roots: Arc::clone(&self.roots), - server_name: self.server_name.clone(), - } - } -} - -impl PeerTlsClient { - /// Builds a client pinned to the enrolled server certificate and key. - /// - /// Invalid trust configuration or client identity returns an error. - pub fn client(&self, certificate: CellDigest, public_key: [u8; 32]) -> Result { - let verifier = WebPkiServerVerifier::builder(Arc::clone(&self.roots)) - .build() - .map_err(|source| TlsError::Setup { - context: "Cell peer CA cannot verify servers", - source: Box::new(source), - })?; - let verifier = Arc::new(PinnedServerVerifier { - verifier, - certificate, - public_key, - server_name: self.server_name.clone(), - }); - let mut config = ClientConfig::builder() - .dangerous() - .with_custom_certificate_verifier(verifier) - .with_client_auth_cert(self.certificates.clone(), self.private_key.clone_key()) - .map_err(|source| TlsError::Setup { - context: "Cell peer client identity is invalid", - source: Box::new(source), - })?; - config.alpn_protocols = vec![b"http/1.1".to_vec()]; - reqwest::Client::builder() - .https_only(true) - .redirect(reqwest::redirect::Policy::none()) - .pool_idle_timeout(Duration::from_secs(30)) - .pool_max_idle_per_host(8) - .use_preconfigured_tls(config) - .build() - .map_err(|source| TlsError::Setup { - context: "Cell peer HTTP client initialization failed", - source: Box::new(source), - }) - } -} - -impl PeerHttpClientFactory for PeerTlsClient { - fn client( - &self, - certificate: CellDigest, - public_key: [u8; 32], - ) -> crab_cell_runtime::Result { - PeerTlsClient::client(self, certificate, public_key).map_err(|source| { - crab_cell_runtime::Error::PeerTransport { - context: "peer mTLS client initialization failed", - source: Box::new(source), - } - }) - } -} - -#[derive(Debug)] -struct PinnedServerVerifier { - verifier: Arc, - certificate: CellDigest, - public_key: [u8; 32], - server_name: ServerName<'static>, -} - -impl ServerCertVerifier for PinnedServerVerifier { - fn verify_server_cert( - &self, - end_entity: &CertificateDer<'_>, - intermediates: &[CertificateDer<'_>], - _server_name: &ServerName<'_>, - ocsp_response: &[u8], - now: UnixTime, - ) -> std::result::Result { - let verified = self.verifier.verify_server_cert( - end_entity, - intermediates, - &self.server_name, - ocsp_response, - now, - )?; - let key = certificate_public_key(end_entity).map_err(|_| pin_error())?; - if sha256_digest(end_entity.as_ref()) != self.certificate || key != self.public_key { - return Err(pin_error()); - } - Ok(verified) - } - - fn verify_tls12_signature( - &self, - message: &[u8], - certificate: &CertificateDer<'_>, - signature: &DigitallySignedStruct, - ) -> std::result::Result { - self.verifier - .verify_tls12_signature(message, certificate, signature) - } - - fn verify_tls13_signature( - &self, - message: &[u8], - certificate: &CertificateDer<'_>, - signature: &DigitallySignedStruct, - ) -> std::result::Result { - self.verifier - .verify_tls13_signature(message, certificate, signature) - } - - fn supported_verify_schemes(&self) -> Vec { - self.verifier.supported_verify_schemes() - } -} - -const fn pin_error() -> rustls::Error { - rustls::Error::InvalidCertificate(CertificateError::ApplicationVerificationFailure) -} - -/// Identity extracted only after rustls validates the complete client chain. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct PeerTlsIdentity { - certificate: CellDigest, - public_key: [u8; 32], -} - -impl PeerTlsIdentity { - /// Returns the verified client's leaf certificate digest. - pub const fn certificate(&self) -> CellDigest { - self.certificate - } - - /// Returns the verified client's Ed25519 leaf public key. - pub const fn public_key(&self) -> [u8; 32] { - self.public_key - } -} - -/// Axum listener that accepts only CA-trusted mutual TLS peers. -pub struct PeerTlsListener { - listener: tokio::net::TcpListener, - acceptor: TlsAcceptor, - handshakes: FuturesUnordered, -} - -type Handshake = Pin> + Send>>; - -impl axum::serve::Listener for PeerTlsListener { - type Io = PeerTlsStream; - type Addr = SocketAddr; - - async fn accept(&mut self) -> (Self::Io, Self::Addr) { - loop { - tokio::select! { - ready = self.handshakes.next(), if !self.handshakes.is_empty() => { - if let Some(Some(accepted)) = ready { - return accepted; - } - } - accepted = self.listener.accept(), if self.handshakes.len() < MAX_PENDING_HANDSHAKES => { - match accepted { - Ok((stream, address)) => { - let acceptor = self.acceptor.clone(); - self.handshakes.push(Box::pin(handshake(acceptor, stream, address))); - } - Err(error) => { - tracing::warn!(error = %error, "private listener accept failed"); - tokio::time::sleep(Duration::from_secs(1)).await; - } - } - } - } - } - } - - fn local_addr(&self) -> io::Result { - self.listener.local_addr() - } -} - -async fn handshake( - acceptor: TlsAcceptor, - stream: tokio::net::TcpStream, - address: SocketAddr, -) -> Option<(PeerTlsStream, SocketAddr)> { - let stream = match tokio::time::timeout(HANDSHAKE_TIMEOUT, acceptor.accept(stream)).await { - Ok(Ok(stream)) => stream, - Ok(Err(error)) => { - tracing::debug!(error = %error, "private mTLS handshake rejected"); - return None; - } - Err(_) => { - tracing::debug!("private mTLS handshake timed out"); - return None; - } - }; - let identity = match tls_identity(stream.get_ref().1.peer_certificates()) { - Ok(identity) => identity, - Err(error) => { - tracing::warn!(error = %error, "verified private peer identity is invalid"); - return None; - } - }; - Some((PeerTlsStream { stream, identity }, address)) -} - -/// Verified TLS stream retaining the connecting peer's leaf identity. -pub struct PeerTlsStream { - stream: TlsStream, - identity: PeerTlsIdentity, -} - -impl AsyncRead for PeerTlsStream { - fn poll_read( - mut self: Pin<&mut Self>, - context: &mut Context<'_>, - buffer: &mut ReadBuf<'_>, - ) -> Poll> { - Pin::new(&mut self.stream).poll_read(context, buffer) - } -} - -impl AsyncWrite for PeerTlsStream { - fn poll_write( - mut self: Pin<&mut Self>, - context: &mut Context<'_>, - buffer: &[u8], - ) -> Poll> { - Pin::new(&mut self.stream).poll_write(context, buffer) - } - - fn poll_flush(mut self: Pin<&mut Self>, context: &mut Context<'_>) -> Poll> { - Pin::new(&mut self.stream).poll_flush(context) - } - - fn poll_shutdown(mut self: Pin<&mut Self>, context: &mut Context<'_>) -> Poll> { - Pin::new(&mut self.stream).poll_shutdown(context) - } -} - -impl Connected> for PeerTlsIdentity { - fn connect_info(stream: axum::serve::IncomingStream<'_, PeerTlsListener>) -> Self { - stream.io().identity.clone() - } -} - -fn load_certificates(path: &Path) -> Result>> { - let file = File::open(path)?; - let certificates = rustls_pemfile::certs(&mut BufReader::new(file)) - .collect::, _>>()?; - if certificates.is_empty() { - return Err(TlsError::Config("Cell peer PEM contains no certificates")); - } - Ok(certificates) -} - -fn load_private_key(path: &Path) -> Result> { - rustls_pemfile::private_key(&mut BufReader::new(File::open(path)?))? - .ok_or(TlsError::Config("Cell peer PEM contains no private key")) -} - -fn root_store(authorities: &[CertificateDer<'static>]) -> Result { - let mut roots = RootCertStore::empty(); - for authority in authorities { - roots - .add(authority.clone()) - .map_err(|source| TlsError::Setup { - context: "Cell peer CA certificate is invalid", - source: Box::new(source), - })?; - } - Ok(roots) -} - -fn verify_own_certificate( - certificates: &[CertificateDer<'static>], - roots: Arc, - server_name: ServerName<'static>, -) -> Result<()> { - let leaf = &certificates[0]; - let intermediates = &certificates[1..]; - let client = WebPkiClientVerifier::builder(Arc::clone(&roots)) - .build() - .map_err(|source| TlsError::Setup { - context: "Cell peer CA cannot verify clients", - source: Box::new(source), - })?; - client - .verify_client_cert(leaf, intermediates, UnixTime::now()) - .map_err(|source| TlsError::Setup { - context: "Cell peer certificate is not valid for client authentication", - source: Box::new(source), - })?; - - WebPkiServerVerifier::builder(roots) - .build() - .map_err(|source| TlsError::Setup { - context: "Cell peer CA cannot verify servers", - source: Box::new(source), - })? - .verify_server_cert(leaf, intermediates, &server_name, &[], UnixTime::now()) - .map_err(|source| TlsError::Setup { - context: "Cell peer certificate is not valid for its advertised endpoint", - source: Box::new(source), - })?; - Ok(()) -} - -fn tls_identity(certificates: Option<&[CertificateDer<'static>]>) -> Result { - let leaf = certificates - .and_then(|certificates| certificates.first()) - .ok_or(TlsError::Config( - "verified Cell peer certificate is missing", - ))?; - Ok(PeerTlsIdentity { - certificate: sha256_digest(leaf.as_ref()), - public_key: certificate_public_key(leaf)?, - }) -} - -fn certificate_public_key(certificate: &CertificateDer<'_>) -> Result<[u8; 32]> { - let certificate = - Certificate::from_der(certificate.as_ref()).map_err(|source| TlsError::Setup { - context: "Cell peer certificate is malformed", - source: Box::new(source), - })?; - let key = &certificate.tbs_certificate.subject_public_key_info; - if key.algorithm.oid != ED25519_OID || key.algorithm.parameters.is_some() { - return Err(TlsError::Config("Cell peer certificate must use Ed25519")); - } - key.subject_public_key - .as_bytes() - .and_then(|bytes| bytes.try_into().ok()) - .ok_or(TlsError::Config( - "Cell peer certificate has an invalid Ed25519 public key", - )) -} - -fn sha256_digest(bytes: &[u8]) -> CellDigest { - CellDigest::from_bytes(Sha256::digest(bytes).into()) -} - -fn fleet_digest(authorities: &[CertificateDer<'static>]) -> CellDigest { - let mut authorities = authorities - .iter() - .map(|certificate| certificate.as_ref()) - .collect::>(); - authorities.sort_unstable(); - authorities.dedup(); - let mut digest = Sha256::new(); - digest.update(FLEET_DIGEST_DOMAIN); - for authority in authorities { - digest.update((authority.len() as u64).to_be_bytes()); - digest.update(authority); - } - CellDigest::from_bytes(digest.finalize().into()) -} - -fn install_crypto_provider() { - static PROVIDER: std::sync::Once = std::sync::Once::new(); - PROVIDER.call_once(|| { - let _ = rustls::crypto::aws_lc_rs::default_provider().install_default(); - }); -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn fleet_digest_is_independent_of_ca_order_and_duplicates() { - let first = CertificateDer::from(vec![1, 2, 3]); - let second = CertificateDer::from(vec![4, 5, 6]); - let ordered = fleet_digest(&[first.clone(), second.clone()]); - let reordered = fleet_digest(&[second, first.clone(), first]); - assert_eq!(ordered, reordered); - } -} diff --git a/crates/crab-cell-runtime/AGENTS.md b/crates/crab-cell-runtime/AGENTS.md deleted file mode 100644 index 1803d0eac..000000000 --- a/crates/crab-cell-runtime/AGENTS.md +++ /dev/null @@ -1,123 +0,0 @@ -# crab-cell-runtime - -Root `AGENTS.md`, `crates/AGENTS.md`, and `docs/README.md` apply. - -## Purpose and ownership - -Embedded SQLite Cell runtime: identities, control/CAS authority, the single-Cell -actor and executor, schema installation, publication, follower durability, -fleet placement, and qualification receipts. HTTP, authentication, provider -construction, and product policy stay in `crab-http-server`. - -## Read first - -1. `src/lib.rs` — module declarations and the frozen root prelude. -2. `src/cell/actor.rs` — the actor root. `actor/task.rs` and - `actor/requests.rs` drive the loop and the request paths, `actor/tasks.rs` - and `actor/lifecycle.rs` handle finished tasks and cell scheduling, - `actor/{handle,state,runtime}.rs` own the handle, projections, and runtime - administration, and `actor/admission.rs` fences all of them. -3. `src/cell/executor.rs` and `src/cell/worker.rs` with `src/cell/worker/run.rs` - — command execution and the bounded SQL worker pool. -4. `src/publication.rs` and `src/recovery/manifest.rs` — exact-root publication - and recovery artifacts. -5. `src/coordination.rs` — the pure `pub(crate)` coordination kernel, with the - deterministic simulator in `src/coordination/sim.rs`. -6. `docs/runtime.md` and `docs/delivery.md` — request path and evidence map. - -## Module map - -Subsystem roots keep the shared contract and helpers; the named child modules -own one concern each. Module files sit beside their root (`foo.rs` + `foo/`). - -- Cell: `cell/{actor.rs,catalog.rs,application.rs,executor.rs,schema.rs,worker.rs}`, - `cell/actor/{admission,handle,lifecycle,requests,runtime,state,task,tasks}.rs`, - `cell/worker/run.rs`. -- Control and clients: `control.rs` + `control/authority.rs`, `client.rs`, - `peer.rs` + `peer/{dispatch,protobuf,transport}.rs`. -- Durability and followers: `follower.rs` + `follower/records.rs`, - `node/{advertisement,capacity,durability,lease,log,log_shipper,log_state,log_transport}.rs`, - `node/directory.rs` + `node/directory/{advertisement,log,recovery}.rs`, - `node/log_recovery.rs` + `node/log_recovery/witness.rs`. -- Fleet and recovery: `fleet/{scheduler,placement,pressure,eviction,resource,telemetry}.rs`, - `recovery/{manifest,release,release_progress,artifacts}.rs`, - `recovery/backup/restore.rs`, `recovery/retention.rs`. -- Primitives: `primitives/.rs` with the wire codecs in - `primitives//api.rs`; Effects adds `supervisor.rs` and Workflow adds - `activity.rs`, `activity_api.rs`, `activity_codec.rs`, and `maintenance.rs`. -- Registry and qualification: `registry/{builder,descriptor,handlers,schemas}.rs`, - `qualification/{profile,receipt,workload,cluster}.rs`, - `qualification/receipt/{matrix,runner}.rs`, `qualification/tests/`. - -## Common changes - -| Task | Start here | Also inspect | -| --- | --- | --- | -| Add a primitive operation | `src/primitives/.rs` | `src/registry/schemas.rs`, `tests/primitives/` | -| Add primitive maintenance | `src/fleet/scheduler.rs` | that primitive's `TABLE` constant, `tests/runtime/scheduler.rs` | -| Wire or defer a policy seam | `src/cell/actor.rs` | `crab/scripts/check-policy-entry-points.py` | -| Change admission or lifecycle | `src/cell/actor/admission.rs`, `src/cell/actor/lifecycle.rs` | `src/coordination.rs`, `tests/runtime/lifecycle.rs` | -| Change publication | `src/publication.rs` | `src/recovery/`, `tests/runtime/publication.rs` | -| Change node log or durability | `src/node/` | `src/follower.rs`, `tests/fleet/` | -| Change placement or pressure | `src/fleet/` | `tests/fleet/`, `docs/canonical-ltx-scaling.md` | -| Change pressure sampling or shedding | `src/cell/actor/task.rs` | `src/cell/actor/runtime.rs`, `src/fleet/pressure.rs`, `src/fleet/telemetry.rs`, `tests/fleet/pressure.rs` | - -## Layout and tests - -- `src/` is production code. Integration tests live in `tests/` as one binary - per suite: `runtime`, `primitives`, `protocol`, `contracts`, `fleet`, - `qualification`. Suite modules live in the matching directory. -- `tests/support/` holds the shared harness; suites declare `mod support;` and - refer to `crate::support::…`. Never add a `#[path]` attribute. -- Tests that assert crate-private behavior stay in their module and are listed - in `tests-allow-list.txt` with a reason. New in-src tests must be added there; - prefer moving the behavior behind the public API when that is honest. - Each entry must name a file that still holds tests or test modules, so a moved - or emptied test location cannot leave a stale entry behind. -- `api-prelude.txt` is the frozen root surface. Adding a root re-export means - editing both `src/lib.rs` and that file in the same commit. -- Run `python3 crab/scripts/check-cell-ltx-layout.py` after layout changes. - -## Invariants - -- One fenced writer per Cell; a successful response follows durable publication - or a durable follower proof (`src/cell/actor.rs`, `src/publication.rs`). -- Recovery verifies the authority-pinned root and every referenced object - (`src/recovery/manifest.rs`). -- Staged xorbs flush before any bundle publication. -- Every acquired lock is released on success, error, cancellation, and timeout. -- A node sheds settled Cells only on sustained evidence: the actor samples its own - reservation ledger on its tick, the classifier hands a tier to the bounded - eviction path only after a full dwell window above the enter threshold, and - that tier is what the node reports through - `CellTelemetry::pressure_state` (`src/cell/actor/task.rs`, - `src/fleet/pressure.rs`). -- `src/coordination.rs` stays sans-I/O: no `async`, no clock, no storage - (`crab/scripts/check-cell-ltx-layout.py` enforces it). - -## Features and platform - -- `test-support` enables `src/test_support.rs` and the `cell_movement_probe` - binary. Integration suites that need the process fixture run with - `--features test-support`. -- `crab-ltx` is always consumed with its `replica` feature from this crate. - -## Verification - -```sh -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/ \ - cargo test -p crab-cell-runtime --features test-support --locked -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/ \ - cargo clippy -p crab-cell-runtime --all-targets --features test-support --locked -- -D warnings -python3 crab/scripts/check-cell-ltx-layout.py -``` - -Use a target directory unique to the checkout; never share it between -worktrees. Protected qualification receipts come from the workflows under -`.github/workflows/cell-runtime-*.yml`, not from local runs. - -## Related documentation - -`docs/README.md`, `docs/runtime.md`, `docs/delivery.md`, -`docs/canonical-ltx-scaling.md`, `qualification/README.md`, and -`../../advisor-plans/033-cell-ltx-layout-reorganization.md`. diff --git a/crates/crab-cell-runtime/CLAUDE.md b/crates/crab-cell-runtime/CLAUDE.md deleted file mode 120000 index 47dc3e3d8..000000000 --- a/crates/crab-cell-runtime/CLAUDE.md +++ /dev/null @@ -1 +0,0 @@ -AGENTS.md \ No newline at end of file diff --git a/crates/crab-cell-runtime/Cargo.toml b/crates/crab-cell-runtime/Cargo.toml deleted file mode 100644 index 8c7a42ba7..000000000 --- a/crates/crab-cell-runtime/Cargo.toml +++ /dev/null @@ -1,48 +0,0 @@ -[package] -name = "crab-cell-runtime" -version = "0.1.0" -edition.workspace = true -license.workspace = true -repository.workspace = true -rust-version.workspace = true -publish = false -description = "Embedded SQLite Cell runtime contracts for Crab" - -[[bin]] -name = "cell_movement_probe" -path = "src/bin/cell_movement_probe.rs" -required-features = ["test-support"] - -[features] -default = [] -test-support = ["dep:async-trait", "dep:fs4"] - -[dependencies] -async-trait = { workspace = true, optional = true } -blake3.workspace = true -bytes.workspace = true -crab-storage.workspace = true -crab-ltx = { workspace = true, features = ["replica"] } -ed25519-dalek.workspace = true -fs4 = { version = "0.13", optional = true } -futures-util.workspace = true -object_store.workspace = true -prost.workspace = true -rusqlite = { workspace = true, features = ["hooks", "blob"] } -rand = "0.9" -serde.workspace = true -serde_json.workspace = true -tempfile.workspace = true -thiserror.workspace = true -tokio = { workspace = true, features = ["rt", "sync", "time"] } -tokio-util.workspace = true -tracing.workspace = true - -[build-dependencies] -prost-build = "0.13" -protoc-bin-vendored = "3" - -[dev-dependencies] -async-trait.workspace = true -proptest = "1" -tokio = { workspace = true, features = ["macros", "rt-multi-thread"] } diff --git a/crates/crab-cell-runtime/README.md b/crates/crab-cell-runtime/README.md deleted file mode 100644 index 2f0adc996..000000000 --- a/crates/crab-cell-runtime/README.md +++ /dev/null @@ -1,51 +0,0 @@ -# crab-cell-runtime - -Embedded SQLite Cell runtime for Crab services: Cell identities, control/CAS -authority, one single-writer actor per Cell, schema installation, exact-root -LTX publication, follower durability, fleet placement, and qualification -receipts. HTTP, authentication, and provider construction stay in -`crab-http-server`. - -An opt-in library read path can open an exact S3-rooted, read-only Cell view -through `CellReadReplica`. Its view and replacement refresh are charged to the -node runtime's memory, descriptor, and disk ledgers. The charges are provisional; -product routing and measured capacity qualification remain open under -[Plan 036](../../advisor-plans/036-cell-read-replicas-and-fenced-promotion.md). - -## Module map - -| Module | Responsibility | -| --- | --- | -| `identity` | Cell, tenant, session, namespace, node, and digest identities | -| `control` | Control record, transitions, and CAS authority | -| `codec` | Bounded wire codec used by modules and peers | -| `registry` | Module/command/query descriptors and the compiled registry | -| `cell` | Actor, executor, worker pool, catalog, schema, application identity | -| `client` | Typed client, prepared commands, state streams | -| `primitives` | SQL, KV, Blob, Queue, Cron, Workflow, Effects, activity pool | -| `publication` | Exact-root LTX publication | -| `follower` | Follower store, lanes, and tail pages | -| `node` | Signed advertisements, node log, recovery, durability, leases | -| `recovery` | Recovery manifests, artifacts, releases, pins, retention | -| `fleet` | Placement, pressure, admission accounting, eviction, scheduling | -| `peer` | Authenticated peer protocol | -| `qualification` | Qualification profiles, workloads, and receipts | -| `ltx` | LTX types this crate exposes to embedders | - -The root also re-exports a small prelude for embedders, frozen in -[`api-prelude.txt`](api-prelude.txt). - -## Tests - -`tests/` holds one binary per suite (`runtime`, `primitives`, `protocol`, -`contracts`, `fleet`, `qualification`) with shared fixtures in -`tests/support/`. Modules whose tests must assert crate-private behavior are -listed in [`tests-allow-list.txt`](tests-allow-list.txt). - -```sh -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/ \ - cargo test -p crab-cell-runtime --features test-support --locked -``` - -See `docs/README.md` for the runtime design and `AGENTS.md` for contributor -rules. diff --git a/crates/crab-cell-runtime/api-prelude.txt b/crates/crab-cell-runtime/api-prelude.txt deleted file mode 100644 index 63d39d176..000000000 --- a/crates/crab-cell-runtime/api-prelude.txt +++ /dev/null @@ -1,60 +0,0 @@ -ApplicationId -BlobArtifactStore -BlobModule -BlobNamespace -BuildDescriptor -CatalogRole -CellClient -CellReadReplica -CellId -CellModule -CellRuntime -CellRuntimeStats -CellTarget -Command -Committed -CronModule -CronNamespace -Digest -EffectModule -EffectSource -Error -FollowerStore -InvocationError -KvModule -KvNamespace -MigrationDescriptor -ModuleDescriptor -MutationIdentity -NamespaceDescriptor -NamespaceId -NodeDurabilityConfig -NodeLeaseGuard -Observed -PendingMutation -PreparedCommand -QualificationExecution -QualificationOperation -QualificationOperationExecutor -QualificationProfile -QualificationRunSummary -QualificationWorkload -Query -QueueModule -QueueNamespace -Receipt -Registry -RegistryBuilder -Resolution -Result -SessionId -SqlCell -SqlModule -SqlWorkerPool -TenantId -WorkflowActivities -WorkflowActivityModule -WorkflowModule -WorkflowNamespace -partition_for_shard -shard_for_scope diff --git a/crates/crab-cell-runtime/build.rs b/crates/crab-cell-runtime/build.rs deleted file mode 100644 index 1376b072c..000000000 --- a/crates/crab-cell-runtime/build.rs +++ /dev/null @@ -1,9 +0,0 @@ -fn main() -> Result<(), Box> { - let proto = "docs/contracts/peer.proto"; - println!("cargo:rerun-if-changed={proto}"); - let protoc = protoc_bin_vendored::protoc_bin_path()?; - let mut config = prost_build::Config::new(); - config.protoc_executable(protoc); - config.compile_protos(&[proto], &["docs/contracts"])?; - Ok(()) -} diff --git a/crates/crab-cell-runtime/docs/README.md b/crates/crab-cell-runtime/docs/README.md deleted file mode 100644 index f0e884462..000000000 --- a/crates/crab-cell-runtime/docs/README.md +++ /dev/null @@ -1,249 +0,0 @@ -# Understand the embedded Cell runtime - -Crab stores each repository's collaboration state in one SQLite **Cell**. A Cell has one active writer, publishes immutable Log Transaction (LTX) data to object storage, and can reopen on another Crab node. Product code stays in Rust and is compiled into `crab-http-server`. - -| Document intent | Value | -| --- | --- | -| Content type | Conceptual landing page | -| Audience | Crab contributors and fleet operators | -| Goal | Explain the runtime boundary, request path, durability point, and reading order | -| Status | Implemented, with production capacity and multi-Pod fault qualification still required | - -## See the system in one diagram - -The public HTTP server owns authentication and repository policy. The runtime owns deterministic execution, SQLite state, publication, and takeover. - -![Crab Cell runtime request, ownership, execution, and storage architecture](diagram/system-architecture.svg) - -The direct green route is local execution. The orange route is the single authenticated peer hop when another node owns the Cell. Both converge on the same registry, actor, SQLite, and LTX publication path. - -`CellClient::local_runtime` resolves a newly admitted local Cell through its -catalog and owner record for each call. It serves embedded, single-node routing; -`CellClient::runtime_with_peer` uses the same local path when this node owns the -Cell and an authenticated peer round trip when another node owns it. The product -server supplies the peer transport and owner lookup; neither constructor -acquires an idle Cell. - -The dependency direction follows the same boundary: - -```text -crab-storage <- crab-ltx <- crab-cell-runtime <- crab-http-server -``` - -Lower crates never import HTTP, Git, repository authorization, or provider configuration. - -## Follow one mutation - -A successful mutation response means its SQLite outcome is covered either by -the exact object-store root or by every selected follower's fsynced node-log -tail. Fleet-only outcomes are recovered before a successor serves the Cell; -object publication continues while later work stays queued on that Cell. - -```mermaid -sequenceDiagram - participant C as Client - participant H as HTTP route - participant A as Cell actor - participant S as SQLite - participant O as Object store - - C->>H: Authenticated product request - H->>A: Typed command + stable request ID - A->>S: Savepoint, handler, request outcome - S-->>A: Committed WAL cut - A->>O: Upload immutable LTX dependencies - A->>O: CAS control.json to exact root - O-->>A: New ETag - A-->>H: Committed + receipt - H-->>C: Product response -``` - -If the control compare-and-swap (CAS) result is unknown, the actor reloads authority. It accepts only the exact proposed successor. It never reruns the SQL callback to guess the result. - -## Know what a Cell contains - -Each Cell combines runtime metadata and one application schema in the same SQLite transaction. - -| Layer | Stored data | Owner | -| --- | --- | --- | -| Runtime | Request outcomes, effects, inbox, sequence, due summary | `crab-cell-runtime` | -| Application | Repository collaboration rows or one primitive shard | Compiled Rust module | -| Local cache | SQLite main file, WAL, retained LTX, sparse pages | Current Crab node | -| Durable data | Immutable roots, LTX bodies, indexes, control record | Object store | -| External product data | Git, Xet, LFS, release assets | Existing Crab subsystems | - -One command changes one Cell. Cross-Cell work uses durable effects and idempotent destination inboxes, not distributed SQL transactions. - -## Understand the ownership model - -The object-store control record is the authority for the Cell's owner and root. - -```mermaid -stateDiagram-v2 - [*] --> Recovering: provision - Recovering --> Serving: publish initial root - Serving --> Serving: command or renewal - Serving --> Idle: clean drain - Idle --> Recovering: acquire exact root - Serving --> Recovering: stale-owner takeover - Recovering --> Idle: activation fails cleanly - Idle --> Tombstoned: administrative delete -``` - -The runtime applies these rules: - -- **Single writer**: one owner session and epoch may publish the next root -- **Fencing**: an ownership mismatch closes admission before more SQL runs -- **Exact recovery**: takeover opens the root named by authoritative control -- **Disposable owner-local SQLite**: selected followers may durably fsync recent - LTX tails, but the owner's mutable SQLite files remain disposable caches -- **Bounded work**: commands, results, queues, workers, memory, and disk have explicit limits - -An application query can use `QueryContext::database_used_bytes()` to inspect its -Cell's occupied SQLite pages, including indexes and runtime tables. Reusable -freelist pages, WAL, and LTX files are excluded. Capacity control must account -for the latter resources separately. The optional capacity primitive protects -durable page claims for deferred work; see [runtime.md](runtime.md). - -Read [runtime.md](runtime.md) for the actor and failure state machines. Read [storage.md](storage.md) for identity, control, root, and LTX formats. - -## Build applications as native Rust modules - -V1 is not a general code-hosting platform. A Crab contributor registers typed Rust handlers at build time. - -```rust,ignore -impl Command for CreateIssue { - const MODULE: &'static str = "repository"; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - - type Input = CreateIssueInput; - type Output = Issue; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> Result> { - let rows = context.sql(&input.insert_batch())?; - Ok(CommandResult::Success(Issue::decode(rows)?)) - } -} -``` - -Handlers receive bounded transaction capabilities. They don't receive raw storage credentials, database paths, or network access. - -The runtime has no JavaScript host, WebAssembly host, dynamic library loader, or public primitive endpoint. Browsers continue to use Crab's product HTTP API. - -Read [rust-api.md](rust-api.md) for module registration, typed commands, queries, activities, and peer routing. - -## Choose a persistence primitive - -The primitives share the same actor, transaction, publication, recovery, and admission path. - -| Primitive | Use it for | Partition key | Delivery contract | -| --- | --- | --- | --- | -| SQL | Repository-local relational state | Explicit repository UUID | Serializable single-Cell command | -| KV | Scoped metadata and atomic checks | Scope hash | Atomic batch within one shard | -| Blob | Object-store parts and range reads | Object-key hash | SQLite manifest and content-addressed part references within one shard | -| Queue | Deferred work | Producer hash for send, explicit shard for claim | At least once | -| Cron | Recurring typed triggers | Schedule-ID hash | Atomic occurrence effect and schedule advance | -| Workflow | Durable state machines, timers, activities | Workflow ID hash | Deterministic transition plus retryable activity | - -Read [primitives.md](primitives.md) for schemas, state transitions, limits, and examples. - -## Deploy one Crab server per node - -Each Kubernetes Pod or virtual machine runs one `crab-http-server` process. Every eligible node compiles the same registry and advertises its release, capacity, and scheduler progress. - -```mermaid -flowchart TB - LB[External load balancer] - N1[Crab node A] - N2[Crab node B] - N3[Crab node C] - Origin[(Shared object-store origin)] - - LB --> N1 - LB --> N2 - LB --> N3 - N1 <-->|private mTLS| N2 - N2 <-->|private mTLS| N3 - N1 --> Origin - N2 --> Origin - N3 --> Origin -``` - -The external load balancer may send a request to any node. That node resolves the authoritative Cell owner and either executes locally or forwards once over the authenticated management network. - -Read [deployment.md](deployment.md) for node sizing, release activation, Kubernetes lifecycle, backup, and cutover. - -## Apply the hard cutover contract - -Cell data starts empty. Crab does not import, dual-read, dual-write, or fall back to the retired native-bucket collaboration format. - -The operator performs this sequence: - -1. Stop and fence every legacy writer -2. Manually delete retired `app/v1` collaboration keys and the old HTTP catalog -3. Keep canonical Git, Xet, LFS, and release-asset objects -4. Run repository adoption for each retained Git repository -5. Verify that adoption published a new empty Cell before enabling traffic - -This rule removes compatibility branches from product code. It does not permit deletion of canonical Git objects. - -## Use the documentation by task - -| Task | Read | -| --- | --- | -| Design an application from entity, shard, workflow, and read-model Cells | [Application framework](application-framework.md) | -| Understand the actor, publication, timeout, or takeover path | [Runtime execution](runtime.md) | -| Design follower durability, response gating, and warm failover | [Follower durability and warm failover](failover-and-followers.md) | -| Complete canonical LTX scaling and decide standalone replication | [Canonical Cell LTX scaling](canonical-ltx-scaling.md) | -| Execute owner routing, VFS/LTX, and application scale qualification | [SQLite VFS and LTX scaling plan](vfs-ltx-scale-plan.md) | -| Prioritize LTX latency and publication-capacity experiments | [LTX performance audit](ltx-performance-audit.md) | -| Inspect persistent identities, paths, control JSON, or LTX roots | [Storage and recovery](storage.md) | -| Implement SQL, KV, Blob, Queue, Cron, Workflow, or effects | [Primitive contracts](primitives.md) | -| Add a native product feature | [Rust programming model](rust-api.md) | -| Size or operate a fleet | [Deployment and operations](deployment.md) | -| Verify implementation coverage and remaining gates | [Delivery and qualification](delivery.md) | -| Inspect normative schemas or peer messages | [`contracts/`](contracts/) | - -## Treat these contracts as normative - -Contract precedence is: - -1. SQL and Protocol Buffers files in [`contracts/`](contracts/) -2. Rust types and validation in `crab-cell-runtime` -3. This documentation for ordering, ownership, and operational policy - -The validation script checks SQL constraints, Protocol Buffers fixtures, Markdown links, and code fences: - -```bash -node crates/crab-cell-runtime/docs/validate.mjs -``` - -## Respect the initial limits - -These values are admission contracts, not benchmark results. - -| Resource | Initial limit | -| --- | ---: | -| Command input or result | 1 MiB by default; KV may declare 4 MiB plus 64 KiB framing | -| Workflow state | 1 MiB | -| SQL batch | 128 statements | -| SQL query result | 1,000 rows and 1 MiB | -| KV atomic batch | 128 mutations | -| KV value | 4 MiB | -| Blob part / range read | 256 KiB / 512 KiB | -| Queue payload | 256 KiB | -| Cron payload / interval | 256 KiB / 1 second to 1 year | -| Queue or activity lease | 5s to 300s, 30s default | -| Native transaction wall budget | 5s | -| Public transport wait | 30s default, 60s maximum | -| Owner renewal | Every 3s | -| Self-fence threshold | 10s | -| Takeover observation | 15s unchanged control | -| Open Cells per node target | 1,000 to 10,000 | -| Aggregate command target | 1,000 commands/s per node | - -Node profiles do not promise these targets without the qualification described in [delivery.md](delivery.md#capacity-qualification). diff --git a/crates/crab-cell-runtime/docs/application-framework-example.md b/crates/crab-cell-runtime/docs/application-framework-example.md deleted file mode 100644 index 58b88e12c..000000000 --- a/crates/crab-cell-runtime/docs/application-framework-example.md +++ /dev/null @@ -1,1053 +0,0 @@ -# Build a complete Commerce application - -This example exercises the complete proposed application framework over the -Cell runtime: custom SQL entity and read-model Cells, KV, Blob, Queue, Cron, -Workflow, cross-Cell effects, external activities, generated clients, node -composition, authenticated HTTP adaptation, receipted reads, ambiguous-result -resolution, and owner-loss recovery. - -| Document intent | Value | -| --- | --- | -| Content type | End-to-end target API example | -| Audience | Application owners and framework implementers | -| Goal | Make the proposed application framework concrete enough to implement and evaluate | -| Status | Mixed status: the handwritten `crab-cell-app` reference test registers SQL, KV, Blob, Queue, Cron, Workflow, Activity, and Effects through one descriptor, and `CellNode` is used by the server; the Commerce snippets below remain an illustrative target while generated clients, full operator ownership, and protected owner-loss evidence remain open | - -[Back to the application framework design](application-framework.md) - -The example is an executable design target, not a claim that every snippet -compiles today. Existing low-level contracts named here—`Command`, `Query`, -`CellClient`, primitive mechanics, receipts, effects, activities, exact roots, -and runtime publication—are implemented. The current `crab-cell-app` reference -test proves registration and one successful typed invocation for SQL, KV, Blob, -Queue, Cron, Workflow, Activity, and Effects through a bounded local multi-Cell -router, while -`crab-cell-host` and `crab-http-server` prove the initial node-facade adoption; -generated clients, complete operator ownership, and protected provider -qualification still require the remaining plans. - -## Follow the application flow - -The Commerce application uses every persistence and coordination primitive for -one coherent request: - -```mermaid -flowchart LR - HTTP[Authenticated HTTP request] - Cart[KV cart shard] - Order[SQL Order Cell] - Checkout[Workflow Cell] - Stock[SQL inventory shard] - Payment[Payment activity] - Jobs[Queue shard] - Invoice[Blob shard] - Index[SQL read-model shard] - Renew[Cron shard] - - HTTP --> Cart - HTTP --> Order - Order -->|effect| Checkout - Checkout -->|effect| Stock - Stock -->|effect| Checkout - Checkout --> Payment - Checkout -->|effect| Jobs - Checkout -->|effect| Index - Jobs --> Invoice - Renew -->|effect| Checkout -``` - -1. A customer builds a cart in sharded KV. -2. `PlaceOrder` commits the order and a workflow-start effect in one Order Cell. -3. The Checkout workflow sends typed reservation effects to inventory shards. -4. Inventory shards deduplicate reservations and signal the workflow. -5. A payment activity calls the external provider with a stable idempotency key. -6. The workflow sends a fulfillment job and customer-index projection effects. -7. A worker claims the job, renders and stores an invoice through Blob, sends - mail, and acknowledges the exact queue lease. -8. Cron starts the same workflow contract for subscription renewals. - -There is no multi-Cell transaction. Each arrow is either a published command, -a durable effect with inbox deduplication, or a leased activity with an explicit -external idempotency contract. - -## Organize the application crate - -The application keeps domain code separate from node and transport policy: - -```text -commerce/ -├── Cargo.toml -├── migrations/ -│ ├── orders/0001.sql -│ ├── inventory/0001.sql -│ └── customer_order_index/0001.sql -└── src/ - ├── lib.rs - ├── ids.rs - ├── values.rs - ├── orders.rs - ├── inventory.rs - ├── carts.rs - ├── invoices.rs - ├── fulfillment.rs - ├── renewals.rs - ├── checkout.rs - ├── customer_order_index.rs - ├── activities.rs - ├── worker.rs - ├── http.rs - └── main.rs -``` - -The generated module contributes `generated::CommerceClient`, descriptor -fixtures, typed namespace accessors, operation dispatch, and release bytes. It -does not contain application authorization or provider credentials. - -## Declare stable values and identifiers - -All values that cross the runtime boundary use a bounded canonical codec. The -derive is proposed syntax for generating the existing `WireValue` contract. - -```rust,ignore -use crab_cell_app::CellValue; - -#[derive(Clone, Copy, CellValue, PartialEq, Eq)] -pub struct OrderId(pub [u8; 16]); - -#[derive(Clone, Copy, CellValue, PartialEq, Eq)] -pub struct CustomerId(pub [u8; 16]); - -#[derive(Clone, CellValue, PartialEq, Eq)] -pub struct LineItem { - #[cell(max_bytes = 64)] - pub sku: String, - pub quantity: u32, - pub unit_price_cents: u64, -} - -#[derive(Clone, CellValue, PartialEq, Eq)] -pub struct PlaceOrderInput { - pub order_id: OrderId, - pub customer_id: CustomerId, - #[cell(max_items = 64)] - pub lines: Vec, -} - -#[derive(Clone, CellValue, PartialEq, Eq)] -pub enum PlaceOrderOutcome { - Placed, - AlreadyExists, - Empty, -} -``` - -The application stores stable identifiers in source rather than deriving them -from names or registration order: - -```rust,ignore -pub const ORDERS: NamespaceId = NamespaceId::from_bytes([0x01; 16]); -pub const INVENTORY: NamespaceId = NamespaceId::from_bytes([0x02; 16]); -pub const CARTS: NamespaceId = NamespaceId::from_bytes([0x03; 16]); -pub const INVOICES: NamespaceId = NamespaceId::from_bytes([0x04; 16]); -pub const FULFILLMENT: NamespaceId = NamespaceId::from_bytes([0x05; 16]); -pub const RENEWALS: NamespaceId = NamespaceId::from_bytes([0x06; 16]); -pub const CHECKOUTS: NamespaceId = NamespaceId::from_bytes([0x07; 16]); -pub const CUSTOMER_ORDER_INDEX: NamespaceId = NamespaceId::from_bytes([0x08; 16]); -``` - -Changing a constant creates a different namespace and therefore different Cell -IDs. A rename leaves the constant unchanged. - -## Register the complete application - -The application registry includes every target before a node becomes ready. -The builder verifies migrations, bindings, byte limits, effect targets, -workflow definitions, activities, primitive roles, and shard topology. - -```rust,ignore -use crab_cell_app::{CellApplication, CellApplicationBuilder}; - -pub struct Commerce; - -impl CellApplication for Commerce { - const NAME: &'static str = "commerce"; - - fn register(builder: &mut CellApplicationBuilder) -> Result<()> { - builder.entity::()?; - builder.sharded_sql::()?; - builder.kv::()?; - builder.blob::()?; - builder.queue::()?; - builder.cron::()?; - builder.workflow::()?; - builder.read_model::()?; - - builder.activity::()?; - builder.activity::()?; - builder.finish_module::() - } -} -``` - -The resulting generated client exposes only the declared capabilities: - -```rust,ignore -// The generated surface; bodies are elided because the generator writes them. -pub struct CommerceClient; - -impl CommerceClient { - pub fn orders(&self, id: OrderId) -> OrderClient { - todo!() - } - pub fn inventory(&self, sku: &str) -> InventoryClient { - todo!() - } - pub fn shopping_carts(&self) -> KvNamespace { - todo!() - } - pub fn invoice_documents(&self) -> BlobNamespace { - todo!() - } - pub fn fulfillment_jobs(&self) -> QueueNamespace { - todo!() - } - pub fn subscription_renewals(&self) -> CronNamespace { - todo!() - } - pub fn checkout_runs(&self) -> WorkflowNamespace { - todo!() - } - pub fn customer_orders(&self, customer: CustomerId) -> CustomerOrderIndexClient { - todo!() - } -} -``` - -## Store the Order aggregate in a SQL Cell - -Each order is an entity Cell selected by `OrderId`. Its SQL migration is -application-owned and digest-checked: - -```sql -CREATE TABLE orders ( - order_id BLOB PRIMARY KEY CHECK(length(order_id) = 16), - customer_id BLOB NOT NULL CHECK(length(customer_id) = 16), - status INTEGER NOT NULL, - total_cents INTEGER NOT NULL, - created_at_ms INTEGER NOT NULL -) STRICT; - -CREATE TABLE order_lines ( - line_number INTEGER PRIMARY KEY, - sku TEXT NOT NULL CHECK(length(sku) BETWEEN 1 AND 64), - quantity INTEGER NOT NULL CHECK(quantity > 0), - unit_price_cents INTEGER NOT NULL CHECK(unit_price_cents >= 0) -) STRICT; -``` - -The Cell type declares the persistent partition and effect targets: - -```rust,ignore -pub struct Orders; - -impl CellEntity for Orders { - const MODULE: &'static str = "commerce.orders"; - const NAMESPACE: NamespaceId = ORDERS; - const DATABASE_LIMIT_BYTES: u64 = 64 * MIB; - const EFFECT_TARGETS: &'static [NamespaceId] = &[ - CHECKOUTS, - CUSTOMER_ORDER_INDEX, - ]; - - type Key = OrderId; - - fn partition(id: &OrderId) -> EntityKey { - EntityKey::new(&id.0) - } - - fn register(registry: &mut RegistryBuilder) -> Result<()> { - registry.migration(1, include_str!("../migrations/orders/0001.sql"))?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_query::()?; - Ok(()) - } -} -``` - -`PlaceOrder` changes only its Order Cell and records a typed workflow-start -effect in the same transaction: - -```rust,ignore -pub struct PlaceOrder; - -impl Command for PlaceOrder { - const MODULE: &'static str = Orders::MODULE; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - - type Input = PlaceOrderInput; - type Output = PlaceOrderOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> Result> { - if input.lines.is_empty() { - return Ok(CommandResult::Rejected(PlaceOrderOutcome::Empty)); - } - if order_exists(context, input.order_id)? { - return Ok(CommandResult::Rejected( - PlaceOrderOutcome::AlreadyExists, - )); - } - - insert_order(context, &input)?; - insert_lines(context, &input.lines)?; - - context.emit_effect(&CheckoutRuns::start_effect( - input.order_id, - CheckoutState::new(&input), - )?)?; - - Ok(CommandResult::Success(PlaceOrderOutcome::Placed)) - } -} -``` - -The handler does not call the workflow Cell. It writes an effect row beside the -order so a crash can lose neither the order nor its workflow intent. - -The query uses an optional write receipt supplied by the generated client: - -```rust,ignore -pub struct GetOrder; - -impl Query for GetOrder { - const MODULE: &'static str = Orders::MODULE; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - - type Input = (); - type Output = Option; - - fn execute(context: &mut QueryContext<'_>, _: ()) -> Result { - load_order_with_lines(context) - } -} -``` - -## Store shopping carts in KV - -Carts are small scoped records, so they share fixed KV shards rather than -opening one SQLite database per cart. - -```rust,ignore -pub struct ShoppingCarts; - -impl KvModule for ShoppingCarts { - const MODULE: &'static str = "commerce.carts"; - const NAMESPACE: NamespaceId = CARTS; - const SHARDS: u32 = 256; - const ATOMIC_COMMAND_ID: u32 = 1; - const GET_QUERY_ID: u32 = 1; - const LIST_QUERY_ID: u32 = 2; -} -``` - -An HTTP command adds an item with optimistic concurrency: - -```rust,ignore -let carts = commerce.shopping_carts(); -let scope = customer_id.0.to_vec(); - -let updated = carts - .atomic( - request.mutation_identity()?, - KvAtomicRequest { - scope: scope.clone(), - checks: vec![KvCheck { - key: b"version".to_vec(), - condition: KvCondition::Version(expected_version), - }], - mutations: vec![ - KvMutation::Put { - key: format!("line/{sku}").into_bytes(), - value: encode(&line)?, - expires_at_ms: Some(request.now_ms() + CART_LIFETIME_MS), - }, - KvMutation::Put { - key: b"version".to_vec(), - value: next_version.to_be_bytes().to_vec(), - expires_at_ms: Some(request.now_ms() + CART_LIFETIME_MS), - }, - ], - }, - ) - .await?; -``` - -Both mutations and the version check execute in one KV shard transaction. The -fixed 256-shard count is part of the release topology. - -## Reserve inventory in sharded SQL Cells - -Inventory needs an atomic quantity invariant and therefore uses custom SQL. -The SKU hash selects one of 1,024 Cells. - -```rust,ignore -pub struct Inventory; - -impl ShardedSqlCell for Inventory { - const MODULE: &'static str = "commerce.inventory"; - const NAMESPACE: NamespaceId = INVENTORY; - const SHARDS: u32 = 1024; - const EFFECT_TARGETS: &'static [NamespaceId] = &[CHECKOUTS]; - - type Scope = String; - - fn shard(sku: &String) -> Result { - shard_for_scope(Self::NAMESPACE, sku.as_bytes(), Self::SHARDS) - } -} -``` - -The reservation command uses a stable reservation ID derived from the order and -line. A duplicate effect returns the first result: - -```rust,ignore -impl Command for ReserveInventory { - const MODULE: &'static str = Inventory::MODULE; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - - type Input = ReserveInventoryInput; - type Output = ReserveInventoryOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> Result> { - let outcome = reserve_if_available(context, &input)?; - context.emit_effect(&CheckoutRuns::inventory_result_effect( - input.order_id, - input.line_number, - outcome.clone(), - )?)?; - - Ok(match outcome { - ReserveInventoryOutcome::Reserved => CommandResult::Success(outcome), - ReserveInventoryOutcome::Unavailable => CommandResult::Rejected(outcome), - }) - } -} -``` - -Inventory and checkout do not commit atomically. The workflow retains the -pending-line set and compensates already reserved lines if another line fails. - -## Coordinate checkout with Workflow - -The workflow Cell is selected by `OrderId`. Its deterministic transition emits -effects and activities but performs no network I/O. - -```rust,ignore -pub struct CheckoutRuns; - -impl WorkflowModule for CheckoutRuns { - const MODULE: &'static str = "commerce.checkout"; - const NAMESPACE: NamespaceId = CHECKOUTS; - const SHARDS: u32 = 256; - const START_COMMAND_ID: u32 = 1; - const SIGNAL_COMMAND_ID: u32 = 2; - const GET_QUERY_ID: u32 = 1; -} - -pub struct CheckoutV1; - -impl WorkflowDefinition for CheckoutV1 { - fn digest(&self) -> Digest { - CHECKOUT_V1_DIGEST - } - - fn effect_targets(&self) -> &'static [NamespaceId] { - &[INVENTORY, ORDERS, FULFILLMENT, CUSTOMER_ORDER_INDEX] - } - - fn transition( - &self, - state: &[u8], - event: &[u8], - context: WorkflowContext, - ) -> Result { - let mut state = CheckoutState::decode(state)?; - match CheckoutEvent::decode(event)? { - CheckoutEvent::Started => { - for line in &state.lines { - context.effect(Inventory::reserve_command( - line.sku.clone(), - state.order_id, - line, - )?)?; - } - state.phase = CheckoutPhase::Reserving; - Ok(context.continue_with(state)) - } - CheckoutEvent::InventoryResult { line, outcome } => { - state.record_inventory(line, outcome)?; - if state.has_failure() { - for reservation in state.successful_reservations() { - context.effect(Inventory::release_command(reservation)?)?; - } - context.effect(Orders::set_status_command( - state.order_id, - OrderStatus::InventoryFailed, - )?)?; - return Ok(context.complete(state.failed())); - } - if state.inventory_complete() { - context.activity::(state.payment_input())?; - state.phase = CheckoutPhase::Charging; - } - Ok(context.continue_with(state)) - } - CheckoutEvent::PaymentCompleted(payment) => { - context.effect(Orders::set_status_command( - state.order_id, - OrderStatus::Paid, - )?)?; - context.effect(FulfillmentJobs::send_command( - state.fulfillment_job(payment)?, - )?)?; - context.effect(CustomerOrderIndex::upsert_command( - state.customer_projection(OrderStatus::Paid), - )?)?; - Ok(context.complete(state.paid())) - } - CheckoutEvent::PaymentFailed(reason) => { - for reservation in state.successful_reservations() { - context.effect(Inventory::release_command(reservation)?)?; - } - context.effect(Orders::set_status_command( - state.order_id, - OrderStatus::PaymentFailed, - )?)?; - Ok(context.complete(state.payment_failed(reason))) - } - } - } -} -``` - -The workflow definition digest is pinned when a run starts. A rolling release -retains `CheckoutV1` while any stored run still names that digest. - -## Charge through an external activity - -The activity runs only after its claim root publishes. The payment provider -receives the stable activity attempt identity as its idempotency key. - -```rust,ignore -pub struct ChargePayment; - -impl Activity for ChargePayment { - const TYPE: &'static str = "commerce.charge-payment"; - type Input = ChargePaymentInput; - type Output = ChargePaymentOutput; - - async fn execute( - context: ActivityContext, - input: Self::Input, - ) -> ActivityExecution { - let result = payment_provider() - .charge(ChargeRequest { - customer: input.customer, - amount_cents: input.amount_cents, - idempotency_key: context.idempotency_key(), - }) - .await; - - match result { - Ok(charge) => ActivityExecution::complete( - ChargePaymentOutput::Paid { charge_id: charge.id }, - ), - Err(error) if error.retryable() => { - ActivityExecution::retry(error.retry_after()) - } - Err(error) => ActivityExecution::fail(error.public_code()), - } - } -} -``` - -The supervisor publishes the completion event back into the workflow. Dropping -the worker future does not erase the durable claim or make an external charge -exactly once; provider idempotency closes that boundary. - -## Project customer queries into a read-model Cell - -Customer history is not queried by scanning Order Cells. A projection command -updates the customer's read-model shard after checkout changes state. - -```rust,ignore -pub struct CustomerOrderIndex; - -impl ReadModelCell for CustomerOrderIndex { - const MODULE: &'static str = "commerce.customer-order-index"; - const NAMESPACE: NamespaceId = CUSTOMER_ORDER_INDEX; - const SHARDS: u32 = 256; - type Scope = CustomerId; -} - -impl Command for UpsertCustomerOrder { - const MODULE: &'static str = CustomerOrderIndex::MODULE; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - type Input = CustomerOrderProjection; - type Output = (); - - fn execute( - context: &mut CommandContext<'_, '_>, - projection: Self::Input, - ) -> Result> { - upsert_projection(context, projection)?; - Ok(CommandResult::Success(())) - } -} -``` - -Projection identity derives from the source order and source commit sequence, -so repeated effect delivery is harmless. The API documents that this read model -is asynchronous and cannot satisfy an Order Cell receipt. - -## Deliver fulfillment through Queue - -Fulfillment is at least once. Producers hash by order ID; workers claim one -explicit shard at a time. - -```rust,ignore -pub struct FulfillmentJobs; - -impl QueueModule for FulfillmentJobs { - const MODULE: &'static str = "commerce.fulfillment"; - const NAMESPACE: NamespaceId = FULFILLMENT; - const SHARDS: u32 = 128; - const SEND_COMMAND_ID: u32 = 1; - const CLAIM_COMMAND_ID: u32 = 2; - const LEASE_COMMAND_ID: u32 = 3; - const VALIDATE_QUERY_ID: u32 = 1; - const CONTROL_COMMAND_ID: u32 = 4; - const INFO_QUERY_ID: u32 = 2; -} -``` - -The worker validates the published lease before external work, writes the -invoice, sends mail with an idempotency key, and then acknowledges the exact -lease token: - -```rust,ignore -pub async fn run_fulfillment_shard( - commerce: CommerceClient, - shard: u32, - cancellation: CancellationToken, -) -> Result<()> { - while !cancellation.is_cancelled() { - let claim = commerce - .fulfillment_jobs() - .claim( - Request::new_random()?, - shard, - QueueClaimRequest { - limit: 16, - lease_ms: 30_000, - }, - ) - .await?; - - let valid = commerce - .fulfillment_jobs() - .validate_claim(shard, claim.output.clone(), Some(claim.receipt)) - .await?; - if !valid.output { - continue; - } - - for message in claim.output { - match fulfill(&commerce, &message).await { - Ok(()) => { - commerce - .fulfillment_jobs() - .ack( - Request::new_random()?, - shard, - message.message_id, - message.token, - ) - .await?; - } - Err(error) if error.retryable() => { - commerce - .fulfillment_jobs() - .retry( - Request::new_random()?, - shard, - message.message_id, - message.token, - 10_000, - ) - .await?; - } - Err(error) => return Err(error), - } - } - } - Ok(()) -} -``` - -The application supplies a stable request identity for every ack or retry. A -worker crash after external work but before ack repeats `fulfill`, so each -external destination must deduplicate by the job or order identity. - -## Store invoices through Blob - -Invoice metadata and its manifest live in one Blob shard transaction domain; -invoice bytes live in the configured object store. The worker uses multipart -publication even though this example produces a small document, keeping the -same bounded path for larger invoices. - -```rust,ignore -pub struct InvoiceDocuments; - -impl BlobModule for InvoiceDocuments { - const MODULE: &'static str = "commerce.invoices"; - const NAMESPACE: NamespaceId = INVOICES; - const SHARDS: u32 = 128; - const MUTATE_COMMAND_ID: u32 = 1; - const QUERY_QUERY_ID: u32 = 1; -} - -async fn store_invoice( - commerce: &CommerceClient, - order: OrderId, - bytes: Vec, -) -> Result { - let blobs = commerce.invoice_documents(); - let key = format!("orders/{}/invoice.pdf", encode_hex(&order.0)).into_bytes(); - let upload = blobs - .begin(Request::derived(order.0, b"invoice-begin"), key.clone()) - .await?; - - for (index, part) in bytes.chunks(256 * KIB).enumerate() { - blobs - .put_part( - Request::derived(order.0, &(index as u32).to_be_bytes()), - upload.output.upload_id, - index as u32 + 1, - part.to_vec(), - ) - .await?; - } - - let completed = blobs - .complete( - Request::derived(order.0, b"invoice-complete"), - upload.output.upload_id, - BlobCondition::CreateOnly, - ) - .await?; - Ok(completed.output) -} -``` - -Part digests are verified on write and range read. `complete` publishes the -manifest atomically with request outcome and Blob state. It does not expose a -separate uncommitted body path. - -## Trigger subscription renewals through Cron - -Cron stores the schedule and advances one occurrence in the same transaction -that records its workflow-start effect. - -```rust,ignore -pub struct SubscriptionRenewals; - -impl CronModule for SubscriptionRenewals { - const MODULE: &'static str = "commerce.renewals"; - const NAMESPACE: NamespaceId = RENEWALS; - const SHARDS: u32 = 64; - const MUTATE_COMMAND_ID: u32 = 1; - const QUERY_QUERY_ID: u32 = 1; - type Target = CheckoutRuns; -} - -let scheduled = commerce - .subscription_renewals() - .upsert( - Request::new(request_id)?, - SubscriptionId(subscription_id), - CronSchedule::every(Duration::from_days(30)) - .starting_at(first_renewal_ms), - RenewalInput { - customer_id, - subscription_id, - }, - ) - .await?; -``` - -The registry verifies the Cron target command, codec, namespace, and input -limit before readiness. An owner crash cannot lose an occurrence after its -schedule advance publishes, and destination inbox deduplication prevents the -same occurrence from starting the workflow twice. - -## Compose and start a Cell node - -The service binary constructs one node. Application code never assembles -`CellRuntime`, `CellAuthority`, or `CellReplica` directly. - -```rust,ignore -#[tokio::main] -async fn main() -> Result<()> { - let config = Config::load()?; - let provider = build_storage_provider(&config.storage).await?; - let identity = load_application_identity(&provider, &config.root).await?; - let registry = Commerce::compile(BuildDescriptor { - source_revision: build_revision().to_owned(), - cargo_lock_digest: cargo_lock_digest(), - })?; - - let node = CellNode::builder() - .identity(identity) - .storage(provider, config.root) - .data_directory(config.data_directory) - .registry(registry) - .resources(config.resources) - .cluster( - config.peer_transport()?, - config.node_signer()?, - config.private_endpoint, - ) - .durability(Durability::FollowersOrObjectStore { followers: 2 }) - .telemetry(config.telemetry()?) - .build() - .await?; - - node.start().await?; - let commerce = CommerceClient::new(node.application::()?); - let workers = FulfillmentWorkers::start(commerce.clone(), node.resources())?; - let http = serve_http(config.public_listener, commerce, node.readiness()).await?; - - shutdown_signal().await; - http.close_admission(); - http.drain().await?; - workers.stop().await?; - node.drain(config.shutdown_deadline()).await?; - node.shutdown().await -} -``` - -Provider construction, credentials, public listeners, authentication, and -process signals remain service concerns. The node owns Cell admission, -activation, peer routing, followers, publication, recovery, scheduling, -placement, eviction, backup integration, and ordered shutdown. - -## Adapt an authenticated HTTP route - -The external route authorizes the product action before calling the generated -application capability. It maps durable outcomes explicitly. - -```rust,ignore -pub async fn place_order_route( - State(state): State, - Authenticated(principal): AuthenticatedPrincipal, - Json(input): Json, -) -> HttpResult { - state - .authorizer - .require(&principal, Action::PlaceOrder, input.customer_id) - .await?; - - let request = Request::new(input.request_id)? - .issued_at(input.issued_at_ms) - .expires_at(input.expires_at_ms) - .build()?; - let order_id = OrderId(input.order_id); - - match state - .commerce - .orders(order_id) - .create(order_id, request, input.into_domain()) - .await - { - Ok(committed) => Ok(created(committed.output, committed.receipt)), - Err(ApplicationInvocationError::Rejected(rejection)) => { - Ok(conflict(rejection.output, rejection.receipt)) - } - Err(ApplicationInvocationError::Pending(pending)) => { - Ok(accepted_for_resolution(pending)) - } - Err(ApplicationInvocationError::Unavailable(error)) => Err(error.into()), - } -} -``` - -The HTTP request ID is stable across client retries. A `Pending` response means -the mutation may have started and must be resolved; the adapter must not create -a new request ID and submit the business operation again. - -## Resolve an ambiguous mutation - -The generated pending token contains the target, incarnation, request identity, -operation digest, and result bound. It contains no storage credential or raw -SQL input. - -```rust,ignore -pub async fn resolve_order( - commerce: &CommerceClient, - pending: PendingApplicationMutation, -) -> Result> { - loop { - match commerce.resolve(&pending).await? { - Resolution::Committed(result) => return Ok(result), - Resolution::Absent => return Ok(Resolution::Absent), - Resolution::Expired => return Ok(Resolution::Expired), - Resolution::Unknown => tokio::time::sleep(RETRY_DELAY).await, - } - } -} -``` - -`Absent` means the authoritative current incarnation has no matching ledger -row. `Unknown` means the framework cannot yet prove absence or a committed -outcome. Incarnation change fails closed rather than searching stale local -state. - -## Exercise the public application surface - -An ordinary application test uses generated APIs and observes every primitive: - -```rust,ignore -#[tokio::test] -async fn customer_checkout_reaches_a_queryable_invoice() -> Result<()> { - let cluster = TestCluster::::new(3).await?; - let commerce = cluster.client(); - let customer = CustomerId([1; 16]); - let order = OrderId([2; 16]); - - commerce - .shopping_carts() - .atomic(cart_request(), add_line(customer, "sku-1", 2)) - .await?; - - let placed = commerce - .orders(order) - .create(order, place_request(), order_input(customer, order)) - .await?; - - let observed = commerce - .orders(order) - .get_order(ReadConsistency::After(placed.receipt)) - .await?; - assert_eq!(observed.output.status, OrderStatus::Pending); - - cluster.activities().complete_next::(paid()).await?; - cluster.workers().run_fulfillment_once().await?; - - let invoice = commerce - .invoice_documents() - .read_range(invoice_key(order), 0..4096, None) - .await?; - assert!(invoice.output.starts_with(b"%PDF")); - - let history = commerce - .customer_orders(customer) - .list(CurrentRead, first_page()) - .await?; - assert_eq!(history.output.items[0].order_id, order); - Ok(()) -} -``` - -The test may poll workflow and projection state because those paths are -asynchronous. It uses the Order receipt only for an Order query, not for the -customer read model or Blob namespace. - -## Prove owner loss and retry safety - -The application qualification test kills the owner after the command is -accepted, loses its local directory, and verifies exact recovery through public -APIs: - -```rust,ignore -#[tokio::test] -async fn acknowledged_order_survives_owner_and_local_disk_loss() -> Result<()> { - let cluster = TestCluster::::new(3).await?; - let commerce = cluster.client(); - let order = OrderId([7; 16]); - let request = place_request(); - - let committed = commerce - .orders(order) - .create(order, request.clone(), order_input(customer(), order)) - .await?; - - cluster - .kill_owner_and_remove_local_state(committed.receipt.cell) - .await?; - - let restored = commerce - .orders(order) - .get_order(ReadConsistency::After(committed.receipt)) - .await?; - assert_eq!(restored.output.id, order); - - let replay = commerce - .orders(order) - .create(order, request, order_input(customer(), order)) - .await?; - assert_eq!(replay.receipt.commit_sequence, committed.receipt.commit_sequence); - - cluster.assert_one_checkout_start(order).await?; - Ok(()) -} -``` - -Additional qualification injects: - -- Lost control-CAS responses after publication -- Object-store timeout with follower proof available -- Follower loss with object-store proof available -- Duplicate inventory and projection effects -- Payment completion after activity lease expiry -- Fulfillment worker death after invoice publication but before queue ack -- Blob part corruption and range-read checksum failure -- Cron owner death between occurrence publication and delivery -- Workflow definition retention across rolling deployment -- Disk-full activation, capture, compaction, and hydration -- Pressure eviction followed by exact-root reacquisition - -Every acknowledged order must remain queryable. Every duplicate request must -return the same durable decision or an explicit unresolved state. Corruption, -fencing, and incompatible releases fail closed. - -## Understand what each primitive contributes - -| Primitive | Commerce use | Boundary demonstrated | -| --- | --- | --- | -| Custom SQL | Order aggregate, inventory invariant, customer read model | Serializable state within one Cell | -| KV | Shopping carts | Atomic checks and mutations within one scope-derived shard | -| Blob | Invoice documents | Object-store parts and manifest references in one transactional shard | -| Queue | Fulfillment jobs | Published leases and at-least-once worker delivery | -| Cron | Subscription renewals | Failover-safe occurrence effect and schedule advance | -| Workflow | Checkout | Durable deterministic saga, timers, effects, and activities | -| Effects | Order, inventory, projection, and queue transitions | Transactional outbox and idempotent destination inbox | -| Activities | Payment and mail | Retryable external work after published claim | -| Receipts | Order create followed by Order query | Same-Cell read watermark | -| LTX and exact roots | All stateful primitives | Verified publication, source-loss recovery, and takeover | - -The framework is successful when this application contains no object-store -path, owner election, peer message, LTX segment, control CAS, SQLite file, -runtime task, or recovery branch in its domain modules. Those mechanics remain -observable through outcomes and metrics but are owned by the node facade and -runtime. diff --git a/crates/crab-cell-runtime/docs/application-framework.md b/crates/crab-cell-runtime/docs/application-framework.md deleted file mode 100644 index 0a46a475b..000000000 --- a/crates/crab-cell-runtime/docs/application-framework.md +++ /dev/null @@ -1,823 +0,0 @@ -# Build large applications on the Cell runtime - -Crab should expose an application framework above `crab-cell-runtime` so an -application owner defines durable Cell types, typed operations, partitioning, -and cross-Cell workflows without constructing catalogs, authority records, -LTX replicas, worker pools, or peer routes. The framework keeps the existing -single-writer and exact-root contracts; it does not turn Cells into a globally -distributed relational database. - -| Document intent | Value | -| --- | --- | -| Content type | Target design with implemented boundary slice | -| Audience | Application framework, runtime, and product contributors | -| Goal | Define the application-owner programming model and the platform API needed to host it | -| Status | Boundary slice implemented: `crab-cell-app` supplies deterministic author compilation and a handwritten all-primitive reference application; `crab-cell-host` supplies the provider-neutral lifecycle shell and `crab-http-server` uses it for serving and offline maintenance. Code generation, full operator ownership, and protected qualification remain open | - -[Back to the Cell runtime index](README.md) - -[`crates/crab-cell-app/examples/authoring.rs`](../../crab-cell-app/examples/authoring.rs) -is the minimal compile-checked version of the flow below: one module with its -migration, one namespace, one cell type, and a finished `CompiledApplication`. -The prose snippets stay illustrative; the example is what CI compiles. - -The [complete Commerce example](application-framework-example.md) remains the -target qualification shape for custom SQL Cells, KV, Blob, Queue, Cron, -Workflow, effects, activities, generated clients, node composition, HTTP -adaptation, and owner-loss qualification. It is not a production evidence -claim until the full-primitive workload and protected provider gates pass. - -## Design for application owners - -An application owner should make five durable decisions: - -1. Which state must change atomically? -2. Which stable key selects that state? -3. Which commands and queries may access it? -4. Which operations cross a Cell boundary? -5. Which external work requires an idempotent activity? - -The framework turns those decisions into compiled module descriptors, catalog -entries, typed clients, and release evidence. It owns the mechanics after the -application declares them. - -```text -application source - -> Cell type declarations - -> commands, queries, workflows and activities - -> compiled registry and typed client - -> CellNode application host - -> crab-cell-runtime - -> crab-ltx - -> crab-storage -``` - -Application code does not open SQLite files, select owners, publish LTX roots, -interpret peer messages, or retry ambiguous SQL mutations. Product adapters -continue to own authentication, authorization, HTTP or RPC policy, and mapping -external identities to application identities. - -## Use four topology patterns - -Large applications compose four Cell patterns. The application declares the -pattern for each namespace rather than treating every record as a separate -database. - -| Pattern | Partition | Use | Main tradeoff | -| --- | --- | --- | --- | -| Entity Cell | Stable aggregate ID | Orders, repositories, projects, accounts, game sessions | Strong local invariants; one hot entity remains one writer | -| Shard Cell | Stable hash modulo fixed shard count | KV, queues, counters, rate limits, small records | Shares lifecycle overhead; shard contention must be sized | -| Workflow Cell | Workflow or business-process ID | Checkout, deployment, merge, provisioning | Durable orchestration; cross-Cell work is not one transaction | -| Read-model Cell | Query-domain shard | Search, dashboards, feeds, secondary indexes | Fast queries; updated asynchronously from source Cells | - -State that must commit together belongs in one Cell. State that can be retried, -compensated, or projected may cross Cells through durable effects, workflows, -and activities. - -The framework cannot transparently split a hot Cell because arbitrary SQLite -state has application-defined invariants. Repartitioning is a declared schema -and data migration with explicit source and destination ownership. Increasing a -namespace's shard count is therefore a release operation, not a live tuning -knob. - -The implemented `CellType::with_entity_partitions` mode addresses distinct -entity Cells by a stable 33-byte partition digest under a namespace declared -with one shard. Generated accessors and `ApplicationHandle::target_for_scope` -derive that partition from the entity key. `CellType::entity_partition` exposes -the same derivation for provisioning, and `ApplicationHandle` checks its encoding. -This provides an application-validated target for an application's own split protocol; it -does not repartition or migrate data automatically. - -## Separate the author and operator APIs - -The public framework has two capability levels. - -### Application author API - -Application authors use: - -- `CellApplication` to assemble modules into one release -- `CellType` to declare namespace, partitioning, schema, and limits -- `Command` and `Query` for deterministic in-Cell behavior -- typed primitive capabilities for KV, Queue, Blob, Cron, and Workflow -- `EffectContext` for typed cross-Cell commands -- `Activity` for external asynchronous work -- generated application clients for targeting and invocation - -These APIs never expose `CellAuthority`, `CellReplica`, `PreparedRoot`, object -store credentials, database paths, or peer signing material. - -### Platform operator API - -Platform operators use: - -- `CellNodeBuilder` to compose storage, local paths, runtime resources, and the registry -- one cluster transport for authenticated peer requests -- node identity, lease, follower, and placement configuration -- release activation and migration controls -- backup, retention, telemetry, and graceful shutdown controls - -The operator API owns the current `CellCatalog`, `CellAuthority`, `CellRuntime`, -`CellReplica`, scheduler, supervisors, node durability, and routing composition. -It returns an application-bound client instead of exposing those parts -individually. - -Binding an `ApplicationHandle` is fallible: its author type must name the -compiled application, and its `CellClient` must carry the same release digest -as the compiled registry. The host checks this before returning the handle, -so a client assembled with a different release cannot dispatch through an -application descriptor that validated a different set of operations. - -## Declare one application - -The initial framework remains statically linked Rust. Attributes reduce -descriptor boilerplate but do not introduce uploaded code, dynamic libraries, -JavaScript, WebAssembly, or subprocess handlers. - -The following API is illustrative: - -```rust,ignore -use crab_cell_app::{CellApplication, CellApplicationBuilder}; - -pub struct Commerce; - -impl CellApplication for Commerce { - const NAME: &'static str = "commerce"; - - fn register(builder: &mut CellApplicationBuilder) -> crab_cell_app::Result<()> { - builder.entity::()?; - builder.sharded::()?; - builder.queue::()?; - builder.workflow::()?; - builder.read_model::()?; - Ok(()) - } -} -``` - -The implemented `crab-cell-app::ApplicationBuilder::finish` produces the -existing canonical runtime registry plus an application topology descriptor. -Registration order does not change either digest, and every namespace in the -compiled registry must have exactly one matching `CellType` declaration; -undeclared runtime namespaces fail closed before the descriptor is emitted. - -Every namespace declaration contains: - -- Stable 16-byte namespace ID -- Human-readable name -- Topology pattern -- Partition codec and version -- Fixed shard count when sharded -- Initial schema and ordered migration digests -- Commands, queries, workflows, activities, and effect targets -- Input, output, state, database, and capture limits -- Provisioning policy -- Retained code and codec versions required for rolling rollout - -Names are diagnostic. IDs, partition bytes, operation IDs, codec versions, and -migration digests are persistent contracts. - -## Declare an entity Cell - -An entity Cell co-locates one aggregate's transactional state. The application -declares its stable partition encoding and installs its schema through checked -migrations. - -```rust,ignore -use crab_cell_app::{CellEntity, EntityKey}; - -pub struct Orders; - -impl CellEntity for Orders { - const MODULE: &'static str = "orders"; - const NAMESPACE: NamespaceId = NamespaceId::from_bytes(*b"commerce-orders1"); - const DATABASE_LIMIT_BYTES: u64 = 64 * 1024 * 1024; - - type Key = OrderId; - - fn partition(key: &Self::Key) -> EntityKey { - EntityKey::new(key.as_bytes()) - } - - fn register(registry: &mut RegistryBuilder) -> Result<()> { - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_query::()?; - Ok(()) - } -} -``` - -Partition encoders must be canonical, bounded, and covered by byte fixtures. -Changing an entity key's encoding requires a new namespace or an explicit -repartitioning migration. The implemented entity mode stores a domain-separated -digest of the scope as its partition bytes. The application retains the mapping -from logical entity ID to target; the catalog retains the target bytes needed -for routing and recovery. - -## Write deterministic commands - -The framework reuses the current typed `Command` contract. Attributes may -generate descriptor entries, but operation IDs and codec versions remain -explicit in source so renaming or reordering code cannot change stored work. - -```rust,ignore -#[cell_command(id = 1, codec = 1, input_limit = "64KiB", output_limit = "64KiB")] -pub struct PlaceOrder; - -impl Command for PlaceOrder { - const MODULE: &'static str = Orders::MODULE; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - - type Input = PlaceOrderInput; - type Output = PlaceOrderOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> Result> { - let existing = load_order(context, input.order_id)?; - if existing.is_some() { - return Ok(CommandResult::Rejected( - PlaceOrderOutcome::AlreadyExists, - )); - } - - insert_order(context, &input)?; - context.emit_effect(&Inventory::reserve_effect(input.inventory_request())?)?; - - Ok(CommandResult::Success(PlaceOrderOutcome::Placed)) - } -} -``` - -A command may: - -- Read and mutate its current Cell through bounded authorized SQL -- Use the runtime's sampled logical time -- Allocate deterministic transition-local identities -- Insert typed cross-Cell effects -- Start or signal a workflow in the same Cell -- Return a durable success or durable business rejection - -A command may not perform network I/O, call object storage, spawn work, read the -system clock directly, commit or roll back the transaction, change SQLite -configuration, or access another Cell synchronously. - -The handler's application savepoint and the runtime request ledger commit in -one SQLite transaction. A business rejection rolls back application writes but -still records and publishes its typed outcome. - -## Write receipted queries - -Queries are typed, bounded, and read-only. Generated clients expose consistency -as an input instead of making callers manually reconstruct receipt checks. - -```rust,ignore -let placed = commerce - .orders(order_id) - .place_order(request, input) - .await?; - -let observed = commerce - .orders(order_id) - .get_order(ReadConsistency::After(placed.receipt), order_id) - .await?; -``` - -The initial consistency choices are: - -```rust,ignore -pub enum ReadConsistency { - CurrentOwner, - After(Receipt), -} -``` - -`CurrentOwner` uses the current owner and its actor-ordered query path. -`After(receipt)` additionally requires the same Cell and incarnation and a -commit sequence at or beyond the receipt. The framework does not expose an -unfenced local-file read or a global timestamp spanning Cells. -The implemented client API uses `client::ReadPolicy::{CurrentOwner, Replica}` -on a cloned capability and an optional minimum `Receipt` on each typed query. -`CellClient::with_read_policy` and `ApplicationHandle::with_read_policy` select -the policy; generated clients retain it when deriving scoped accessors. -Commands, resolution, state streams, and Queue/Effects/Workflow activity lease -validation still use the owner. Replica queries -report their actual snapshot receipt, reject a newer minimum with -`ReplicaBehind`, and never fall back when readers are unavailable. - -The public host installs `ReadReplicaManager` for admitted-view refresh and -drain, and `CellNode::install_read_replica_recruitment` for owner recruitment -across the application's compiled namespaces. Products supply scope, signed -membership, and an authenticated activation client. Both the reference app -and repository service use this host loop and the shared peer activation/status -dispatch. Reader selection remains advisory; each recipient rechecks owner, -policy, membership, and resource admission before opening a snapshot. - -Host code supplies `CellClient::with_read_replicas` with the shared runtime -`ReplicaReadRouter` built from its existing instrumented `CellAuthority` and -live directory, an authenticated peer client, and an optional local -admitted-view resolver. The same router serves explicit HTTP issue-detail -reads. It loads authority and the S3 desired count concurrently, rejects a -policy from a different incarnation, and consults signed live membership. It -prefers lower ingress-observed in-flight load, and bounds selection and all -attempts by one five-second deadline. Peer attempts carry the remaining budget. -Advisory reader discovery shares a bounded one-second membership snapshot -across directory clones and Cells, with one concurrent refresh. Selection -filters expired advertisements each time; a failed expired refresh returns an -error. Successful local enrollment and withdrawal invalidate this discovery -snapshot. -The same signed snapshot supplies the owner's immutable boot identity for -physical-node exclusion; an absent owner is inspected directly, including its -retirement tombstone. Authority, peer authentication, and offline maintenance -scans remain fresh. -The author handle does not expose storage, local files, or routing internals. -Fresh authority checks remain mandatory before a replica releases a result. -Blob queries hydrate content-addressed parts with digest and length checks; -missing or reclaimed parts fail instead of returning unverified bytes. -The object durability profile supplies the all-node-loss contract; -[Plan 036](../../../advisor-plans/036-cell-read-replicas-and-fenced-promotion.md) -tracks remaining qualification work. Replica views now fault authenticated -pages from their exact root, with no full local database restore. Each view -provisionally reserves 12 MiB and four descriptors, including a conservative -charge for the shared page cache; refresh retains both views until old queries -finish. Sparse I/O uses the query deadline and preserves its source error. - -## Generate an application client - -The generated client binds tenant, application, registry, and routing once. -Each namespace accessor accepts only its declared key type. - -The current Rust `crab_cell_app::cell_client!` binding generates namespace -accessors and typed command, prepare, query, and resolution methods from -explicit stable IDs. Construction checks the compiled registry and operation -traits; each accessor accepts a declared `CellKey` whose canonical bytes feed -the compiled `CellType` entity or fixed-shard contract. Namespace-level primitive -helpers require fixed shards; explicitly targeted SQL and effect handles accept -entity Cells. The reference application's independent descriptor-byte test, compile-fail -examples, and three-node `CellNode` suite cover this initial binding. The -rollout test adds a generated query and retains predecessor code while old and -new clients overlap. It then publishes a code-only migration, rejects stale -capabilities and predecessor clients, and recovers exact receipts on a fresh -host. The RustFS variant uses the same path with real object storage. This is -a correctness gate with one process hosting the nodes; additive schema changes -and continuous traffic during rolling container replacement remain unqualified. -A general schema-driven generator and generated transport adapters remain open. - -```rust,ignore -let commerce = CommerceClient::new(node.application::()?); - -let order = commerce.orders(order_id); -let result = order - .place_order( - Request::new(request_id).expires_in(Duration::from_minutes(5))?, - PlaceOrderInput { - order_id, - customer_id, - lines, - }, - ) - .await; - -match result { - Ok(committed) => return_order(committed.output, committed.receipt), - Err(ApplicationInvocationError::Rejected(rejection)) => { - return_conflict(rejection.output, rejection.receipt) - } - Err(ApplicationInvocationError::Pending(pending)) => { - enqueue_resolution(pending) - } - Err(ApplicationInvocationError::Unavailable(error)) => return_unavailable(error), -} -``` - -The client: - -1. Canonically encodes the entity or shard key. -2. Constructs the deterministic `CellTarget`. -3. Validates the operation against the compiled registry. -4. Routes to a resident local actor or one authenticated peer. -5. Activates from exact authority when no usable owner exists. -6. Preserves the caller's request identity across transport retries. -7. Converts an ambiguous started mutation into `Pending`, never a blind replay. -8. Resolves a pending outcome through the durable request ledger. - -Generated clients are internal application capabilities. They do not generate -a public HTTP authorization model. A product may add generated Axum, tonic, or -other transport adapters later, but those adapters must require an explicit -authorization function before constructing an application invocation. - -## Make provisioning explicit - -Queries never create state. A mutating API chooses one of two declared -provisioning modes: - -| Mode | Behavior | Intended use | -| --- | --- | --- | -| Explicit | An administrator or trusted workflow provisions the target before use | Repositories, projects, regulated entities | -| Create command | One named command may provision an absent target and execute after the empty root publishes | Orders, sessions, user-created entities | - -The generated API keeps the distinction visible: - -```rust,ignore -let orders = commerce.orders(); -let order = orders - .create(order_id, request, PlaceOrderInput { /* ... */ }) - .await?; - -let existing = orders.open(order_id).await?; -``` - -Provisioning publishes the catalog entry before creating control, reserves node -resources before claiming ownership, installs runtime and application schemas, -publishes the initial root, and only then executes later mutations. Concurrent -create attempts adopt only the exact same catalog and control result. - -## Compose built-in primitives - -Applications should use primitive capabilities when their contract fits rather -than recreate queue, lease, expiry, or workflow state machines. - -```rust,ignore -pub struct CommercePrimitives; - -impl CellModule for CommercePrimitives { - fn register(self, registry: &mut RegistryBuilder) -> Result<()> { - register_kv::(registry)?; - register_queue::(registry)?; - register_blob::(registry)?; - register_cron::(registry)?; - register_workflow::(registry)?; - Ok(()) - } -} -``` - -The application client exposes pre-bound typed capabilities: - -```rust,ignore -let updated = commerce - .shopping_carts() - .atomic(request, cart_mutation) - .await?; - -let sent = commerce - .fulfillment_jobs() - .send(request, FulfillOrder { order_id }) - .await?; - -let claim = commerce - .fulfillment_jobs() - .claim(claim_request, shard, 16, Duration::from_secs(30)) - .await?; -``` - -Primitive registration contributes its schema, maintenance work, operations, -and effect targets to the same application descriptor. It does not start a -second runtime or durability path. - -## Coordinate across Cells - -Cross-Cell effects use a transactional outbox and destination inbox: - -```mermaid -sequenceDiagram - participant O as Order Cell - participant E as Effect supervisor - participant I as Inventory Cell - - O->>O: Commit order + reserve effect - O-->>E: Published source receipt - E->>I: Typed Reserve command + effect ID - I->>I: Deduplicate inbox + commit reservation - I-->>E: Published destination receipt - E->>O: Acknowledge source effect -``` - -This supplies durable at-least-once delivery and idempotent destination -execution. It does not supply an atomic transaction across Order and Inventory. -`EffectSource::status(effect_id, minimum_receipt)` reads the source state, -attempt, lease deadline, expiry, and recorded result through the typed host. -It reports whether a lease token exists without returning the token. A source -effect can be removed after its delivery horizon, so callers must treat an -absent status as unknown rather than proof that delivery never happened. - -Application owners choose one of three outcomes: - -- Accept asynchronous convergence for projections and notifications. -- Use a workflow to wait, retry, compensate, and expose business progress. -- Co-locate the state in one Cell when the invariant truly requires atomicity. - -The registry rejects undeclared destination namespaces and cross-tenant effect -targets before writes. - -## Run external work as activities - -Activities are the only application extension that may call external systems. -The workflow or command first publishes an activity intent. A supervisor then -claims the exact lease, validates its published receipt, invokes the registered -activity, and publishes completion or retry. - -```rust,ignore -#[cell_activity(name = "charge-payment", input_limit = "64KiB")] -impl Activity for ChargePayment { - async fn execute( - context: ActivityContext, - input: ChargePaymentInput, - ) -> ActivityExecution { - payment_provider - .charge(input, context.idempotency_key()) - .await - .into_activity_result() - } -} -``` - -The external destination must honor the supplied idempotency key when duplicate -execution is unacceptable. A Cell transaction cannot roll back an external -side effect, and activity lease expiry can cause another attempt. - -## Host applications through one node facade - -`CellNode` is the missing public composition boundary. It owns the current -runtime facilities and exposes application clients plus lifecycle operations. - -```rust,ignore -let node = CellNode::builder() - .identity(application_identity) - .storage(store, root_prefix) - .data_directory(data_directory) - .registry(Commerce::compile(build_descriptor)?) - .resources(NodeResources { - memory_bytes, - local_disk_bytes, - file_descriptors, - sql_workers, - primitive_jobs, - }) - .cluster(cluster_transport, node_signer, node_endpoint) - .durability(Durability::FollowersOrObjectStore { followers: 2 }) - .telemetry(telemetry) - .build() - .await?; - -node.start().await?; -let commerce = CommerceClient::new(node.application::()?); -``` - -The builder validates all process-wide facilities before readiness. It must -not provide partially configured modes that silently fall back to weaker -durability or unfenced local execution. - -`CellNode` owns these operations: - -```rust,ignore -impl CellNode { - pub fn application(&self) -> Result>; - pub async fn provision( - &self, - key: &C::Key, - ) -> Result; - pub async fn prepare_release(&self, release: ReleaseArtifact) -> Result; - pub async fn activate_release(&self, plan: ReleasePlan) -> Result<()>; - pub async fn status(&self) -> NodeStatus; - pub async fn drain(&self, deadline: Instant) -> Result<()>; - pub async fn shutdown(self) -> Result<()>; -} -``` - -Application code cannot obtain the internal runtime handle from -`ApplicationHandle`. This prevents a generated client from bypassing namespace, -schema, codec, or authorization boundaries with an arbitrary closure. - -## Keep deployment artifacts canonical - -One application build produces: - -| Artifact | Purpose | -| --- | --- | -| Release descriptor | Canonical modules, operations, codecs, schemas, workflows, activities, namespaces, and limits | -| Topology descriptor | Cell patterns, partition codecs, shard counts, and provisioning policies | -| Migration inventory | Ordered SQL and content digests | -| Compatibility inventory | Retained code, codec, workflow, and activity versions | -| Typed Rust client | Compile-time targeting and invocation | -| Qualification identity | Source revision, lockfile digest, image digest, and test evidence bindings | - -The runtime persists and validates canonical descriptor bytes. A deployment is -rolling-compatible only when it can execute every code, schema, codec, -workflow, activity, and stored effect referenced by authoritative Cells. - -An incompatible rollout uses maintenance activation and an explicit transform. -The framework does not retain aliases, fallback readers, or dual-write paths -for unreleased formats. - -## Make local development representative - -The framework supplies a single-process development host that uses the same -registry, actor, SQLite, LTX, control, and typed client path as production. - -```rust,ignore -#[tokio::test] -async fn checkout_survives_owner_loss() -> Result<()> { - let cluster = TestCluster::::new(3).await?; - let client = cluster.client(); - - let placed = client.orders(order_id).create(request, input).await?; - cluster.kill_owner(placed.receipt.cell).await?; - - let restored = client - .orders(order_id) - .get_order(ReadConsistency::After(placed.receipt), order_id) - .await?; - - assert_eq!(restored.output.status, OrderStatus::Pending); - Ok(()) -} -``` - -The test host supports deterministic failure points for: - -- Cancellation before and after SQL begins -- Lost object-store CAS responses -- Owner fencing and takeover -- Source-directory loss -- Follower unavailability and recovery overlays -- Disk admission and filesystem failure -- Duplicate effects and activity attempts -- Rolling-compatible and incompatible releases - -An in-memory object store is useful for fast tests but does not count as -provider, filesystem, process-loss, or power-loss qualification. - -## Apply one resource model - -Application declarations provide bounds, not separate resource schedulers. The -node converts declared maximums and observed local state into its existing -shared resource ledger. - -Admission covers: - -- Active Cells and SQLite connections -- Page-cache and fixed native state -- File descriptors -- Queued and executing commands -- Encoded inputs and results -- WAL, retained LTX, sparse pages, and directory cache -- Page faults and object-store I/O -- Hydration, compaction, recovery, and scratch disk -- Effect, activity, and maintenance jobs -- Follower node-log tails - -Per-application or per-namespace quotas may reserve a share of the node-wide -envelope, but they cannot create capacity outside it. Admission failure happens -before a command handler starts whenever the runtime can know the required -capacity in advance. - -Metrics aggregate by application, namespace, operation, role, and outcome. -They do not use Cell ID as an unbounded metric label. Traces and bounded debug -status may include a Cell ID when authorized. - -## Preserve explicit consistency contracts - -The application API documents these guarantees: - -| Scope | Guarantee | -| --- | --- | -| One command | Application state, request outcome, effects, workflow intents, sequence, and due summary commit together | -| Command retry | Same request ID and operation digest resolves one stored outcome | -| Query after receipt | Same Cell/incarnation at the receipt's commit sequence or later | -| One Cell | Serialized accepted mutations through one actor and SQLite writer | -| Cross-Cell effect | Durable at-least-once delivery with destination inbox deduplication | -| Activity | Retryable leased execution; external idempotency remains destination-owned | -| Workflow | Deterministic durable transition plus retryable activities and effects | -| Failover | Successor opens exact authority and consumes any pinned recovery overlay before serving | - -The API does not claim: - -- Multi-Cell ACID transactions -- A global serial order or global receipt -- Exactly-once external side effects -- Transparent hot-key splitting -- Synchronous global secondary indexes -- Reads from stale local SQLite files -- Success after a local SQLite commit without a durability proof - -## Keep security at the correct boundary - -The application framework validates compiled capability relationships. The -product or service boundary authenticates users and authorizes actions. - -```text -external request - -> transport authentication - -> application authorization - -> bounded typed input - -> generated application capability - -> CellClient - -> local actor or authenticated peer -``` - -Peer transport is private to compatible nodes. A peer accepts only signed, -bounded, registered operations for the same fleet and release compatibility -window. Application SQL, handler closures, credentials, and raw database paths -never cross that protocol. - -Tenant and application IDs are bound into every target Cell ID. Generated -clients bind those IDs once and cannot construct a target for another tenant -without receiving a different authorized `ApplicationHandle`. - -## Deliver in vertical slices - -### Phase 1: stabilize the author surface - -- Define `CellApplication`, `CellType`, entity and shard partition contracts. -- Generate descriptors and typed clients from explicit stable IDs. -- Reuse the existing `Command`, `Query`, `WireValue`, and primitive APIs. -- Add compile-fail and canonical-byte tests for generated bindings. -- Keep current server composition unchanged. - -Completion proof: a small application uses only the author API after receiving -an existing `ApplicationHandle`; generated release bytes match an independently -constructed current `Registry`. - -### Phase 2: add the node facade - -- Introduce `CellNodeBuilder`, `CellNode`, and `ApplicationHandle`. -- Move current server assembly behind that facade without adding another runtime. -- Keep HTTP authentication and provider construction in the product boundary. -- Expose explicit provision, status, drain, and shutdown operations. - -Completion proof: `crab-http-server` uses the facade, and architecture checks -reject direct production composition around it. - -### Phase 3: complete application lifecycle - -- Add explicit and create-command provisioning. -- Generate migration and compatibility inventories. -- Integrate effects, activities, scheduler work, backup, and retention through - the application descriptor. -- Add a three-node deterministic test host. - -Completion proof: an entity plus queue plus workflow application survives -source loss, owner loss, duplicate delivery, activity retry, and rolling -compatible deployment through only public framework APIs. - -### Phase 4: qualify scale and operations - -- Run entity-heavy, shard-heavy, workflow-heavy, and mixed workloads. -- Measure resident Cells, aggregate throughput, hot-key behavior, object-store - operations, WAL/LTX amplification, local disk, RSS, file descriptors, and - takeover latency. -- Exercise pressure shedding, paced movement, follower loss, disk-full, - ambiguous CAS, compaction, backup, and restore under load. -- Bind signed evidence to source, image, provider, topology, and workload. - -Completion proof: published limits describe measured profiles rather than the -current target envelope. - -### Phase 5: publish a supported framework - -- Make crate publication and semantic-versioning decisions. -- Freeze the supported author and operator contracts. -- Publish upgrade, rollback, migration, and deprecation policy. -- Retain low-level LTX and authority surfaces as implementation APIs unless a - separate expert contract is explicitly approved. - -## Reject convenient but unsafe shortcuts - -- Do not expose `CellHandle::execute` as the normal application API. -- Do not let generated clients submit arbitrary SQL or operation IDs. -- Do not infer current state by listing local files or object prefixes. -- Do not acknowledge a command merely because SQLite committed locally. -- Do not retry a mutation after an ambiguous start without resolving its ledger entry. -- Do not hide shard-count changes behind configuration. -- Do not run network I/O inside command or workflow transition callbacks. -- Do not describe effects or activities as exactly once. -- Do not create a second primitive-specific durability or scheduler path. -- Do not make routing, placement, or cached metadata an ownership authority. -- Do not advertise target scale before the application workload matrix passes. - -## Define success from the owner's perspective - -The application framework is complete when an owner can: - -1. Declare entity, shard, workflow, and read-model Cells with stable keys. -2. Implement typed commands and queries without touching runtime internals. -3. Compose built-in primitives through the same application client. -4. Coordinate Cells with typed effects and workflows whose delivery semantics - are visible in the API. -5. Run external activities with framework-supplied durable identity and leases. -6. Test failover, ambiguity, retries, and rollout locally through public APIs. -7. Deploy one canonical artifact through a `CellNode` host. -8. Observe bounded application and namespace metrics without operating one - SQLite service per Cell. -9. Scale by adding nodes and distributing Cells while preserving one-writer - semantics for every individual Cell. -10. Diagnose a hot Cell as an application partitioning problem rather than have - the framework silently weaken its invariants. - -Until the node facade, generated author API, lifecycle tests, and capacity -qualification exist, application modules remain internal Crab integrations -rather than a supported general application platform. diff --git a/crates/crab-cell-runtime/docs/canonical-ltx-scaling.md b/crates/crab-cell-runtime/docs/canonical-ltx-scaling.md deleted file mode 100644 index f4b1caeba..000000000 --- a/crates/crab-cell-runtime/docs/canonical-ltx-scaling.md +++ /dev/null @@ -1,1432 +0,0 @@ -# Complete and qualify canonical Cell LTX scaling - -Crab will finish production scaling on one canonical persistence path: -`Db` captures SQLite, `CellReplica` prepares immutable roots, -`CellRuntime` owns execution and durability, and `crab-http-server` composes the -product. The older standalone epoch-head, paged, and scheduler surfaces were -present in release tag `v1.2.4` but are now hard-removed under the recorded -compatibility decision; their stored prefixes are never interpreted as Cell -roots. - -| Document intent | Value | -| --- | --- | -| Content type | Target design and delivery contract | -| Audience | `crab-ltx`, `crab-cell-runtime`, and `crab-http-server` contributors | -| Goal | Bound canonical Cell resources and qualify production scale/failover on one native Rust path | -| Status | In progress; implementation slices are tracked in `advisor-plans/004`–`017`; standalone hard removal is executed | -| Scope | LTX preparation, authenticated metadata, resident lifecycle, resource accounting, qualification, and standalone-contract consolidation | - -[Back to the Cell runtime index](README.md) - -## Make one path canonical - -The production dependency and authority path is: - -```text -crab-http-server RepositoryCellRouter - -> crab-cell-runtime CellRuntime - -> CellExecutor + CellPublisher + CellAuthority - -> crab-ltx Db + CellReplica - -> crab-ltx CellStorageLayout -``` - -Each module has one responsibility: - -| Module | Responsibility | Explicitly does not own | -| --- | --- | --- | -| `Db` | Exclusive SQLite session, WAL capture, checkpoints, retained local cuts | Remote authority or response release | -| `CellReplica` | Verify and prepare immutable Cell roots, sparse reads, restore, compaction | Mutable owner or root publication | -| `CellExecutor` | Serialized SQL, request ledger, captured mutation outcome | Object-store authority | -| `CellPublisher` | Immutable preparation, ordered publication, ambiguous-result reconciliation | Owner selection | -| `CellAuthority` | Sole mutable owner, epoch, recovery overlay, and root CAS | SQLite or LTX parsing | -| `CellRuntime` | Activation, admission, durability gate, recovery, drain, and local lifecycle | HTTP authentication or provider construction | -| `crab-http-server` | Product routing, authorization, peer transport, deployment, and composition | LTX parsing or alternate publication | - -`CellReplica` writes immutable objects and never changes mutable authority. -`CellAuthority` remains the only module allowed to publish a prepared root into -Cell control. The only legal mutable publication chain is: - -```text -CellPublisher - -> Control::publish_prepared - -> CellAuthority::transition -``` - -Production server code must not call `crab-ltx` directly. Test fixtures may use -the server's `crab-ltx` development dependency to construct exact inputs, but -product behavior crosses the `crab-cell-runtime` interface. - -## Preserve intentional redundancy - -This design removes duplicate ownership, not useful independent mechanisms. - -- `Db` and `CellReplica` are not competing paths. One owns local SQLite - capture; the other owns immutable remote mechanics. -- `CellReplica` and `CellAuthority` are not competing roots. One prepares an - immutable proposal; the other conditionally publishes the sole authoritative - successor. -- Object-store durability and follower durability are intentionally independent - proofs. Either may release a response, but object publication continues in - order and remains long-term recovery authority. -- Logical and published heads are bounded monotonic watermarks around one SQLite - writer and one publisher. They are not independently writable histories. -- Followers store recent verified LTX tails. They do not run writable standby - SQLite databases or serve reads. - -The runtime must continue to enforce: - -```text -response(commit) => object_root_covers(commit) OR fleet_covers(commit) - -fleet_covers(commit) => every selected follower fsynced the commit ticket - -cell.serving => control names this owner and exact root - AND every attached recovery overlay was consumed -``` - -## Meet these design goals - -1. Bound publication memory by configured buffers, not database or capture size. -2. Bound authenticated metadata memory and local cache disk across 10,000 open - Cells. -3. Evict quiescent Cells without losing any acknowledged result or reopening - unverified mutable state. -4. Account SQLite, WAL, retained LTX, sparse pages, follower tails, immutable - cache, Git/LFS staging, and full-job scratch under one node envelope. -5. Prove exact recovery and continued publication through process, disk, - follower, network, and object-store faults. -6. Drive production coordination and deterministic simulation through the same - sans-I/O decision kernel, then model-check its safety invariants at small - scale. -7. Make a fully hydrated resident read route and execute with zero object-store - operations; keep cold activation and durability proof costs explicit. -8. Rebalance quiescent ownership with live, signed resource observations, - hysteresis, bounded movement, and cgroup-aware pressure shedding. -9. Produce signed capacity receipts tied to one source revision, image, node - profile, provider, workload, and fault scenario. -10. Preserve every still-relevant safety test while consolidating the removed - standalone replication contract into Cell proofs. - -The following are not goals: - -- A V8, JavaScript, WebAssembly, dynamic-library, or public primitive host. -- A second mutable SQLite owner or hot SQL standby. In object durability mode, - an explicit desired-reader policy can activate admitted read-only exact-root - views through private peers. Repository issue-detail reads can explicitly - select a replica and return its observed receipt; other product reads still - use the owner while [Plan 036](../../../advisor-plans/036-cell-read-replicas-and-fenced-promotion.md) - completes routing and qualification. -- A fallback from Cell roots to standalone epoch heads. -- Listing local files or object prefixes to infer the latest state. -- Reusing an old mutable SQLite file merely because it exists locally. -- Maintaining a second replication protocol beside the canonical Cell path. -- Claiming Celld performance superiority without matched measurements. -- Treating a route cache, placement plan, or fleet sample as ownership - authority. -- Hot migration of a writable SQLite process or direct owner-to-owner transfer. -- A central mutable scheduler whose loss can stop safe request routing. - -## Close the three major architecture gaps - -This plan adds three first-class workstreams beyond LTX allocation and local -lifecycle. They are gaps in the current canonical Crab path, not evidence that -the path should be replaced. - -| Gap | Current Crab evidence | Target architecture | Required proof | -| --- | --- | --- | --- | -| Protocol assurance | `Control` retains pure persistent transitions. The runtime now has a private coordination state machine, deterministic simulator, and pinned TLA+ model; async adapters carry activation generations and typed per-effect intents/IDs while parity coverage is still expanding. | One private sans-I/O coordination kernel used by production and simulation, a replayable adversarial scheduler, and a TLA+ model of the same durable state machine. | Pinned seeds find deliberately broken variants; model configurations check single-writer and acknowledged-durability invariants; remaining work is full decision extraction/parity, not a second policy path. | -| Warm request latency | `RepositoryCellRouter::route_existing` first asks the actor-owned resident lookup; sparse activation selects and installs bounded background page batches on the SQL worker while fetching asynchronously outside it. The zero-origin post-promotion qualification is still outstanding. | Actor-owned resident lookup before remote metadata, plus bounded background hydration. A fully hydrated local read performs zero object-store operations from route through SQL result. | An instrumented store observes zero calls for qualified resident reads; cold, sparse, hydrating, resident, local-write, fleet-proof, and object-proof latency are reported separately. | -| Fleet balancing | Signed versioned placement observations carry measured node headroom, Cell/job counts, and three backlog counters. The private server loop plans bounded transfers, the actor confirms exact settled releases, and the receiver restores through ordinary authority acquisition. Ownership counts now balance by weighted share beside the material headroom-gain path: one elected donor per complete snapshot, a two-percent receiver deadband, and batch, surplus, and room bounds. Cold activation also sends one authenticated hint to a preferred live node. Local, planner, and process race tests cover exact-root preservation, stale-owner fencing/recovery, donation without headroom gain, refusal to mix pre-batch counts, convergence at target, and failed receiver rollback; protected multi-process movement proof remains. | Deterministic weighted placement over signed live capacity, actor-approved quiescent release, idle eviction, cgroup-aware pressure tiers, hysteresis, and paced drains. Placement remains advisory; existing control CAS remains authoritative. | Skew, membership change, stale samples, pressure, receiver death, rolling drain, and oscillation tests preserve authority and converge within declared movement and latency bounds. | - -The ownership margin reserves only whole Cells: `floor(target * 2 / 100)`. -Rounding it up makes an empty receiver ineligible at a one-Cell target. -Each donation also consumes the receiver's projected room within its batch; -fleet-wide room alone cannot prevent a preferred receiver from overshooting. -The 3/5/10/20-node convergence regression covers these small targets, while -the existing stale-view, settlement and resource gates remain in force. - -The Celld comparison is pinned to upstream commit `10cb1303dac710dcb3b557e318e08c855261f68b`. -Its documentation reports about 1.1 ms p50 and 7 ms p99 for one fixed-host -warm resident request. Those numbers are a comparison baseline, not a Crab -measurement or an unconditional acceptance threshold. Celld also documents a -pure decision core, seeded simulation, small-state specification, weighted -ownership balancing, idle eviction, pressure shedding, and paced drains. Its -current balancing limitation is equally relevant: Cells are weighted by node -capacity but counted uniformly rather than by measured per-Cell CPU or memory. -Crab should close the assurance and routing gaps, then exceed that placement -model without weakening its exact-root and dual durability proofs. - -The current-code evidence map is: - -The standalone compatibility boundary is recorded in -[standalone-replication-audit.md](standalone-replication-audit.md). Its -decision is HARD REMOVE; the public standalone surface has been deleted while -shared authenticated mechanics remain private to Cell roots. - -| Surface | Current owner and behavior | -| --- | --- | -| Request entry | [`RepositoryCellRouter::route_target`](../../crab-http-server/src/cells/router.rs) calls `route_existing` twice around an activation lock, then repeats catalog and control loads for activation. | -| Metadata lookup | [`CellCatalog::lookup`](../src/cell/catalog.rs) loads the shard head and every referenced immutable catalog page; [`CellAuthority::load`](../src/control/authority.rs) separately reads exact control. | -| Local residency | [`CellRuntime::resident_handle`](../src/cell/actor.rs) asks the actor for a fully resident owner before remote metadata; [`local_handle`](../src/cell/actor.rs) remains the verified slow-path lookup for sparse or activation callers. Fenced, draining, and non-resident actors miss safely. | -| Sparse hydration | [`Db::prepare_hydration` and `install_hydration`](../../crab-ltx/src/db.rs) bracket asynchronous fetch; a separate hydration effect permits foreground work and retains drain obligations. Cancellation, overwrite and takeover tests cover the split; fleet latency qualification remains. | -| Fleet observation | [`NodePublisher`](../../crab-http-server/src/peer.rs) signs short-lived measured capacity and backlog observations; `NodeAdvertisement` carries a versioned placement signature. [`RepositoryCellRouter`](../../crab-http-server/src/cells/router.rs) plans movement from live signed samples and actor-settled candidates, then records confirmed release and receiver activation separately. Advertised disk headroom is clamped by the runtime ledger, server memory resolves nested cgroup-v1/v2 membership, and cold activation sends a bounded direct-node hint before normal authority acquisition. The test-only process race covers one shared-control winner; unified process-wide probe parity and protected multi-process movement proof remain. | -| Existing rendezvous | [`preferred_scanner`](../src/fleet/scheduler.rs) elects a catalog scheduler scanner. It does not rank or move Cell owners. | -| Transition safety | [`Control`](../src/control.rs) validates named single-record transitions; [`coordination.rs`](../src/coordination.rs) allocates and retires typed per-effect intents/IDs, while the actor fences completions by activation generation and effect family, drains the kernel-owned pending-effect set before fenced deactivation, and keeps effect timing coupled to the production publisher. Background hydration, renewal, persisted-work inventory refresh, drain, and shutdown pass queue/publisher/lease observations through the same kernel schedule transition before an adapter starts work. | - -### Closed-book LTX telemetry and the prefetch gate - -Capture telemetry is a fixed-size per-batch ledger. It attributes schema checks, -WAL existence and position resolution, WAL reads and page collection, encoding, -local writes, file sync, parent sync, verification, and checkpoint time. The -same ledger records logical WAL work, physical WAL file/read bytes, allocated -image bytes, finite read/snapshot strategy counters, and checkpoint runs, busy -outcomes, frames, backfill, and restarts. `Db` emits the ledger for both -successful and failed capture attempts; publication is not a second reporting -boundary. None of these observations is persisted or participates in authority, -checkpoint, or recovery decisions. - -Replica telemetry crosses into `crab-cell-runtime` only as the closed enums -`LtxPhase`, `LtxReadOrigin`, and `LtxRequestOutcome`. Logical reads are counted -separately from provider attempts. Provider attempts retain succeeded/failed -outcomes and bytes returned before failure for cold, sparse, and hydrating -reads; resident reads increment only the logical counter and perform no -provider operation. The runtime exports phase result/duration and these finite -counters for cold, sparse, hydrating, and resident reads. -Cell IDs, paths, object keys, digests, and arbitrary caller strings cannot be -labels. Root-open, authenticated-directory, frame-fetch, ordered restore-write, -and compaction paths report success and failure through the same host hook. - -Exact-root compaction downloads every selected authenticated body once into -scratch in bounded 1 MiB chunks. The same admitted spool supplies frame decode -and output encoding, eliminating the previous hash-verification pass followed -by a second provider read. Source-body bytes are included in scratch admission; -body BLAKE3, frame hash, page number, page checksum, final LTX checksum, and -no-clobber publication checks remain unchanged. - -B-tree-guided speculative prefetch remains disabled until traces collected by -these counters demonstrate a scan workload whose p95 improves without -regressing the existing point-read contract. The current baseline already -coalesces one authenticated 64-page window: `cold_open_and_restore_improve_p95_under_object_latency` -proves one body request for a point fault under injected latency, and -`sparse_hydration_coalesces_contiguous_cell_frames` proves fewer range requests -than hydrated pages. A future predictor is acceptable only when all of the -following are verified: - -- malformed SQLite pages produce no prediction; -- prediction changes fetch timing only, never exact-root authority checks; -- point reads issue no additional provider request; -- speculative workers, bytes, and cache residency use existing host admission; -- scan p95 improves on recorded workloads at 5/20/100 ms provider latency. - -### Implementation evidence and remaining qualification - -The first implementation slices now have one code path each: the architecture -guard rejects production `crab-ltx` imports from `crab-http-server`; the actor -uses a private coordination state machine for admission, scheduling, fencing, -publication, renewal, migration, inventory refresh, and drain, while the kernel allocates and retires typed -effect intents/IDs and the actor fences completions by activation generation -and effect family before fenced -deactivation; the deterministic simulator and bounded TLA+ model exercise the same lifecycle -predicates; the simulator's movement release also passes queue/publisher -observations through the same deactivation gate; resident-only lookup is -actor-owned and attempted before -catalog/control I/O; sparse restored Cells receive bounded -page selection and installation on the existing SQL worker with asynchronous -fetch outside it; active-cell admission -uses an exact RAII resource ledger (including resident native bytes, active-Cell -file-descriptor reservations, bounded SQL-worker, hydration-job, and primitive -activity/effect reservations, with runtime metrics for hydration and descriptor -usage/capacity); the runtime installs a weak -ledger admission on `crab_ltx::DiskBudget`, imports existing local bytes, and -keeps LTX reserve/resize/release usage identical to the advertised disk total; -persisted -Queue/Workflow rows are re-inspected after durable work before eviction; the -eviction seam now has a pure -deterministic selector that -excludes unsafe obligations; directory-cache files are restart-persistent, -verified, and charged to the shared local-disk budget, while the HTTP startup -path inventories every regular file in prior process session directories and -holds those bytes in the same budget, rejecting symlinked or special layouts; -native and bundle -publication share the authenticated LTX inspection path with replayable scratch -sources; placement/pressure decisions are pure fixed-point functions with -versioned signed observations whose advertised disk total and free headroom are -reconciled with the runtime ledger; placement's bounded `u32` wire projection saturates -host-sized counters rather than wrapping; and qualification receipts are signed, bounded, -artifact-bound records. The `qualification_receipt` binary verifies exact -source, image, and artifact identity, and the release workflow consumes only -that bound evidence. The isolated local RustFS LTX, Cell takeover/retention, -HTTP collaboration, native-push, and receive-fault qualifications now pass; -they are provider evidence, not release receipts. Matched warm-restart -zero-origin latency receipts, complete advertised/metric parity, multi-process -movement, and protected Kubernetes faults remain release gates. The -standalone-surface decision is recorded and its execution is in this change. -The cold-activation planner seam -and its local receiver-failure rollback are implemented locally: a failed -rooted idle acquisition or fenced-owner takeover releases the takeover through -the canonical publisher path and leaves the exact root unowned. A pinned -recovery overlay remains owned until its follower proof is replayed and sealed. -The multi-process movement, membership-loss, and fleet-convergence receipts -still belong to the protected qualification gate. - -The canonical native publication path now has its dedicated multi-GiB receipt: -the release `rustfs_cell_replica_scale_load` example grew a 5,368,709,120-byte -incompressible SQLite source through 160 bounded captures (320 immutable -segments) against RustFS 1.0.0-rc.1, deleted the source, restored the published -root, compacted the complete range, restored the compacted root, and matched the -source BLAKE3/length exactly. `/usr/bin/time -l` recorded 1,496.96 seconds wall -time and 592,805,888 bytes maximum resident set size (~565 MiB); the largest -observed compaction scratch LTX was about 5.1 GiB on the external qualification -volume. This closes the native 5 GiB/RSS publication gate; provider matrices and -protected fleet receipts remain separate gates. - -The local warm-path regression -`resident_route_reports_zero_origin_reads_and_latency_percentiles` runs 64 -resident-handle plus SQL reads after activation through an instrumented -`Store`; the latest run recorded p50 67us, p95 90us, p99 364us, max 364us, -and zero origin reads. It is intentionally labeled local evidence rather than -a matched-hardware or signed release receipt. - -The companion -`restored_sparse_route_promotes_before_zero_origin_reads` publishes and drains -a Cell, reacquires its exact root through a new runtime, waits for verified -sparse hydration to promote the resident route, and observes zero origin calls -on the subsequent SQL read. This closes the local warm-restart regression seam; -provider-scale and signed release receipts remain separate gates. - -The upstream comparison is supported by Celld's pinned -[`docs/testing.md`](https://github.com/denoland/celld/blob/10cb1303dac710dcb3b557e318e08c855261f68b/docs/testing.md), -[`crates/logic/rebalance.rs`](https://github.com/denoland/celld/blob/10cb1303dac710dcb3b557e318e08c855261f68b/crates/logic/rebalance.rs), -and -[`docs/limitations.md`](https://github.com/denoland/celld/blob/10cb1303dac710dcb3b557e318e08c855261f68b/docs/limitations.md). - -## Deepen existing modules instead of multiplying surfaces - -The design adds implementation behind three narrow interfaces: - -| Module | Interface | Hidden implementation and leverage | -| --- | --- | --- | -| Coordination kernel | Step from one explicit state and input to one decision | Lifecycle predicates, fencing, durability release, recovery, timers, and movement stay local. Production and simulation gain the same behavior without learning its internal branches. | -| `CellRuntime` resident lookup | Resolve one target to a current local handle or a miss | Actor map, admission generation, node lease, owner epoch, root, hydration class, and invalidation stay local. Routers do not assemble a second cache policy. | -| Placement planner | Rank eligible nodes and propose bounded actions from one signed snapshot | Resource normalization, weighted rendezvous, deadband, cooldown, pressure tiers, and movement budgets stay local. Execution still crosses the existing actor and authority interfaces. | - -These modules pass the deletion test: deleting any one would spread the same -rules back across production routing, simulation, pressure handling, and tests. -They are private by default. The effect seam is real because production and the -simulator provide different adapters. The resource-probe seam is real because -Linux cgroup and portable process/host adapters differ. Do not add a public -route-cache trait, ownership-provider trait, or generic planner framework for a -single adapter. - -The interface is also the test surface. Protocol properties enter through the -coordination step, route tests enter through `CellRuntime` lookup, and placement -tests enter through a complete signed fleet observation. Tests must not reach -past those seams to mutate internal maps or manufacture authority. - -## Make coordination deterministic - -### Extract decisions, not storage abstractions - -Keep the existing storage implementations and public interfaces. Add a private -`coordination` module inside `crab-cell-runtime` that accepts plain immutable -observations and returns decisions. It owns no Tokio handle, object-store -client, filesystem path, SQLite connection, wall clock, random source, or -network client. - -The kernel covers decisions that currently span `actor.rs`, `publication.rs`, -`node_lease.rs`, `node_log_state.rs`, and `node_log_recovery.rs`: - -- Admit, queue, reject, or fence a command. -- Start, reconcile, retry, or abandon an immutable publication. -- Release an output after object or fleet durability proof. -- Renew, activate, publish, quiesce, release, take over, or tombstone control. -- Attach and consume one exact recovery overlay. -- Select a due timer, retry, or bounded maintenance action. -- Start or refuse an eviction, pressure drain, or ownership move. - -`Control::validate_transition` remains the single-record predicate. The kernel -composes that predicate with node lease, publisher, follower, recovery, and -local lifecycle observations. It must not copy transition rules into a second -simulator-only implementation. - -The coordination seam has three value families: - -```text -CoordinationState - = durable observations + local actor state + admitted work watermarks - -CoordinationInput - = request | timer | I/O completion | lease observation | shutdown | fault - -CoordinationDecision - = next local state + ordered effects + externally releasable outputs -``` - -An effect carries a stable operation ID, complete preconditions, bounded size, -and the authority token it observed. Production adapters execute effects and -feed typed completions back into the kernel. They never mutate kernel state -behind its back. Retried completions are idempotent; unknown completions fail -closed. Only the production adapter owns secrets, byte bodies, ETags, file -handles, and network connections. - -### Preserve one production implementation - -The async actor becomes an executor around the kernel: - -1. Poll one external input or completed effect. -2. Call the pure step function. -3. Persist or dispatch the returned effects in order. -4. Feed every success, explicit rejection, timeout, cancellation, and - ambiguous result back as a typed input. -5. Release a response only when the returned decision names a satisfied - durability proof. - -Production may coalesce safe reads or immutable uploads, but coalescing cannot -hide a completion from the state machine. Timers carry explicit logical -deadlines. Random identifiers and backoff jitter are supplied as inputs. This -makes the test driver and production loop exercise the same decisions without -making object I/O synchronous or moving large bytes into the kernel. - -### Drive a seeded adversarial simulator - -Add a test-only simulator whose complete run is determined by a printed seed -and scenario version. It models several nodes, Cells, clients, one linearizable -conditional object store, local disks, follower logs, and independently -advancing clocks. At every step it chooses among enabled inputs and may: - -- Delay, duplicate, reorder, reject, or lose the response to an accepted I/O. -- Crash a process or individual async effect at every production await seam. -- Expire, renew, or observe a node lease near its deadline. -- Fill owner, follower, cache, or scratch disk. -- Partition peer traffic independently from object storage. -- Restart with empty mutable state and retained immutable cache. -- Race publish, takeover, recovery attachment, migration, eviction, and drain. -- Pause a handler or stream after acceptance but before completion. - -Every failure prints the seed, minimized event trace, initial state, and final -invariant violation. CI keeps a bounded deterministic seed corpus plus every -historical failing seed. A scheduled broad job explores new seeds. A protocol -change may retire a seed only when the scenario is invalidated and the reason -is recorded. - -The simulator continuously asserts: - -```text -at most one output-capable owner per Cell epoch -acknowledged(commit) => object_covered(commit) OR recoverable_fleet_covered(commit) -serving => live node lease AND matching owner/epoch/root AND no pending overlay -published_root and logical sequence never rewind -takeover cannot consume a partial or unpinned recovery tail -released owner cannot emit output or revive its mutable database -all reservations, leases, and durable dependencies remain bounded and owned -eventually, after faults stop, accepted work resolves or reports unknown -``` - -Deliberately broken variants disable one fence, durability gate, CAS predicate, -or overlay precondition. The checker must find each defect within a pinned seed -budget. This proves that green properties are capable of observing the class of -failure they claim to prevent. - -### Model the smallest durable protocol - -Add a TLA+ model under `crates/crab-cell-runtime/model/` for two Cells, up to -three nodes, bounded commits, object publication, follower proof, lease expiry, -takeover, recovery attachment, release, and stale-owner output. Keep SQL, -payload bytes, transport encoding, Blob, Queue, Workflow, Cron, and JavaScript -outside the model; they matter only as accepted work and durable effects. - -Each checked configuration pins an expected verdict. Passing configurations -must establish single-writer, no-lost-acknowledgement, monotonic-root, and -fencing invariants. Negative configurations intentionally remove a rule and -must produce a counterexample. The model is a reviewed specification, not a -generated mirror: `model/README.md` records the production commit, modeled -transitions, known abstractions, and every deliberate delta. - -The pinned TLC version and checksum live in the verification tooling, not in a -production dependency. A fast small configuration runs for protocol changes; -the broader state space runs in scheduled CI. A model result never replaces -Rust simulation or real-fleet qualification. - -### Exit the assurance phase only with evidence - -- Production and simulation call the same pure transition functions. -- Every coordination await seam is representable as a simulator completion or - crash point. -- Historical seeds replay exactly and broken variants fail as expected. -- The TLA+ model and delta ledger name every authority and durability - transition in the production protocol. -- The simulator covers recovery, rollout, eviction, and pressure movement in - addition to the happy path. -- Existing async integration tests remain as adapter and real-I/O proof. - -## Make warm resident reads local - -### Put local lookup before remote discovery - -`RepositoryCellRouter` first derives the deterministic `CellTarget`, then asks -`CellRuntime` for an active local route. The actor map already owns the only -valid process-local admission capability and the verified `CatalogProof` used -to create it. Extend that lookup to return an opaque local route certificate -containing the handle and its current incarnation, code, schema, owner session, -epoch, root, and local hydration state. - -The certificate is process-local, non-serializable, and short-lived. It is not -an ownership cache. Lookup succeeds only when: - -- The runtime node lease is currently usable. -- The actor is `Active`, not activating, quiescing, draining, migrating, or - fenced. -- The requested target exactly reconstructs the actor's verified catalog - entry and Cell ID. -- The actor's publisher still holds the matching owner session, epoch, and - authoritative root. -- The current admission generation is the same generation embedded in the - returned handle. - -On any miss, the router uses the existing slow path: load and verify catalog, -load exact control, route to a live peer or activate from the authoritative -root. Catalog and control observations from that path may populate immutable -process-local acceleration, but they never permit takeover or publication -without the normal fresh CAS preconditions. - -### Invalidate from the owner seam - -The runtime removes or rejects a local route before it starts migration, -quiescing, pressure drain, explicit drain, or shutdown. A node-lease fence -atomically closes route lookup and Cell admission before peer takeover is -possible. A failed control renewal fences the actor; a later local request -cannot fall back to its stale handle. - -No timeout-based cache invalidation is required for local ownership. The -capability lifetime is coupled to actor admission and the node lease. Remote -owner endpoints and cold Cells are never served from this local index. - -### Finish bounded background hydration - -Sparse activation remains legal and may begin serving after exact-root -verification. It is called `ActiveSparse`, not fully resident. The actor -selects up to 64 pages with `Db::prepare_hydration` on its SQL worker, -fetches authenticated pages asynchronously, then dispatches -`Db::install_hydration` to that activation. Foreground queries and mutations -can run while fetch is in flight. Hydration owns a separate effect identity -that still prevents drain or transfer from releasing an unfinished activation; -its completion cannot clear another command's foreground slot. - -Hydration: - -- Reserves incremental local disk before each page batch. -- Uses existing page-I/O, object-I/O, and job admission. -- Reserves fetched payload bytes before origin work and carries the reservation - through queued installation; canceled callers cannot release live worker bytes. -- Verifies every directory node, frame, page checksum, and final hydration - count. -- Starts batches only while the Cell's foreground queue and publication are idle. -- Defers preparation/fetch timeouts and retryable fetch errors for at least one - second, honoring longer provider delays. Permanent fetch errors and uncertain - or failed installation still fence the owner. Demand reads retain their - synchronous VFS contract; shared-worker installation latency remains to qualify. -- Is cancel-safe on fence and eviction; partial verified pages remain only as - disposable local state. -- Promotes the actor to `ActiveResident` only after every inherited allocated - page is locally materialized or superseded by a local write. - -`ActiveResident` is a performance classification, not durable authority. A -subsequent database growth stays local, while a migration, root change, or -takeover creates a new activation and must earn the classification again. - -### Define the zero-operation claim precisely - -A **qualified warm resident read** is an authenticated read-only request whose -Cell is `ActiveResident`, whose handler needs no remote Blob or object-store -value, and whose result fits existing response admission. Local SQL and KV -reads remain part of the claim. From the start of Cell -routing through completion of its SQLite query, it performs zero calls to the -fleet object store. TLS, ingress, application authorization, and peer transport -are measured separately. - -The claim does not include: - -- Cold or sparse activation and first-page faults. -- A mutation, which waits for object or fleet durability proof. -- Explicit application access to remote Blob or object storage. -- Migration, backup, retention, or scheduler scans. -- A request initially received on a non-owner and forwarded to the owner. - -Instrument the storage adapter with request origin (`route`, `page_fault`, -`publication`, `primitive`, or `maintenance`). Qualification fails if a warm -resident read increments any origin. Report p50, p95, p99, and maximum for -local-route lookup, actor queue, SQL execution, and full request separately. - -### Exit the resident-routing phase only with evidence - -- Repeated local lookups perform no catalog-head, catalog-page, control, node - directory, or immutable-root object reads. -- A qualified warm resident read records zero object-store calls end to end. -- Concurrent fence, drain, migration, and takeover tests never use an old - admission generation. -- Sparse reads remain exact while hydration proceeds, and promotion occurs - only after complete verified local coverage. -- Hydration and route acceleration stay within the node memory, disk, - descriptor, and job envelope. -- Crab latency is reported from matched hardware; the Celld fixed-host figures - remain an external baseline until reproduced under the same workload. - -## Balance ownership under live pressure - -### Separate placement from authority - -Add a private fleet placement controller in `crab-http-server` and a pure -planner in `crab-cell-runtime`. The planner consumes a signed, revision-pinned -fleet observation and produces advisory actions. It cannot write Cell control, -construct an owner, or bypass actor admission. - -```text -signed node observations + owned Cell summaries - -> pure weighted planner - -> keep | stop-acquiring | hydrate | evict | quiesce-and-release - -> actor proves quiescence and durability - -> CellAuthority release CAS - -> ordinary preferred-node activation -``` - -The existing scheduler rendezvous continues to assign catalog maintenance -scans. Ownership placement is a different decision and must not overload that -interface or make scheduler progress an authority prerequisite. - -### Advertise live usable capacity - -Extend the signed node advertisement with a versioned placement block derived -from the same resource envelope used by admission. It includes: - -- Effective memory limit, current cgroup working set or process RSS fallback, - allocator estimate, and memory headroom. -- Effective disk limit, physical free bytes, admitted local/follower/scratch - bytes, and disk headroom. -- Available file descriptors, worker/job credits, activation backlog, and - publication/follower backlog. -- Active, sparse, resident, quiescing, and owned Cell counts. -- Draining and pressure tier, placement weight, sample generation, and sample - time. - -On Linux, the resource probe reads cgroup v2 limits, current charge, and memory -events. On other platforms or unreadable cgroups it uses the already-qualified -process and host probes and marks the source. The controller reasons about -usable headroom, never host totals hidden behind a container limit. - -Advertisements remain short-lived and signed. A mixed fleet that does not -publish the required placement version may route existing ownership normally -but performs no proactive movement. This is a rollout gate, not a compatibility -fallback. - -The current signed placement schema is version 2. It carries three bounded -backlog counters: publication pressure in 1 MiB units of retained native work -and unrooted node-log bytes, plus admitted hydration and primitive job counts. -An advertisement without a runtime measurement has no signed placement block. -Draining nodes retain a signed block with zero free capacity so donors remain -visible but cannot receive new Cells. The pure transfer planner caps one -tick at two Cells and 8 GiB of projected disk restore, with absolute receiver -memory, disk, Cell-slot, and job-credit checks. It requires two stable samples -and a 60-second residence/cooldown for ordinary movement; explicit drain and -sustained shedding bypass the score-gain gate only. - -Ownership movement has two ordinary reasons, and both stay advisory. A -balancing move answers a count question: each member's weight is its declared -Cell capacity, its target is the fleet's owned Cells shared by that weight -(rounded up, so the targets always cover the fleet), and only the member with -the most Cells per unit of weight may donate. That donor releases at most its -surplus, at most the two-Cell batch, and at most the receivers' room below a -two-percent deadband. Receivers are members below their own target that are -fresh, eligible, and not shedding, least dense first. One snapshot therefore -elects one donor and cannot hand a Cell to a member that its own next sample -would send back. A relief move answers a resource question and keeps the -material headroom-gain gate. Both paths share settlement, residence, cooldown, -and projected receiver capacity, so a balancing move cannot skip an actor -gate. - -A balancing view fails closed. Every live member must publish a fresh signed -placement block, and every sample must be taken after the instant this node -dispatched its previous movement batch: a partial or mixed total lowers every -target and moves Cells that come straight back. Without that complete view the -loop moves nothing on the count rule and keeps the drain and relief paths, -which carry their own per-node freshness checks. These are advisory limits; -the actor still rechecks the exact Cell generation and a transfer-specific, -indexed durable-work inspection before release. Retained request/inbox -outcomes, Blob metadata, Queue producer identities, and future Cron schedules -may follow the exact root. Live or due source effects, ready or leased Queue -messages, pending Workflow activities/timers, due Cron delivery, and unknown -inspection state block movement; the maintenance-release inventory remains -conservative and unchanged. -The private server controller runs every 15 seconds, samples signed live nodes -and actor-approved local candidates, then releases exact generations through -the actor before sending an authenticated receiver activation hint. If receiver -activation fails, the exact unowned root remains available for normal routing. -Residence evidence survives temporary work or renewal while the activation -remains resident. A changed activation generation resets it; removal or drain -discards it. Movement still requires the ordinary 60-second idle window and -the actor's fresh transfer inspection before release. -Each tick reports confirmed source releases and successful receiver activations -separately; a started drain is not counted as a completed move. -The scale-down host state stops new acquisition, paces exact actor releases, -and reports released, blocked, and remaining Cell counts while retaining the -node lease and facilities for incomplete deadlines. A successful drain invokes -the existing terminal shutdown only after ownership reaches zero. Measured -per-Cell disk demand, shared fleet-wide movement accounting, and protected -provider/Kubernetes evidence remain qualification work before production -rollout. -The mTLS management listener exposes `POST /internal/cells/v1/scale-down` for -an operator or orchestrator to request this same drain: `200` means the node -reached `Stopped`, while `202` reports a bounded incomplete drain that is safe -to retry without releasing blocked ownership. - -### Weight Cells by measured cost - -Each actor emits a bounded placement summary with exponentially weighted -recent demand: - -- Reserved SQLite/native memory and local disk bytes. -- Foreground CPU time and request rate. -- Queue depth and oldest accepted-work age. -- Publication and follower-proof backlog. -- Sparse bytes remaining and estimated restore cost. -- Last foreground use and next durable scheduler deadline. - -The first implementation uses existing reservations as hard cost and recent -CPU/request observations only as tie-breakers. It does not guess unmeasured -memory. Later weights require receipt-backed evidence that they improve tail -latency or convergence. Missing or stale summaries use conservative declared -reservations. - -Weighted rendezvous ranks eligible nodes for a Cell using the fleet snapshot -digest, Cell ID, node session, and normalized resource headroom. The ranking is -stable for one snapshot and changes minimally when membership changes. A node -is ineligible when draining, hard-pressured, lease-stale, release-incompatible, -over its activation backlog, or unable to reserve the Cell's conservative -cost. - -### Add hysteresis and bounded movement - -The planner uses three pressure states derived from existing reserves rather -than new environment switches: - -| State | Entry | Exit | Action | -| --- | --- | --- | --- | -| Normal | All declared headroom above soft reserve | N/A | Admit preferred cold Cells and retain active Cells. | -| Soft pressure | Any resource crosses its soft reserve for multiple samples | All resources clear a higher exit margin for multiple samples | Stop new acquisition, pause background hydration, evict oldest eligible idle Cells, and plan bounded quiescent releases. | -| Hard pressure | Memory, disk, descriptors, or job backlog crosses its hard reserve, or cgroup events show sustained reclaim/OOM risk | Return through soft pressure; never jump directly to Normal | Reject new local activation, shed eligible Cells in paced batches, preserve durability work, and fail readiness if safe progress cannot restore reserve. | - -Entry and exit use separate thresholds, consecutive-sample requirements, a -minimum Cell residence time, a post-move cooldown, and a per-node movement -budget. Only one bounded donor set moves for a snapshot generation. The fleet -caps concurrent activation, hydration, and drain bytes as well as Cell count, -so many large Cells cannot evade a count-only limit. - -Idle eviction first closes SQLite and retains verified immutable cache within -budget. A balancing or pressure move additionally releases ownership to -`Idle`. Normal cache eviction may keep an active owner only if the local actor -can safely reopen within its existing authority; this design initially avoids -that additional hibernation state and uses full release. - -### Move through ordinary recovery - -Crab does not add a direct owner-transfer record. A movement is: - -1. Planner proposes a destination and records the fleet snapshot digest. -2. Current actor rechecks eligibility and enters `Quiescing`. -3. It closes new admission, drains accepted work, and makes every acknowledged - commit object-covered or recovery-pinned. -4. It CASes the exact owner, epoch, and root to unowned `Idle`. -5. Router preference sends the next activation to the highest-ranked eligible - node, which rereads control and acquires through the existing idle- - acquisition path. -6. The destination restores the exact immutable root and only then serves. - -The proposed destination is a hint. If it disappears or loses capacity before -step five, the next eligible node may acquire. The donor never releases merely -because a receiver promised capacity. Failure before release leaves the donor -authoritative; failure after release leaves an exact unowned root that any -eligible node can acquire. Hot live migration and writable local-file transfer -remain outside scope. - -Node shutdown uses the same mechanism with a stricter paced-drain budget. It -stops acquisition first, drains quiescent Cells in bounded batches, and reports -remaining active, durability-blocked, and restore-in-flight counts. A deadline -may terminate availability, but it cannot skip the durability or authority -preconditions for a clean handoff. - -### Exit the balancing phase only with evidence - -- Adding and removing nodes converges weighted resource load without changing - a Cell's acknowledged contents or producing two output-capable owners. -- A skewed mix of tiny, memory-heavy, disk-heavy, hot, sparse, Queue, and - Workflow Cells balances by measured cost better than count-only placement. -- Stale, missing, mixed-version, or forged samples stop proactive movement and - do not stop ordinary authority routing. -- Hysteresis and cooldown prevent ping-pong under oscillating cgroup memory, - disk, descriptor, and workload pressure. -- Receiver death before and after release recovers through the ordinary exact- - root path. -- Movement rate, concurrent restore bytes, foreground p99 impact, and time to - restore reserve remain within the signed profile thresholds. -- Fleet drain with one failed receiver remains bounded, observable, and safe. - -## Apply the architecture uniformly to native primitives - -SQL, KV, Blob, Queue, Workflow, Cron, and effects remain behaviors behind one -Cell interface. None receives a separate owner, publisher, scheduler, or -replication protocol. - -| Primitive | Resident and movement rule | -| --- | --- | -| SQL and KV | Read-only operations participate in the qualified resident-read claim when their complete SQLite state is local. Mutations use the same output and durability gates. | -| Blob and object storage | Blob metadata remains Cell state, while explicit remote body access is reported as `primitive` object I/O and is outside the zero-object-operation read claim. Blob references needed by an acknowledged result remain retention dependencies during drain. | -| Queue | Ready messages, leases, dead-letter work, and oldest due age contribute to eviction eligibility and placement cost. A live delivery lease prevents movement. | -| Workflow | Pending activities, timers, retries, and terminal cleanup contribute to eviction eligibility and placement cost. Activity completion still enters through the normal command and durability path. | -| Cron | The next durable fire time remains in authoritative Cell control. A move must leave it visible to scheduler scans; no process-local timer may be the only record. | -| Effects | Pending delivery and resolution prevent unsafe eviction. The coordination kernel models their durable acceptance and completion without understanding application payloads. | - -Primitive transition logic remains deterministic and separately testable. The -coordination kernel models it as accepted work, durable state change, due work, -and an optional external effect. This keeps the kernel deep without making its -interface grow for every primitive operation. The placement planner consumes -bounded summaries, not Queue messages, Workflow histories, Blob bodies, or SQL -rows. - -Qualification includes mixed primitive Cells so a Queue backlog, long-running -Workflow, imminent Cron fire, or large Blob dependency cannot be hidden by an -otherwise idle request rate. A Cell that cannot safely quiesce stays owned and -reports the blocking class; pressure policy may reject new work but never -silently drops the primitive's durable obligation. - -## Stream canonical Cell publication - -### Describe the current allocation - -Native capture preparation copies each admitted segment into an owned scratch -file and reopens it through the bounded authenticated inspector. Bundle -preparation now has the same source shape: `Bundle::decode_file` and recovery -manifest reopen retain a verified file path plus row metadata, while selected -rows are read by exact extent and uploaded through a replayable multipart -source. The bundle is first written to a deterministic digest-scoped staging -key and promoted through the content-addressed CAS before the staging key is -removed. - -Admission bounds the total bytes, but admission is not the same as bounded -resident memory. The remote recovery path no longer retains a complete bundle -body: it streams the provider response to a runtime-owned session/cell scratch -file, checks the outer digest before CRB1 parsing, reserves the exact remote -size in the shared recovery `DiskBudget`, and performs structural/LTX -verification on a blocking worker. The reservation travels with the returned -overlay until its temporary file is dropped. The node restart inventory counts -regular files left in stale session directories—including compaction and -recovery scratch—before admitting new work; it rejects symlinked or special -entries. Server startup warns with the stale-session count, charged bytes, -remaining shared disk budget, and budget capacity when earlier session -directories remain. That reservation is conservative accounting, not cleanup; -the files stay charged until an exclusive reclaim protocol is proved. Newly -encoded node-log overlays still begin in memory and remain a -separate peak-residency qualification item. A 5 GiB Cell is built from bounded -cuts; its size must not increase the memory used by any later incremental -append. - -### Replace body ownership with admitted sources - -The private append implementation will consume an ordered list of admitted -sources rather than `Vec` bodies. A source is one of: - -- An exact captured local segment path plus its immutable `SegmentInfo`. -- An exact byte range within an already verified recovery bundle. - -The source list owns routing and expected metadata, not decoded contents. It is -private to `crab-ltx`; callers continue to use `CellReplica::prepare`, -`prepare_bundle`, and `prepare_recovered_overlay`. - -For every source, preparation performs this sequence: - -1. Reserve dirty and scratch capacity for the complete operation. -2. Open the exact source through the injected `Host` filesystem. -3. Stream LTX verification through bounded reads. -4. Stream the authenticated page index into a synced scratch file. -5. Verify the complete body BLAKE3, metadata, page order, TXID range, database - checksum, and index digest. -6. Upload the body and index through bounded multipart reads. -7. Retain only the validated descriptor and the scratch reference needed by - directory construction. -8. Feed authenticated index entries into initial or incremental directory - construction. -9. Remove owned scratch after success or failure. - -Preparation still returns one `PreparedRoot`. Uploading immutable bytes is not -publication. A cancellation, failed upload, or caller drop may leave orphan -immutable objects, but it cannot create a publishable root missing a dependency -or change Cell control. - -### Keep verification exact - -Streaming must preserve the current verification contract: - -- Every segment matches its captured `SegmentInfo`. -- TXID ranges are contiguous and begin at the expected predecessor. -- Page numbers are strictly ordered and exclude the SQLite lock page. -- Every frame hash and decoded page checksum matches. -- The resulting database checksum equals the declared endpoint. -- A bundled segment remains bound to its bundle digest and exact extent. -- An index digest binds the exact encoded index bytes. -- The root binds Cell, incarnation, sequence, schema, TXID, checksum, directory, - and segment pages. - -No error may cause a retry to reinterpret bytes through a less strict reader. - -The file-backed bundle contract is deliberately fail-closed. A provider read, -digest, footer, row, LTX, or multipart error is returned to the caller; it does -not fall back to object listing, a native row, or an alternate bundle source. -Temporary files are owned by the bundle and are removed when the overlay is -dropped or decoding fails. Remote staging keys are deterministic for the -content digest, so a retry or failover converges on one unreferenced upload -target rather than creating a fresh key for every attempt. A process death can -still leave that one private staging key; remote staging scavenging remains a -provider-retention qualification and is never used as a recovery reader. -### Bound transfers - -The implementation uses existing configured facilities rather than new -environment options: - -- Bounded filesystem transfers for verification and scratch. -- Existing object I/O permits. -- Existing CPU, dirty-job, recovery, and scratch admission. -- Existing multipart object uploads. -- At most one bounded body window and one bounded index window active per - worker. - -The exact buffer sizes remain implementation constants selected from existing -limits. They are recorded in qualification receipts and changed only with -before/after measurements. - -### Test every ambiguous point - -Fault injection covers: - -- Source read failure before and after verified bytes. -- Scratch create, write, sync, rename, and parent-sync failure. -- Index encoding or checksum failure. -- Multipart failure before and after an accepted part. -- Cancellation while verification or upload is running. -- Local disk exhaustion before admission and after an installed scratch file. -- Object-store timeout and accepted-write response loss. -- Retry with already-uploaded immutable objects. - -The observable assertions are unchanged root authority, bounded retained -admission, cleaned owned scratch, and exact retry behavior. - -### Exit the publication phase only with evidence - -The phase is complete when: - -- A 5 GiB incompressible Cell grown through legal bounded cuts restores and - compacts exactly under declared peak RSS and scratch ceilings. -- The maximum legal captured append uses memory proportional to configured - transfer buffers rather than its complete body and index size. -- No remote body read or upload begins before complete operation admission. -- Corrupt body, index, bundle, or directory bytes fail closed. -- Cancellation and disk-full tests leak neither capacity nor owned scratch. -- Existing exact-root, compaction, follower recovery, and source-loss suites - remain green. - -## Persist only verified directory acceleration - -### Keep authority out of the cache - -The authenticated radix directory is the canonical metadata structure. Its -incremental publisher already changes only affected leaves and ancestors, and -initial construction streams completed leaves. The remaining scale concern is -repeated remote metadata reads after the process-wide 8 MiB memory cache turns -over. - -The server may supply `crab-ltx::Host` with a caller-owned local directory-node -cache. The default host remains memory-only. These are two real adapters at the -existing host seam: ephemeral library use and server-managed persistent cache. - -A cache key binds: - -```text -backing store identity -+ Cell identity -+ incarnation -+ immutable object path -+ BLAKE3 digest -``` - -Cache installation uses an exclusive scratch file, file sync, atomic rename, -and parent-directory sync. Every hit is rehashed before decoding. A corrupt, -short, or missing entry is deleted or ignored and rebuilt from the authoritative -origin. It never causes root rollback or fallback to another format. - -The persistent cache is byte-bounded and evictable. Eviction changes latency, -not correctness. Backup and retention traversal continues to use uncached -origin reads where it must prove that remote dependencies still exist. - -### Measure before sharding locks - -The current in-memory cache uses one process-global mutex. The implementation -first adds contention telemetry and the 1,000/10,000-Cell workload. It shards -or replaces that lock only when measurements show material wait time. This -avoids adding a speculative concurrency structure. - -### Exit the metadata phase only with evidence - -- 1,000 and then 10,000 genuinely open Cells have bounded metadata RSS. -- Persistent cache disk use stays within its admitted share. -- Cold, memory-cache, and disk-cache lookup latency are reported separately. -- Cache corruption and eviction cannot change restored bytes. -- Remote reachability checks do not accept local presence as proof. - -## Make resident lifecycle explicit - -### Separate local residency from distributed control - -Cell control continues to use `Recovering`, `Serving`, `Idle`, and `Tombstoned`. -The runtime additionally tracks a local, non-serialized lifecycle for active -handles: - -```text -Cold -> Activating -> ActiveSparse -> ActiveResident - | | | - +-------+-> Quiescing -> Cold - | | | - +-------+-> Fenced <---+ -``` - -This local lifecycle does not create a second authority record. `Active` is -the common admission state represented by `ActiveSparse` and `ActiveResident`. -Either is valid only while authoritative Cell control still names the process -session and epoch. The distinction is performance-only: both use the same -SQLite writer, authority, root, and durability gates. - -### Let the actor own eviction eligibility - -The actor has the information required to decide whether a Cell is quiescent. -Router timers or an external LRU cannot independently prove safety. - -A Cell is not evictable while it has any of: - -- Accepted commands or queries. -- Tentative SQLite state. -- Pending object publication or ambiguous control transition. -- Fleet-covered data not yet object-covered or recovery-pinned. -- Live Workflow activity, Queue lease, effect delivery, or state stream. -- Migration or maintenance work. -- A scheduler deadline that cannot be transferred safely. - -When active admission is exhausted, the runtime selects the least-recently-used -eligible actor. No new configuration option is added. If no actor is eligible, -the request receives bounded capacity failure rather than forcing unsafe -eviction. - -### Drain in one order - -Eviction performs: - -1. Close new local admission. -2. Wait for accepted work and output gates. -3. Reconcile or publish every pending root. -4. Ensure follower-covered tails are object-covered or recovery-pinned. -5. Stop Cell-local activities and state streams through their existing - cancellation contracts. -6. Close SQLite and release its reservations. -7. CAS authoritative control to `Idle` only if this exact owner, epoch, and root - still win. -8. Remove mutable files for this activation. -9. Retain only independently verified immutable cache entries within budget. - -Failure before step seven leaves authority owned and the actor fenced or -recoverable. Failure after step seven cannot revive the closed local writer. - -### Define warm activation narrowly - -Warm activation means that verified immutable directory nodes and pages may be -available locally. It still acquires authority, opens the exact control-pinned -root, and creates a fresh sparse writable SQLite session. - -Do not call cache-warm activation a resident request. Cache warmth can reduce -activation I/O but cannot prove that every SQLite page needed by a request is -local. Only `ActiveResident` supports the zero-object-operation read claim. - -Reopening an old mutable SQLite session is outside this design. It would need a -separate crash-safe claim, clean-close marker, exact-root binding, and filesystem -qualification. Implement it only if the measured fresh sparse path remains a -material bottleneck after immutable caching. - -### Exit the lifecycle phase only with evidence - -- Sustained churn beyond active capacity produces bounded eviction and - activation rather than permanent exhaustion. -- Every acknowledged outcome survives eviction and reactivation. -- Ineligible actors are never selected under pressure. -- Fencing during every drain step prevents further output. -- Hot, immutable-cache-warm, and cold activation have separate measurements. -- Process restart treats ambiguous mutable files as quarantine, not authority. - -## Use one resource envelope - -The runtime already shares many admission facilities. Qualification must prove -that the total model covers every local consumer without omission or double -counting. - -| Consumer | Required accounting | -| --- | --- | -| SQLite main, WAL, and SHM | Active Cell disk and descriptors; the runtime ledger reserves eight descriptors per active Cell and exports used/capacity gauges | -| Retained captured LTX and checksum sidecar | Managed session disk | -| Sparse materialized pages | Incremental Cell disk | -| Directory-node and immutable-page cache | Evictable cache disk | -| Follower node-log tails | Follower disk budget | -| Verification, restore, and compaction scratch | Full-job scratch budget | -| Git, LFS, archive, and Release staging | Shared product local disk | -| SQLite page caches and actor state | Active Cell memory | -| Codec, dirty, recovery, and activity jobs | Shared job and memory credits | -| Background Cell hydration | Incremental disk, page I/O, object I/O, and job credits | -| Coordination simulator traces | Test-only bounded memory and artifact retention | - -Before a large operation reads remote bodies, it reserves its complete estimate -and remeasures physical free space against the node reserve. A failed operation -retains conservative accounting until installed files are reconciled or the -owning handle is discarded. - -The work adds no environment variable. Resource-derived server configuration -and `cells capacity --json --live` remain the operator interface. - -This phase also qualifies retention at scale: - -- Backup pins and in-flight backup advertisements race collection safely. -- Collection streams large inventories and bounds deletion batches. -- Current controls, retained releases, and every pin remain exact roots. -- Retired follower lanes disappear only after authority covers their epochs. -- Cross-provider export is implemented before it is advertised as a recovery - contract. - -## Qualify production behavior - -### Define claims before running load - -Every qualification run declares: - -- Source commit, release tag, image digest, and chart digest. -- Provider and object-store version/topology. -- Node CPU, memory, effective local disk, filesystem, and file-descriptor limit. -- SQLite page size and database size distribution. -- Number of genuinely open Cells. -- Transaction body size and pages changed per transaction. -- Read/write mix and hot-key distribution. -- Follower count and durability policy. -- Object-store latency and injected failure schedule. -- Target throughput and latency thresholds. - -Profile names do not imply a result. Small, medium, and large profiles receive -separate receipts. The 10,000-Cell or 1,000-mutation/s result is claimed only on -profiles that actually achieve it. - -### Run the capacity matrix - -| Stage | Workload | Required evidence | -| --- | --- | --- | -| Large database | 100 MiB and 5 GiB; compressible and incompressible | Exact source-loss recovery; bounded peak RSS and scratch | -| Residency | 1,000, 5,000, then 10,000 open Cells | Bounded RSS, threads, descriptors, cache, and SSD | -| Resident reads | Repeated read-only requests to fully hydrated local owners | Zero object-store calls; route, queue, SQL, and end-to-end latency percentiles | -| Aggregate writes | 1,000 mutations/s across eight or more Cells | At least 95% success, latency percentiles, exact replay | -| Hot Cell | One Cell under bounded concurrent writes | Queue bounds, logical/published-head lag, no starvation | -| Maintenance overlap | Hydration, compaction, backup, retention, scheduler | Foreground latency, bounded queues, no root rewind | -| Lifecycle churn | Working set exceeds active admission | Bounded eviction, activation, and authority movement | -| Fleet convergence | Add/remove nodes with uneven Cell costs and shifting hot sets | Weighted balance, bounded moves, no oscillation, declared p99 impact | -| Resource pressure | Oscillating cgroup memory, disk, descriptor, and job pressure | Hysteretic shedding, reserve recovery, no unsafe release | -| Takeover | Cold and immutable-cache-warm successors | Exact root, monotonic epoch and sequence, takeover time | - -### Inject the fault matrix - -- Kill the owner before and after SQLite commit. -- Kill after follower proof but before object publication. -- Lose an accepted control-CAS response. -- Delay or throttle immutable uploads. -- Partition peer management traffic without partitioning object storage. -- Expire the owner advertisement and reconnect the stale process later. -- Fill owner, follower, and scratch disk independently. -- Lose follower ACKs and restart followers simultaneously. -- Kill recovery after each overlay attachment and pinning step. -- Restart with empty owner-local SQLite and LTX directories. -- Roll between compatible releases while work continues. -- Enter maintenance with retained incompatible runtime work. -- Return stale or mixed-version placement samples during a rebalance. -- Kill a planned receiver before release, after release, and during restore. -- Oscillate cgroup memory around both pressure thresholds. -- Replay every historical simulator seed and crash each modeled await seam. - -After every fault, verify one authoritative owner, no stale-owner output, -monotonic sequence and root, stable replay, no lost acknowledged result, and no -leaked lease or reservation. - -### Retain these measurements - -- Process RSS and allocator peak. -- Descriptors and threads. -- Main, WAL, LTX, sparse, cache, follower, and scratch bytes. -- Object-store requests and bytes per command. -- Object-store requests by origin: route, page fault, publication, primitive, - and maintenance. -- WAL and LTX write amplification. -- Publication and follower backlog. -- Command p50, p95, and p99 latency. -- Local-route hit rate and lookup latency. -- Sparse hydration remaining bytes, throughput, pauses, and promotion time. -- Scheduler pass duration and due lag. -- Placement-sample age, planned and completed moves, rejected moves, pressure - tier, convergence error, and cooldown suppressions. -- Cold and warm takeover duration. -- Compaction throughput and peak scratch. -- Graceful drain and shutdown duration. - -Qualification fails on any correctness violation. It also fails if the process -swaps, exceeds its descriptor reserve, admits beyond disk capacity, allows a -scheduler pass above five seconds, or records less than 95% successful responses -at the configured 1,000-mutation/s target. - -Signed receipts bind the measurements to the exact source, image, Pod UID, -provider, profile, and completion time. Local unit tests and in-memory object -stores cannot substitute for these receipts. - -### Compare Crab and Celld fairly - -A comparative benchmark uses identical: - -- Hardware and filesystem. -- SQLite version, page size, and database distribution. -- Object-store implementation, placement, and injected latency. -- Mutation payload, pages changed, and read/write mix. -- Follower count and acknowledgement policy. -- Warmup, duration, and failure schedule. - -Exclude V8 and JavaScript handler execution from both measurements. Report -throughput, latency, resource use, recovery time, and write amplification. -Crab may claim an advantage only for a metric demonstrated under the matched -configuration; architectural expectations are not benchmark results. - -Compare architecture as well as headline throughput: - -- Run the same resident-read instrumentation and report bucket calls, sparse - page faults, and routing cost separately. -- Run the same seeded fault classes where both protocols expose the seam, and - publish non-equivalent assumptions rather than normalizing them away. -- Compare count-weighted placement with Crab's reservation- and demand-weighted - placement on heterogeneous Cells. -- Measure convergence, movement amplification, and foreground p99 during node - addition, node loss, cgroup pressure, and graceful drain. -- Record model and simulator coverage as assurance evidence, not as a runtime - performance score. - -## Standalone replication contract: hard removal recorded and executed - -Commit `4d097cce362` introduced the standalone replication interface and is -reachable from tag `v1.2.4`. That tag exported the epoch-head, standalone paging, -and standalone scheduling APIs. The repository maintainer explicitly authorized -hard removal on 2026-09-18 for the current unreleased breaking change (or the -next breaking release if this branch is cut into a release). The full evidence -record is [the standalone replication audit](standalone-replication-audit.md). - -The crate remains `publish = false`, but tagged source, docs, and examples were -treated as potentially shipped. The audit found no workspace production caller; -unknown external usage is handled by an explicit offline migration boundary, -not by keeping a parallel runtime path. - -### Preserve proof before removal - -Before physical deletion, move still-relevant evidence to the canonical path: - -| Standalone evidence | Canonical owner after migration | -| --- | --- | -| Mutable-head CAS race and lost response | `CellAuthority` and `CellPublisher` | -| Exact source-loss restore | `CellReplica` and actor takeover tests | -| Bundle corruption and no fallback | Recovery overlay and node-log recovery tests | -| Sparse partial writes and delayed faults | `CellWritableDatabase` and Cell VFS tests | -| Destination admission before I/O | `CellReplica` restore/compaction tests | -| Range-compaction byte bounds | Cell exact-root compaction tests | -| Provider worker lifetime | Shared `Host` and Cell paged-I/O tests | -| RustFS round trip | Runtime and server source-loss qualification | - -Index encoding, decoding, frame verification, and index validation remain in the -private authenticated-index module used by `CellReplica`. The public standalone -facade and its page-map implementation are gone. - -### Delete only standalone ownership - -The authorized hard-removal change deletes: - -- `Replica` and `ReplicaHead`. -- Standalone epoch-head layout and publication. -- `CompactionSchedule` for standalone heads. -- Standalone read-only `PagedDatabase` and `PagedConnection`. -- Standalone-only `Db` helpers. -- Standalone examples, documentation, and tests after evidence migration. -- The standalone branch of `paged_io`; the retained bridge has one Cell variant. - -Retain: - -- `Db`, WAL capture, snapshots, and checkpoints. -- LTX codecs, checksums, and exact local recovery. -- `Host`, filesystem, executor, disk, dirty, recovery, and scratch admission. -- `bundle::Bundle` and node-log recovery use. -- Private authenticated-index and frame utilities. -- Cell sparse VFS, hydration, roots, directory, compaction, and recovery overlay. - -The existing `replica` Cargo feature also enables required Cell remote -mechanics, so its name remains. No feature alias is added. - -Old standalone object prefixes are never reinterpreted as Cell roots. If stored -standalone data needs migration, an operator must use an explicit offline -export/import tool with exact-root verification; this change does not delete -remote data and adds no production fallback reader. - -## Deliver in dependency order - -Each change is independently reviewable and leaves one canonical path. - -| Slice | Work | Depends on | Exit evidence | -| --- | --- | --- | --- | -| 1 | Correct ownership docs and add server dependency guard | None | Architecture check rejects production `crab_ltx` imports | -| 2 | Extract the pure coordination kernel without changing effects | 1 | Production adapter parity tests preserve current behavior | -| 3 | Add seeded simulation, broken variants, and trace replay | 2 | Historical seeds replay; each broken variant is detected | -| 4 | Add the small-state TLA+ model and delta ledger | 2 | Positive and negative verdict matrix is pinned | -| 5 | Add actor-owned resident lookup before remote metadata | 2 | Local route has zero catalog/control object reads and fences exactly | -| 6 | Wire bounded background hydration and resident promotion | 5 | Qualified resident reads perform zero object-store calls | -| 7 | Stream native captured-segment verification and upload | 2 | Large native append has bounded RSS and exact recovery | -| 8 | Stream bundle ranges and remove complete-body copies | 7 | Implemented file-backed reopen, exact row reads, staged CAS upload, and recovery tests; 5 GiB RSS/low-disk qualification remains | -| 9 | Persist verified directory-node acceleration | 7 | Cache corruption/eviction tests and bounded residency | -| 10 | Add actor-owned quiescing, idle eviction, and full resource summaries | 5, 6, 9 | Churn test preserves every acknowledged root | -| 11 | Add signed live placement observations and the pure weighted planner | 2, 10 | Mixed-version gate and deterministic plan tests pass | -| 12 | Add hysteretic pressure shedding and paced release/activation | 11 | Pressure and membership tests converge without unsafe movement | -| 13 | Reconcile node-wide disk and job accounting | 7, 10, 12 | Capacity report matches measured local consumers | -| 14 | Complete simulator, fault, capacity, latency, and balancing receipts | 3 through 13 | Signed provider/profile matrix passes | -| 15 | Record standalone support decision | 14 | Named HARD REMOVE decision and offline migration limit | -| 16 | Port unique standalone evidence | 15 | Canonical tests cover every retained invariant | -| 17 | Remove standalone module | 16 | Public surface and documentation match the decision | - -The streaming work and standalone deletion are now separate reviewable commits; -canonical Cell tests own the retained invariants, and no standalone test owner -or compatibility facade remains. - -Slice 1 implementation evidence: `make architecture-check` now includes an -explicit Cell composition guard. It verifies that `crab-http-server` keeps -`crab-ltx` dev-only and rejects direct `crab_ltx::` use in production source, -while admitting `#[cfg(test)]` modules and dedicated `tests.rs` fixtures. The -guard's temporary-tree regressions live in -`crab/scripts/test_check_architecture_gates.py` under -`CellRuntimeBoundaryTests`; the same gate rejects the retired standalone LTX -epoch-head symbols while admitting Cell-scoped names. Existing actor/publication tests already cover the -required acknowledgement, fence, shutdown, and lost-CAS characterization -cases; no duplicate runtime tests were added. Local Cargo qualification now -passes on the required external workspace target volume; provider, Kubernetes, -and multi-GiB evidence remains pending. - -Slices 2–6 now have reviewable seams: `coordination.rs` owns the volatile -admission/fence/publication/migration/shutdown decisions; the actor has one -schedule adapter that passes queue/publisher, lease, and publication-pressure -observations and the kernel centrally decides dispatch, wait, deactivation, or -fence, so `actor.rs` does not duplicate those predicates. Work, migration, -publication, renewal, and stale-hydration completions also carry their fence -observation into one kernel transition; the actor only maps the returned `Fence` -decision to cleanup, while pending commands remain busy until proof. The -schedule transition also distinguishes `ReadyToDeactivateFenced` from a normal -live drain, so release-path selection does not re-read actor lifecycle state; -background hydration and renewal use the same queue/publication/lease -observation boundary; the test-only -simulator replays fixed seeds and checks acknowledgement/publication invariants; -the pinned TLC runner has positive and deliberately broken configurations; and -`CellRuntime::resident_handle` is attempted before catalog/control reads while -bounded worker hydration promotes only verified sparse roots. Slices 10–12 also -have pure resource, eviction, placement, pressure, and movement-budget -contracts, with the runtime/SQL/hydration/primitive-job ledger, stale-session -restart inventory, actor-owned movement seam, and bounded transport-codec -admission wired; complete advertised/metric parity, cold-placement execution, - and multi-process qualification remain open. The version-4 receipt matrix - is versioned and size-bounded, with signing, profile/threshold binding, - artifact binding, fault/artifact/ownership evidence, runner emission, and - exact source/image release binding implemented. - -## Verify every slice - -Use one worktree-specific external Cargo target directory. - -```bash -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-main \ - cargo test -p crab-ltx --locked - -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-main \ - cargo test -p crab-ltx --features replica --locked - -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-main \ - cargo test -p crab-cell-runtime --locked - -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-main \ - cargo test -p crab-http-server --locked - -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-main \ - cargo clippy -p crab-ltx -p crab-cell-runtime -p crab-http-server \ - --all-targets --locked -- -D warnings - -node crates/crab-cell-runtime/docs/validate.mjs -``` - -After slices 3 and 4, the same verification entry point also runs the pinned -simulation corpus and the fast TLC configuration. The model directory exposes -one checksum-verifying script so local and CI runs use the same TLC build. The -scheduled broad simulator and model jobs archive seeds, traces, counterexamples, -and the source revision; they are not hidden behind a retry-until-green loop. - -Compilation and unit tests are necessary but not production qualification. Run -the real RustFS, Compose, Kubernetes, provider, and signed-receipt gates described -in [delivery.md](delivery.md) before changing readiness claims. - -## Reject these shortcuts - -- Raising `Limits` without removing resident allocations. -- Treating catalog entries as active Cells in capacity reports. -- Using local cache presence as remote-retention proof. -- Reopening an ambiguous local database instead of the authoritative root. -- Letting eviction policy inspect less state than the actor. -- Calling a sparse or cache-warm activation fully resident. -- Claiming zero bucket operations while excluding router or page-fault calls - from instrumentation. -- Caching catalog or control as a substitute for the node lease, actor - admission capability, or authority CAS. -- Copying production transition rules into a simulator-only state machine. -- Accepting a model that cannot detect deliberately broken protocol variants. -- Moving hot writers or releasing ownership before actor quiescence and - durability proof. -- Balancing only by Cell count when declared reservations differ materially. -- Reacting to one pressure sample without a deadband, cooldown, or movement - budget. -- Adding another configuration mode for standalone versus Cell publication. -- Keeping aliases or dual readers after an approved hard cut. -- Deleting standalone tests before moving their unique proof. -- Comparing Crab and Celld with different durability or object-store conditions. - -## Declare completion precisely - -The canonical scaling design is complete only when: - -1. Production and seeded simulation use the same coordination kernel, broken - variants fail, and the TLA+ verdict matrix covers the declared protocol. -2. Publication, metadata, hydration, activation, and eviction satisfy their - bounded tests. -3. A qualified warm resident read performs zero object-store operations, and - its matched latency receipt reports every layer of the request. -4. Weighted placement, pressure shedding, and paced drains converge without - unsafe ownership movement or foreground latency beyond the declared bound. -5. Every acknowledged result survives the complete owner and disk-loss matrix. -6. Signed receipts establish the claimed Cell count, throughput, latency, and - resource envelope for each advertised profile and provider. -7. Backup, retention, follower retirement, release rollout, and shutdown pass - while load continues. -8. The standalone contract has a recorded HARD REMOVE decision and offline - migration limit. -9. The removal retains shared mechanics and canonical proof without an - epoch-head fallback. -10. Documentation describes measured results rather than target capacity. - -Until those conditions hold, describe Crab Cell Runtime as functionally complete -but not production-qualified at the target scale. diff --git a/crates/crab-cell-runtime/docs/contracts/blob.sql b/crates/crab-cell-runtime/docs/contracts/blob.sql deleted file mode 100644 index b31af55ad..000000000 --- a/crates/crab-cell-runtime/docs/contracts/blob.sql +++ /dev/null @@ -1,37 +0,0 @@ -CREATE TABLE blob_uploads ( - upload_id BLOB PRIMARY KEY CHECK (length(upload_id) = 16), - object_key BLOB NOT NULL CHECK (length(object_key) BETWEEN 1 AND 1024), - request_digest BLOB NOT NULL CHECK (length(request_digest) = 32), - condition INTEGER NOT NULL CHECK (condition BETWEEN 0 AND 2), - expected_etag BLOB CHECK (expected_etag IS NULL OR length(expected_etag) = 32), - content_type TEXT CHECK (content_type IS NULL OR length(content_type) BETWEEN 1 AND 256), - metadata BLOB NOT NULL CHECK (length(metadata) <= 8192), - created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0), - expires_at_ms INTEGER NOT NULL CHECK (expires_at_ms > created_at_ms), - completed INTEGER NOT NULL CHECK (completed IN (0, 1)), - etag BLOB CHECK (etag IS NULL OR length(etag) = 32), - size INTEGER NOT NULL CHECK (size >= 0), - part_count INTEGER NOT NULL CHECK (part_count >= 0) -) STRICT, WITHOUT ROWID; -CREATE INDEX blob_upload_expiry ON blob_uploads(expires_at_ms, upload_id); - -CREATE TABLE blob_parts ( - upload_id BLOB NOT NULL REFERENCES blob_uploads(upload_id) ON DELETE CASCADE, - part_number INTEGER NOT NULL CHECK (part_number BETWEEN 1 AND 4096), - digest BLOB NOT NULL CHECK (length(digest) = 32), - size INTEGER NOT NULL CHECK (size BETWEEN 0 AND 262144), - byte_offset INTEGER CHECK (byte_offset IS NULL OR byte_offset >= 0), - PRIMARY KEY (upload_id, part_number) -) STRICT, WITHOUT ROWID; - -CREATE TABLE blob_objects ( - object_key BLOB PRIMARY KEY CHECK (length(object_key) BETWEEN 1 AND 1024), - upload_id BLOB NOT NULL UNIQUE REFERENCES blob_uploads(upload_id), - etag BLOB NOT NULL CHECK (length(etag) = 32), - size INTEGER NOT NULL CHECK (size >= 0), - part_count INTEGER NOT NULL CHECK (part_count BETWEEN 1 AND 4096), - content_type TEXT, - metadata BLOB NOT NULL, - created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0), - updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= created_at_ms) -) STRICT, WITHOUT ROWID; diff --git a/crates/crab-cell-runtime/docs/contracts/cron.sql b/crates/crab-cell-runtime/docs/contracts/cron.sql deleted file mode 100644 index 32d7a778f..000000000 --- a/crates/crab-cell-runtime/docs/contracts/cron.sql +++ /dev/null @@ -1,13 +0,0 @@ -CREATE TABLE cron_schedules ( - schedule_id BLOB PRIMARY KEY CHECK (length(schedule_id) = 16), - target_index INTEGER NOT NULL CHECK (target_index >= 0), - target_partition BLOB NOT NULL CHECK (length(target_partition) <= 1024), - payload BLOB NOT NULL CHECK (length(payload) <= 262144), - interval_ms INTEGER NOT NULL CHECK (interval_ms BETWEEN 1000 AND 31536000000), - next_due_ms INTEGER NOT NULL CHECK (next_due_ms >= 0), - occurrence INTEGER NOT NULL CHECK (occurrence >= 0), - enabled INTEGER NOT NULL CHECK (enabled IN (0, 1)), - generation INTEGER NOT NULL CHECK (generation >= 1), - updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= 0) -) STRICT, WITHOUT ROWID; -CREATE INDEX cron_due ON cron_schedules(enabled, next_due_ms, schedule_id); diff --git a/crates/crab-cell-runtime/docs/contracts/kv.sql b/crates/crab-cell-runtime/docs/contracts/kv.sql deleted file mode 100644 index 0333db5f9..000000000 --- a/crates/crab-cell-runtime/docs/contracts/kv.sql +++ /dev/null @@ -1,11 +0,0 @@ --- KV schema version 1. One namespace per Cell; scope is part of each key. -CREATE TABLE kv_entries ( - scope BLOB NOT NULL, - key BLOB NOT NULL CHECK (length(key) BETWEEN 1 AND 1024), - version BLOB NOT NULL CHECK (length(version) = 28), - value BLOB NOT NULL CHECK (length(value) <= 4194304), - expires_at_ms INTEGER, - PRIMARY KEY (scope, key) -) STRICT, WITHOUT ROWID; -CREATE INDEX kv_expiry ON kv_entries(expires_at_ms) - WHERE expires_at_ms IS NOT NULL; diff --git a/crates/crab-cell-runtime/docs/contracts/peer.proto b/crates/crab-cell-runtime/docs/contracts/peer.proto deleted file mode 100644 index 7233ecb6b..000000000 --- a/crates/crab-cell-runtime/docs/contracts/peer.proto +++ /dev/null @@ -1,186 +0,0 @@ -syntax = "proto3"; -package crab.cell.peer.v1; - -// Private messages for enrolled Crab nodes, not a public gRPC service. -// Rust message types are generated privately inside crab-cell-runtime. -message PeerAuthorization { - bytes origin_session = 1; // 16 bytes; enrollment binds the signing public key. - string principal_issuer = 2; - string principal_subject = 3; - repeated string actions = 4; // Sorted unique compiled action identifiers. - bytes release_digest = 5; // 32 bytes. - int64 issued_at_ms = 6; - int64 expires_at_ms = 7; - bytes payload_digest = 8; // BLAKE3 of exact nested operation bytes. - bytes signature = 9; // Ed25519; 64 bytes. -} -message PeerRequest { - uint32 version = 1; // Exactly 1. - PeerAuthorization authorization = 2; - uint32 hop_count = 3; // 1..2 on wire; local calls do not forward. - uint32 remaining_ms = 4; // 1..60000; never extend the signed expiry. - oneof operation { - MutationRequest mutate = 10; - ReadRequest read = 11; - ResolveRequest resolve = 12; - EffectRequest deliver_effect = 13; - EffectResolveRequest resolve_effect = 14; - MigrationRequest migrate = 15; - } -} -message PeerReply { - oneof outcome { - MutationReply mutation = 1; - ReadReply read = 2; - ResolveReply resolve = 3; - Error error = 4; - MigrationReply migration = 5; - } -} - -message EffectIdentity { - bytes effect_id = 1; - bytes source_cell = 2; - bytes source_incarnation = 3; - uint64 source_sequence = 4; - uint32 ordinal = 5; - int64 expires_at_ms = 6; -} -message EffectRequest { - Target target = 1; - bytes destination_incarnation = 2; - EffectIdentity identity = 3; - oneof operation { - // Native primitives cross the peer boundary through the registered typed - // command codec. There is no second primitive-specific wire path. - CellCommand cell_command = 10; - } - reserved 11 to 17; -} -message EffectResolveRequest { - Target target = 1; - bytes destination_incarnation = 2; - EffectIdentity identity = 3; - bytes operation_digest = 4; -} - -// Release control request. The receiver derives SQL from its frozen registry; -// callers can only bind the exact source and expected successor versions. -message MigrationRequest { - Target target = 1; - bytes incarnation = 2; - bytes from_code = 3; - uint32 from_schema = 4; - bytes to_code = 5; - uint32 to_schema = 6; -} -message MigrationReply { CellDescription description = 1; } - -message Target { - bytes tenant_id = 1; - bytes application_id = 2; - bytes namespace_id = 3; - bytes partition = 4; -} -message MutationIdentity { - bytes request_id = 1; // 16 bytes; immutable across retries. - bytes incarnation = 2; // 16 bytes; obtained by Read or provisioning. - int64 issued_at_ms = 3; - int64 expires_at_ms = 4; -} -// Native primitive payloads are encoded by the registered module codec. -message CellCommand { uint32 command_id = 1; uint32 codec_version = 2; bytes input = 3; } -message MutationRequest { - Target target = 1; - MutationIdentity identity = 2; - uint32 timeout_ms = 3; - CellDescription expected = 4; // Required; checked against the active owner before execution. - oneof operation { - CellCommand cell_command = 10; - } - reserved 11 to 20; -} -message Receipt { - bytes cell_id = 1; - bytes incarnation = 2; - uint64 commit_sequence = 3; -} -message Error { - enum Code { - INVALID = 0; - INVALID_ARGUMENT = 1; - PERMISSION_DENIED = 2; - NOT_FOUND = 3; - PRECONDITION_FAILED = 4; - REQUEST_ID_CONFLICT = 5; - REQUEST_EXPIRED = 6; - LEASE_LOST = 7; - RESOURCE_EXHAUSTED = 8; - OUTCOME_UNKNOWN = 9; - UNAVAILABLE = 10; - SCHEMA_INCOMPATIBLE = 11; - INTERNAL = 12; - REPLICA_BEHIND = 13; - REPLICA_UNAVAILABLE = 14; - } - enum Outcome { UNSPECIFIED = 0; NOT_STARTED = 1; REJECTED = 2; UNKNOWN = 3; } - Code code = 1; - Outcome outcome = 2; - string message = 3; - uint32 retry_after_ms = 4; - bytes application_details = 5; -} -message MutationResult { - oneof result { - bytes command_output = 1; - } - reserved 2 to 7; -} -message MutationReply { - Receipt receipt = 1; // Present for a durable success or stored rejection. - oneof outcome { MutationResult result = 2; Error error = 3; } -} -message CellQuery { uint32 query_id = 1; uint32 codec_version = 2; bytes input = 3; } -message ReadRequest { - Target target = 1; - uint32 timeout_ms = 2; - Receipt minimum = 3; - CellDescription expected = 4; // Required for CellQuery and ReplicaQuery; Describe obtains this observation. - oneof operation { - bool describe = 10; - CellQuery cell_query = 15; - bool replica_activate = 16; // Advisory; receiver verifies policy and current owner. - CellQuery replica_query = 17; // Explicit snapshot-consistent read. - bool replica_reconcile = 18; // Advisory owner wake-up after policy CAS. - bool replica_status = 19; // Fresh readiness check; never activates a view. - } - reserved 11 to 14; -} -message CellDescription { - bytes cell_id = 1; - bytes incarnation = 2; - bytes code = 3; - uint32 schema = 4; -} -message ReadReply { - Receipt receipt = 1; - oneof result { - CellDescription description = 2; - bytes command_output = 6; - Error error = 7; - bool replica_ready = 8; // Exact local snapshot; false means warm only, never readable. - bool replica_reconciled = 9; // Owner accepted a reconciliation hint. - } - reserved 3 to 5; -} -message ResolveRequest { - Target target = 1; - MutationIdentity identity = 2; - bytes operation_digest = 3; - CellDescription expected = 4; -} -message ResolveReply { - enum State { INVALID = 0; COMMITTED = 1; REJECTED = 2; ABSENT = 3; UNKNOWN = 4; EXPIRED = 5; } - State state = 1; - MutationReply reply = 2; -} diff --git a/crates/crab-cell-runtime/docs/contracts/queue.sql b/crates/crab-cell-runtime/docs/contracts/queue.sql deleted file mode 100644 index fd6e2908d..000000000 --- a/crates/crab-cell-runtime/docs/contracts/queue.sql +++ /dev/null @@ -1,78 +0,0 @@ --- Queue schema version 1. state: ready=0, leased=1, acked=2, dead=3. -CREATE TABLE queue_messages ( - message_id BLOB PRIMARY KEY CHECK (length(message_id) = 16), - payload BLOB NOT NULL CHECK (length(payload) <= 262144), - state INTEGER NOT NULL CHECK (state BETWEEN 0 AND 3), - attempt INTEGER NOT NULL CHECK (attempt >= 0), - due_at_ms INTEGER NOT NULL, - expires_at_ms INTEGER NOT NULL, - token BLOB, - lease_until_ms INTEGER, - result_code INTEGER, - dead_letter_effect_id BLOB CHECK (dead_letter_effect_id IS NULL OR length(dead_letter_effect_id) = 32), - CHECK ((state = 1 AND token IS NOT NULL AND length(token) = 16 AND lease_until_ms IS NOT NULL) - OR (state != 1 AND token IS NULL AND lease_until_ms IS NULL)), - CHECK (dead_letter_effect_id IS NULL OR state = 3) -) STRICT, WITHOUT ROWID; -CREATE INDEX queue_ready ON queue_messages(state, due_at_ms, message_id); -CREATE INDEX queue_attempts ON queue_messages(state, attempt); -CREATE INDEX queue_leases ON queue_messages(state, lease_until_ms, message_id); -CREATE INDEX queue_retention ON queue_messages(expires_at_ms); - -CREATE TABLE queue_dedup ( - producer_id BLOB PRIMARY KEY CHECK (length(producer_id) = 16), - payload_digest BLOB NOT NULL CHECK (length(payload_digest) = 32), - message_id BLOB NOT NULL CHECK (length(message_id) = 16), - retain_until_ms INTEGER NOT NULL -) STRICT, WITHOUT ROWID; -CREATE INDEX queue_dedup_expiry ON queue_dedup(retain_until_ms); - -CREATE TABLE queue_control ( - singleton INTEGER PRIMARY KEY CHECK (singleton = 1), - paused INTEGER NOT NULL CHECK (paused IN (0, 1)), - generation INTEGER NOT NULL CHECK (generation >= 0), - updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= 0), - ready_count INTEGER NOT NULL CHECK (ready_count >= 0), - leased_count INTEGER NOT NULL CHECK (leased_count >= 0), - acked_count INTEGER NOT NULL CHECK (acked_count >= 0), - dead_count INTEGER NOT NULL CHECK (dead_count >= 0) -) STRICT; -INSERT INTO queue_control( - singleton, paused, generation, updated_at_ms, - ready_count, leased_count, acked_count, dead_count -) -VALUES (1, 0, 0, 0, 0, 0, 0, 0); - -CREATE TRIGGER queue_messages_count_insert -AFTER INSERT ON queue_messages -BEGIN - UPDATE queue_control - SET ready_count = ready_count + (NEW.state = 0), - leased_count = leased_count + (NEW.state = 1), - acked_count = acked_count + (NEW.state = 2), - dead_count = dead_count + (NEW.state = 3) - WHERE singleton = 1; -END; - -CREATE TRIGGER queue_messages_count_delete -AFTER DELETE ON queue_messages -BEGIN - UPDATE queue_control - SET ready_count = ready_count - (OLD.state = 0), - leased_count = leased_count - (OLD.state = 1), - acked_count = acked_count - (OLD.state = 2), - dead_count = dead_count - (OLD.state = 3) - WHERE singleton = 1; -END; - -CREATE TRIGGER queue_messages_count_state_update -AFTER UPDATE OF state ON queue_messages -WHEN OLD.state <> NEW.state -BEGIN - UPDATE queue_control - SET ready_count = ready_count - (OLD.state = 0) + (NEW.state = 0), - leased_count = leased_count - (OLD.state = 1) + (NEW.state = 1), - acked_count = acked_count - (OLD.state = 2) + (NEW.state = 2), - dead_count = dead_count - (OLD.state = 3) + (NEW.state = 3) - WHERE singleton = 1; -END; diff --git a/crates/crab-cell-runtime/docs/contracts/runtime.sql b/crates/crab-cell-runtime/docs/contracts/runtime.sql deleted file mode 100644 index 8159433a4..000000000 --- a/crates/crab-cell-runtime/docs/contracts/runtime.sql +++ /dev/null @@ -1,58 +0,0 @@ --- Runtime schema version 1. Install in every Cell before its primitive schema. -PRAGMA foreign_keys = ON; - -CREATE TABLE sys_meta ( - singleton INTEGER PRIMARY KEY CHECK (singleton = 1), - cell_id BLOB NOT NULL CHECK (length(cell_id) = 32), - incarnation BLOB NOT NULL CHECK (length(incarnation) = 16), - commit_sequence INTEGER NOT NULL CHECK (commit_sequence >= 0), - logical_time_ms INTEGER NOT NULL CHECK (logical_time_ms >= 0), - schema_version INTEGER NOT NULL CHECK (schema_version >= 1) -) STRICT; - -CREATE TABLE sys_requests ( - request_id BLOB PRIMARY KEY CHECK (length(request_id) = 16), - operation_digest BLOB NOT NULL CHECK (length(operation_digest) = 32), - outcome INTEGER NOT NULL CHECK (outcome IN (1, 2)), - result BLOB NOT NULL, - commit_sequence INTEGER NOT NULL CHECK (commit_sequence > 0), - expires_at_ms INTEGER NOT NULL, - retain_until_ms INTEGER NOT NULL CHECK (retain_until_ms >= expires_at_ms) -) STRICT, WITHOUT ROWID; -CREATE INDEX sys_requests_expiry ON sys_requests(retain_until_ms); - -CREATE TABLE sys_inbox ( - effect_id BLOB PRIMARY KEY CHECK (length(effect_id) = 32), - operation_digest BLOB NOT NULL CHECK (length(operation_digest) = 32), - outcome INTEGER NOT NULL CHECK (outcome IN (1, 2)), - result BLOB NOT NULL, - commit_sequence INTEGER NOT NULL, - expires_at_ms INTEGER NOT NULL, - retain_until_ms INTEGER NOT NULL CHECK (retain_until_ms >= expires_at_ms) -) STRICT, WITHOUT ROWID; -CREATE INDEX sys_inbox_expiry ON sys_inbox(retain_until_ms); - -CREATE TABLE sys_effects ( - effect_id BLOB PRIMARY KEY CHECK (length(effect_id) = 32), - destination BLOB NOT NULL CHECK (length(destination) = 32), - operation BLOB NOT NULL, - state INTEGER NOT NULL CHECK (state IN (0, 1, 2, 3)), - attempt INTEGER NOT NULL CHECK (attempt >= 0), - due_at_ms INTEGER NOT NULL, - expires_at_ms INTEGER NOT NULL CHECK (expires_at_ms >= due_at_ms), - token BLOB, - lease_until_ms INTEGER, - created_sequence INTEGER NOT NULL, - result BLOB, - CHECK ((state = 1 AND token IS NOT NULL AND length(token) = 16 AND lease_until_ms IS NOT NULL) - OR (state != 1 AND token IS NULL AND lease_until_ms IS NULL)) -) STRICT, WITHOUT ROWID; -CREATE INDEX sys_effects_due ON sys_effects(state, due_at_ms); -CREATE INDEX sys_effects_leases ON sys_effects(state, lease_until_ms); -CREATE INDEX sys_effects_expiry ON sys_effects(expires_at_ms); - -CREATE TABLE sys_migrations ( - version INTEGER PRIMARY KEY CHECK (version > 0), - digest BLOB NOT NULL CHECK (length(digest) = 32), - applied_sequence INTEGER NOT NULL CHECK (applied_sequence >= 0) -) STRICT; diff --git a/crates/crab-cell-runtime/docs/contracts/workflow.sql b/crates/crab-cell-runtime/docs/contracts/workflow.sql deleted file mode 100644 index 4b5facc1d..000000000 --- a/crates/crab-cell-runtime/docs/contracts/workflow.sql +++ /dev/null @@ -1,82 +0,0 @@ --- Workflow schema version 1. All identity is qualified by run_id. -CREATE TABLE workflow_runs ( - workflow_id BLOB PRIMARY KEY CHECK (length(workflow_id) BETWEEN 1 AND 1024), - run_id BLOB NOT NULL UNIQUE CHECK (length(run_id) = 16), - definition_digest BLOB NOT NULL CHECK (length(definition_digest) = 32), - status INTEGER NOT NULL CHECK (status BETWEEN 0 AND 4), - state BLOB NOT NULL CHECK (length(state) <= 1048576), - event_sequence INTEGER NOT NULL CHECK (event_sequence >= 0), - result BLOB, - completed_at_ms INTEGER -) STRICT, WITHOUT ROWID; - -CREATE TABLE workflow_events ( - run_id BLOB NOT NULL REFERENCES workflow_runs(run_id), - sequence INTEGER NOT NULL CHECK (sequence > 0), - event_id BLOB NOT NULL CHECK (length(event_id) = 32), - event_digest BLOB NOT NULL CHECK (length(event_digest) = 32), - payload BLOB NOT NULL, - PRIMARY KEY (run_id, sequence), - UNIQUE (run_id, event_id) -) STRICT, WITHOUT ROWID; - -CREATE TABLE workflow_control ( - singleton INTEGER PRIMARY KEY CHECK (singleton = 1), - event_count INTEGER NOT NULL CHECK (event_count >= 0) -) STRICT; -INSERT INTO workflow_control(singleton, event_count) VALUES (1, 0); - -CREATE TRIGGER workflow_events_count_insert -AFTER INSERT ON workflow_events -BEGIN - UPDATE workflow_control - SET event_count = event_count + 1 - WHERE singleton = 1; -END; - -CREATE TRIGGER workflow_events_count_delete -AFTER DELETE ON workflow_events -BEGIN - UPDATE workflow_control - SET event_count = event_count - 1 - WHERE singleton = 1; -END; - -CREATE TABLE workflow_activities ( - run_id BLOB NOT NULL REFERENCES workflow_runs(run_id), - activity_id BLOB NOT NULL CHECK (length(activity_id) = 16), - activity_type TEXT NOT NULL, - input BLOB NOT NULL CHECK (length(input) <= 262144), - state INTEGER NOT NULL CHECK (state BETWEEN 0 AND 4), - attempt INTEGER NOT NULL CHECK (attempt >= 0), - due_at_ms INTEGER NOT NULL, - expires_at_ms INTEGER NOT NULL, - token BLOB, - lease_until_ms INTEGER, - completion_token BLOB, - completion_digest BLOB, - result BLOB, - PRIMARY KEY (run_id, activity_id), - CHECK ((state = 1 AND token IS NOT NULL AND length(token) = 16 AND lease_until_ms IS NOT NULL) - OR (state != 1 AND token IS NULL AND lease_until_ms IS NULL)), - CHECK ((completion_token IS NULL AND completion_digest IS NULL) - OR (completion_token IS NOT NULL AND completion_digest IS NOT NULL - AND length(completion_token) = 16 AND length(completion_digest) = 32)) -) STRICT, WITHOUT ROWID; -CREATE INDEX activities_ready - ON workflow_activities(activity_type, state, due_at_ms, run_id, activity_id); -CREATE INDEX activities_due - ON workflow_activities(state, due_at_ms, run_id, activity_id); -CREATE INDEX activities_leases ON workflow_activities(state, lease_until_ms); -CREATE INDEX activities_expiry ON workflow_activities(expires_at_ms); - -CREATE TABLE workflow_timers ( - run_id BLOB NOT NULL REFERENCES workflow_runs(run_id), - timer_id BLOB NOT NULL CHECK (length(timer_id) = 16), - due_at_ms INTEGER NOT NULL, - state INTEGER NOT NULL CHECK (state IN (0, 1, 2)), - PRIMARY KEY (run_id, timer_id) -) STRICT, WITHOUT ROWID; -CREATE INDEX timers_due ON workflow_timers(state, due_at_ms, run_id, timer_id); -CREATE INDEX workflow_retention ON workflow_runs(completed_at_ms) - WHERE status BETWEEN 1 AND 3; diff --git a/crates/crab-cell-runtime/docs/delivery.md b/crates/crab-cell-runtime/docs/delivery.md deleted file mode 100644 index 7a874ed18..000000000 --- a/crates/crab-cell-runtime/docs/delivery.md +++ /dev/null @@ -1,664 +0,0 @@ -# Verify and qualify the Cell runtime - -The runtime implementation covers identity, authority, SQLite execution, LTX publication, primitives, release control, private peers, and repository adapters. Production qualification still requires measured capacity and real multi-Pod network fault evidence. - -| Document intent | Value | -| --- | --- | -| Content type | Verification plan | -| Audience | Implementers, reviewers, and release engineers | -| Goal | Map every runtime boundary to executable evidence and remaining release gates | - -[Back to the Cell runtime index](README.md) - -## Read the implementation map - -Each layer has one owner and one primary evidence surface. - -| Boundary | Primary source | Evidence | -| --- | --- | --- | -| Identity and Cell derivation | `src/identity.rs` | `src/identity.rs` tests, `tests/contracts/application.rs`, `tests/runtime/catalog.rs` | -| Control CAS and transitions | `src/control.rs`, `src/control/authority.rs` | authority and actor tests | -| SQLite command ledger | `src/cell/executor.rs`, `src/cell/schema.rs` | `tests/runtime/lifecycle.rs`, `tests/runtime/migration.rs` | -| Fixed SQL workers | `src/cell/worker.rs` | `tests/runtime/workers.rs` | -| Publication and exact-root recovery | `src/publication.rs`, `crab-ltx` | `tests/runtime/publication.rs`, `crab-ltx/tests/cell/roots.rs` | -| Catalog | `src/cell/catalog.rs` | `tests/runtime/catalog.rs` | -| Registry and codecs | `src/registry/`, `src/codec.rs` | `tests/contracts/registry.rs`, `tests/contracts/codec.rs` | -| Typed client and peer dispatch | `src/client.rs`, `src/peer.rs` | `tests/protocol/client.rs`, peer unit tests | -| SQL, KV, Blob, Queue, Cron, Workflow | `src/primitives/sql.rs`, `src/primitives/kv.rs`, `src/primitives/blob.rs`, `src/primitives/queue.rs`, `src/primitives/cron.rs`, `src/primitives/workflow.rs` | matching integration tests | -| Effects and activities | `src/primitives/effects.rs`, `src/primitives/activity_pool.rs` | `src/primitives/effects/tests.rs`, `src/primitives/activity_pool/tests.rs`, `tests/primitives/workflow.rs` | -| Scheduler | `src/fleet/scheduler.rs`, `src/primitives/maintenance.rs` | `tests/runtime/scheduler.rs` | -| Release control | `src/recovery/release.rs`, `src/recovery/release_progress.rs` | release unit tests and server command tests | -| Backup pins | `src/recovery/backup.rs`, `crab-ltx::CellReplica::reachable_objects` | runtime pin tests and server create/verify command tests | -| Immutable retention | `src/recovery/retention.rs`, `crab-storage::Store::list_stream` | mark/sweep tests and server maintenance-fence tests | -| Follower mechanics | `src/follower.rs`, `src/node/log.rs`, `src/node/log_recovery.rs` | verified-frame, object-covered queued-prefix, lost-ACK suffix, torn-tail, dual-proof, and seal/gather tests | -| State-observing streams | `src/client.rs`; `crab-http-server/src/state_stream.rs` | `tests/protocol/client.rs`; `CellStateStream` enforces per-output receipts, cancellation, deadlines, and fencing; `state_observing_body` adapts it to one-at-a-time HTTP chunks without a second queue | -| Product composition | `crab-http-server/src/cells/` | server route, restore, and lifecycle tests | - -Celld-style follower durability is connected to product command and schema- -migration response release. A response may be released by either an exact -object-store root or a write-all follower proof. The actor retains the Cell -until the exact root is published, and takeover consumes any fleet-only tail -before serving. The remaining release gate is live multi-node fault and -capacity qualification in -[Follower durability and warm failover](failover-and-followers.md). - -Use the map during review. A change to one boundary needs caller, callee, sibling, and source-loss evidence where applicable. - -## Run design-contract validation - -The documentation keeps executable SQL and Protocol Buffers inputs beside the crate. - -```bash -node crates/crab-cell-runtime/docs/validate.mjs -``` - -The script checks: - -- Runtime, KV, Blob, Queue, Cron, and Workflow schemas load in SQLite -- Foreign keys, lease states, token lengths, and unique constraints reject invalid rows -- The peer descriptor compiles and round-trips representative messages -- The peer contract defines messages, not a public service -- Markdown links, anchors, fences, and trailing whitespace remain valid - -This script validates contracts. It does not prove runtime behavior. - -## Run crate-level proof - -Set a worktree-specific external Cargo target directory before every Rust command. - -```bash -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-b347 \ - cargo test -p crab-cell-runtime - -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-b347 \ - cargo test -p crab-ltx --features replica - -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-b347 \ - cargo test -p crab-http-server -``` - -Run Clippy for all changed crates: - -```bash -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-b347-clippy \ - cargo clippy -p crab-ltx -p crab-cell-runtime \ - -p crab-http-server --all-targets -- -D warnings -``` - -Use the checkout's actual stable target suffix when it differs from `b347`. - -Release jobs bind schema-v5 qualification matrices and threshold-profile -digests to the exact tagged source, published image manifest, and raw cluster -evidence. The protected bundle must contain exact matrices for -`local-provider-v1`, `scale-v1`, `compatibility-v1`, each of -`provider-{s3,gcs,azure}-v1`, and each of `fault-{s3,gcs,azure}-v1`. -The crate-owned validator requires a pinned Ed25519 qualification public key, -canonical receipts, passing thresholds, exact source/image identity, and every -matrix row. Protected receipts also retain and verify a `release` execution -profile; debug or otherwise non-release runs cannot satisfy the release gate. -Fixture or self-signed evidence cannot satisfy the release gate. -The release job also compares the supplied protected profile byte-for-byte with -the checked-in profile from the tagged source before invoking the validator. -GitHub's workflow attestation remains the trust anchor for the release job and -source identity. - -The protected bundle gate is centralized in the fresh-process -`verify-protected-bundle` command. It checks the complete nine-profile bundle, -rejects symlinks below the evidence root, and applies the same source, image, -profile, signer, and freshness checks to every matrix; the workflows retain -only the exact-source profile byte comparison. - -Typed application qualification runs through `CellNode`: protected serial -workloads use `run_qualification`, while independent or idempotent operations -may use the readiness-gated `run_qualification_concurrent` entry point with an -explicit bounded concurrency. Local wiring/smoke checks that intentionally -observe only a subset of lifecycle cases use `run_qualification_observed`; that -result cannot satisfy a protected profile unless its artifact independently -proves the required case coverage. All paths reject results if the node begins -draining or a supervised task fails before the run completes. - -The raw four-process receipt is validated first with the fail-closed v6 command: - -```bash -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-main \ - cargo run -p crab-cell-runtime --bin qualification_receipt --locked -- \ - verify-matrix qualification-matrix.json "$SOURCE_SHA" "$IMAGE_DIGEST" \ - scale-v1.json "$QUALIFICATION_SIGNER" -``` - -This command verifies pinned attestation, canonical encoding, passed threshold -metrics, source/image identity, and BLAKE3 digests of every raw artifact. It -does not turn local or in-memory evidence into provider qualification; the -release matrix still needs the real RustFS/Kubernetes and multi-GiB runs below. -When a threshold profile other than `pr-contract-v1` is supplied, the CLI -requires the pinned signer argument and applies the protected freshness and -clock-skew gate. The profile-less form below is retained only for generic -historical receipt inspection and is not a release decision. - -Each profile is verified as one bounded schema-2 matrix instead of a -caller-owned row loop. `QualificationMatrixManifest` requires exactly one entry -for each of -these rows: `protocol`, `storage`, `publication`, `warm-path`, `churn`, -`fleet`, `failover`, `primitives`, `accounting`, and `compatibility`. Each entry -binds a relative receipt path and one or more relative raw-artifact paths. The -manifest and every receipt are canonical JSON; absolute paths, parent-directory -components, duplicate rows, missing rows, dirty receipts, source/image drift, -and any artifact digest mismatch fail closed. The release job runs this -verification independently for every required profile, so a valid scale matrix -cannot substitute for provider or Kubernetes fault evidence. - -After the release job writes the ten row receipts and their raw artifacts into -each protected matrix directory, verify one profile-bound matrix in a fresh -process: - -```bash -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-main \ - cargo run -p crab-cell-runtime --bin qualification_receipt --locked -- \ - verify-matrix \ - protected/qualification-matrix.json "$SOURCE_SHA" "$IMAGE_DIGEST" \ - protected/scale-v1.json "$QUALIFICATION_SIGNER" -``` - -Repeat that command for every matrix/profile pair listed above; the release -workflow does not accept a scale matrix as a substitute for any other profile. -The matrix verifier recomputes every raw digest, checks the primary artifact -digest for each receipt, and requires all rows to use the supplied source and -image identity. It is a release-evidence check only; it does not promote local -or in-memory runs to RustFS, Kubernetes, matched-latency, or capacity proof. - -The local RustFS qualification pass on 2026-09-18 used one isolated bucket and -unique prefixes with explicit credentials (the credentials were not written to -artifacts). It passed the LTX round trip, Cell source-loss takeover and -retention sweep, HTTP collaboration/takeover, native HTTP push, and receive -fault matrix. The same checkout passed the process-level movement probes and -the hydration shutdown-cancellation regression against the in-memory provider. -The provider-backed -`rustfs_mixed_primitive_inventory_churn_preserves_exact_roots` test also -passed Queue/Workflow retained-work protection, exact-root restore, and -capacity reuse against its isolated prefix. -These commands are provider evidence for iteration, not release receipts; -protected release jobs must consume a schema-v5 matrix signed by the pinned -qualification key and bound to the tagged source, immutable image, profile, -and every raw artifact. -Fault profiles additionally require a named injected fault, a non-`none` fault -schedule digest, and a monotonic ownership transition; a signed no-op receipt -cannot stand in for Kubernetes fault evidence. - -The inventory check below uses an in-memory store. The following source-loss, -retention, and public-host checks use the configured RustFS endpoint. Confirm -each selected command reports a nonzero passed-test count; Cargo accepts an -exact filter that matches no tests. The architecture workflow enforces this -check for every invocation in its RustFS recovery step. - -```bash -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-rustfs \ - cargo test -p crab-ltx --features replica --test cell --locked \ - cell::roots::lifecycle::exact_root_inventory_verifies_every_remote_dependency -- --exact - -AWS_ACCESS_KEY_ID="$AWS_ACCESS_KEY_ID" \ -AWS_SECRET_ACCESS_KEY="$AWS_SECRET_ACCESS_KEY" \ -CRAB_CELL_TEST_BUCKET="$BUCKET" \ -CRAB_CELL_TEST_ENDPOINT="$ENDPOINT" \ -CRAB_CELL_TEST_PREFIX="$UNIQUE_PREFIX" \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-rustfs \ - cargo test -p crab-cell-runtime --test runtime \ - runtime::lifecycle::ownership::recovery::rustfs_source_loss_takeover_restores_exact_root_and_continues_publication \ - --locked -- --ignored --exact - -AWS_ACCESS_KEY_ID="$AWS_ACCESS_KEY_ID" \ -AWS_SECRET_ACCESS_KEY="$AWS_SECRET_ACCESS_KEY" \ -CRAB_CELL_TEST_BUCKET="$BUCKET" \ -CRAB_CELL_TEST_ENDPOINT="$ENDPOINT" \ -CRAB_CELL_TEST_PREFIX="$UNIQUE_PREFIX-mixed" \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-rustfs \ - cargo test -p crab-cell-runtime --test runtime \ - runtime::lifecycle::idle::churn::rustfs_mixed_primitive_inventory_churn_preserves_exact_roots \ - --locked -- --ignored --exact - -AWS_ACCESS_KEY_ID="$AWS_ACCESS_KEY_ID" \ -AWS_SECRET_ACCESS_KEY="$AWS_SECRET_ACCESS_KEY" \ -CRAB_CELL_TEST_BUCKET="$BUCKET" \ -CRAB_CELL_TEST_ENDPOINT="$ENDPOINT" \ -CRAB_CELL_TEST_PREFIX="$UNIQUE_PREFIX-retention" \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-rustfs \ - cargo test -p crab-cell-runtime --lib \ - recovery::retention::tests::rustfs_maintenance_collection_preserves_live_and_pinned_graphs \ - --locked -- --ignored --exact - -AWS_ACCESS_KEY_ID="$AWS_ACCESS_KEY_ID" \ -AWS_SECRET_ACCESS_KEY="$AWS_SECRET_ACCESS_KEY" \ -AWS_ENDPOINT_URL_S3="$ENDPOINT" AWS_ALLOW_HTTP=true \ -AWS_VIRTUAL_HOSTED_STYLE_REQUEST=false \ -QUALIFICATION_BUCKET="$BUCKET" QUALIFICATION_PREFIX="qualification/http-receive-$UNIQUE_PREFIX" \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-rustfs \ - cargo test -p crab-http-server --lib \ - server::receive_fault_tests::receive_faults_rustfs \ - --locked -- --ignored --exact - -AWS_ACCESS_KEY_ID="$AWS_ACCESS_KEY_ID" \ -AWS_SECRET_ACCESS_KEY="$AWS_SECRET_ACCESS_KEY" \ -CRAB_HTTP_CELL_TEST_BUCKET="$BUCKET" \ -CRAB_HTTP_CELL_TEST_ENDPOINT="$ENDPOINT" \ -CRAB_HTTP_CELL_TEST_PREFIX="http-$UNIQUE_PREFIX" \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-rustfs \ - cargo test -p crab-http-server --lib \ - server::peer_e2e_tests::rustfs_public_collaboration_reaches_remote_owner_and_publishes_ltx \ - --locked -- --ignored --exact - -AWS_ACCESS_KEY_ID="$AWS_ACCESS_KEY_ID" \ -AWS_SECRET_ACCESS_KEY="$AWS_SECRET_ACCESS_KEY" \ -AWS_ENDPOINT_URL_S3="$ENDPOINT" AWS_ALLOW_HTTP=true \ -AWS_VIRTUAL_HOSTED_STYLE_REQUEST=false \ -QUALIFICATION_BUCKET="$BUCKET" QUALIFICATION_PREFIX="qualification/http-push-$UNIQUE_PREFIX" \ -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-rustfs \ - cargo test -p crab-http-server --lib \ - server::receive_tests::native_http_push_rustfs \ - --locked -- --ignored --exact -``` - -The coordination simulator has a deterministic seed replay entry point in the -normal runtime test binary. It never starts I/O or Tokio work: - -```bash -CRAB_COORDINATION_SEED=41 CRAB_COORDINATION_STEPS=256 \ - CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-main \ - cargo test -p crab-cell-runtime --lib coordination::sim::replay_requested_seed_from_environment \ - --locked -- --exact --nocapture -``` - -## Prove the mutation contract - -Mutation tests must cover the complete durability boundary. - -```mermaid -flowchart LR - Submit[Submit typed command] - Commit[Commit SQLite] - Capture[Capture LTX] - Publish[Publish exact root] - Delete[Delete local database] - Restore[Open on successor] - Read[Read same outcome and state] - - Submit --> Commit --> Capture --> Publish --> Delete --> Restore --> Read -``` - -Required cases are: - -| Case | Expected proof | Proven by | -| --- | --- | --- | -| Replay | Same request ID and digest returns stored outcome without rerunning handler | `runtime::publication::lost_publication_response_reconciles_without_replaying_sql` | -| Identity conflict | Same request ID with different digest is a durable rejection | `runtime::lifecycle::execution::resolve_distinguishes_committed_absent_conflict_and_expired` | -| Caller cancellation | Accepted command still publishes and later resolves | `runtime::workers::cancelled_waiter_does_not_cancel_an_accepted_sql_command` | -| Lost CAS response | Exact successor is adopted; a different winner fences | `runtime::lifecycle::ownership::lost_release_response_is_reconciled_before_successor_acquire` | -| Source loss | Successor restores exact database and outcome from object storage | `runtime::lifecycle::ownership::crashed_process_is_fenced_before_successor_restore` | -| Recovery state | Exact-root activation CASes `Recovering` to `Serving` before returning a handle | `runtime::lifecycle::ownership::takeover_resumes_pinned_recovery_before_serving` | -| Timeout | Admission closes; tentative local state never publishes | `runtime::lifecycle::execution::native_handler_deadline_discards_late_commit_and_reopens_authoritative_root`, `runtime::lifecycle::execution::sqlite_query_is_interrupted_at_wall_deadline` | -| Panic | Affected Cell fences; worker thread remains usable | `runtime::workers::panicking_handler_fences_only_its_cell_and_worker_continues` | -| Drain | Accepted commands publish before SQLite close and authority release | `runtime::lifecycle::execution::runtime_shutdown_drains_accepted_work_and_releases_all_owners` | - -Every filter above selects exactly one case in the `runtime` suite; run it with -`cargo test -p crab-cell-runtime --features test-support --test runtime --locked`. - -Mock-only tests do not satisfy source-loss or publication proof. - -`tests/runtime/publication.rs` injects the ambiguous publication window at the object -store boundary: the backend accepts the control `Update`, then the decorator -returns a connection-reset error. The publisher must reload the exact root, -clear the retained cut, and return the recorded outcome without invoking the -SQL handler again. This proves local reconciliation; the three-Pod gate must -still inject the same lost response through the deployed network path. - -## Prove storage and LTX behavior - -`crab-ltx` evidence must cover: - -- Managed capture and explicit snapshots stream pages to atomic local files, - then validate format and BLAKE3 without a database-sized resident buffer -- Pending cuts stay owned until the exact root is confirmed -- Destination capacity is reserved before full restore downloads -- Sparse hydration verifies directory, frame, and page checksums -- Truncate and regrow cannot reuse invalid locators -- Incremental roots load only touched directory paths -- Full restore installs without replacing an existing destination -- Scheduled compaction promotes eligible singleton and multi-segment levels -- Compaction preserves transaction ID, checksum, sequence, and schema -- Injected filesystem and executor failures clean owned scratch state -- Cross-epoch continuation starts from the authoritative root -- Backup pins bind release metadata, catalog revisions, exact controls, and every reachable immutable root dependency -- Offline retention verifies current controls and every pin before deleting, - rejects owned controls, honors grace and deletion bounds, and preserves - unknown object layouts -- Backup creation remains advertised from before its final `Ready` check until - its pin pointer is durable, so maintenance cannot miss an in-flight pin - -Repeat remote storage tests against real RustFS. In-memory object storage cannot prove provider ETag and streaming behavior. - -The server's real-RustFS backup smoke must create a pin, repeat creation with -the same ID, verify it independently, remove one isolated test dependency and -observe fail-closed verification, then republish that content-addressed -dependency from a second pin. It must then restore the pin twice into a fresh -prefix, observe identical summaries, inspect unowned `Idle` authority, and use -a separate process configured only for the destination prefix to verify the -complete graph with the pinned release. - -The RustFS qualification tests require one fresh bucket and a unique Cell prefix. -The first command is the in-memory inventory check. Each command must report -at least one passed test: - -```bash -cargo test -p crab-ltx --features replica --test cell \ - cell::roots::lifecycle::exact_root_inventory_verifies_every_remote_dependency -- --exact - -CRAB_CELL_TEST_BUCKET="$BUCKET" \ -CRAB_CELL_TEST_ENDPOINT="$ENDPOINT" \ -CRAB_CELL_TEST_PREFIX="$UNIQUE_PREFIX" \ -cargo test -p crab-cell-runtime --test runtime \ - runtime::lifecycle::ownership::recovery::rustfs_source_loss_takeover_restores_exact_root_and_continues_publication \ - -- --ignored --exact - -CRAB_CELL_TEST_BUCKET="$BUCKET" \ -CRAB_CELL_TEST_ENDPOINT="$ENDPOINT" \ -CRAB_CELL_TEST_PREFIX="$UNIQUE_PREFIX" \ -cargo test -p crab-http-server --test public_cell_qualification \ - rustfs_public_cell_node_runs_typed_primitive_workload \ - -- --ignored --exact --nocapture - -CRAB_CELL_TEST_BUCKET="$BUCKET" \ -CRAB_CELL_TEST_ENDPOINT="$ENDPOINT" \ -CRAB_CELL_TEST_PREFIX="$UNIQUE_PREFIX" \ -cargo test -p crab-cell-runtime --lib \ - recovery::retention::tests::rustfs_maintenance_collection_preserves_live_and_pinned_graphs \ - -- --ignored --exact - -CRAB_HTTP_CELL_TEST_BUCKET="$BUCKET" \ -CRAB_HTTP_CELL_TEST_ENDPOINT="$ENDPOINT" \ -CRAB_HTTP_CELL_TEST_PREFIX="$UNIQUE_PREFIX" \ -cargo test -p crab-http-server --lib \ - server::peer_e2e_tests::rustfs_public_collaboration_reaches_remote_owner_and_publishes_ltx \ - -- --ignored --exact --nocapture -``` - -The Cell test publishes a command on one session, removes its local database, -takes over from a second session, resolves the original request from the exact -root, and publishes the next sequence. The public-host test drives SQL, KV, -Blob, Queue, Cron, Workflow, Activity, and Effects through the typed -`CellNode` application handle against the same real provider. CI runs these -tests against a pinned RustFS image; they remain iteration evidence until the -protected provider, Kubernetes, and scale matrix receipts pass. - -The same CI job also runs `crab-http-server` through public HTTP and private -mTLS forwarding to a heartbeat-renewed remote owner. Native Git creates main -and feature commits; public APIs then publish an issue, comment, label, status, -check run, pull request and comment, release and Git tag, and branch protection -through SQLite/LTX on that RustFS origin. The test stops the owner endpoint, -withdraws its current advertisement, publishes the authoritative fenced -takeover, deletes the old local Cell directory, and restores the exact root on -the ingress runtime. It reads every saved product surface again, publishes the -next sequence, clones the feature branch, resolves the release tag, and asserts -that no retired `app/v1` collaboration object exists. This combines the product -network, source-loss, hard-cut, and storage boundaries; it does not replace the -three-Pod kill and partition matrix below. - -The local Compose cluster qualification adds a real three-process owner-loss -case on one host. It writes through node B, verifies private forwarding from A -and C, kills B without draining, waits for B's signed session advertisement to -expire, and requires C to restore the same root at a higher epoch before it can -publish the next sequence. Restarted B has empty local Cell storage and must -route to C. The script emits the exact sessions, epochs, root, sequences, and -live admission envelopes as JSON. Before workload, it also compares every -node's `cells capacity --json --live` disk and active-Cell ceilings with the -corresponding runtime Prometheus gauges, the signed placement block returned by -`cells node --session SESSION --json`, and an independent probe of the mounted -Cell filesystem with df (1 MiB tolerance); a mismatch fails the script and the -receipt records all parity results. Because the processes share one network -namespace, this proves local process, signed-placement, and local-disk loss -behavior but not Pod networking or partition behavior. - -The shipped Kubernetes qualification script adds one real three-Pod owner-loss -case. It reads the durable repository Cell control, maps the serving endpoint to -a ready Pod, force-deletes that Pod without grace, then requires a different -session at a higher epoch to restore the exact digest, transaction ID, checksum, -and commit sequence and serve the public status before publishing a strictly -newer root and a second status visible through another replica. Before takeover -traffic, it queries the -old boot session through `cells node --session SESSION --json` until the signed -advertisement is no longer live; Pod deletion alone is not expiry evidence. -The receipt also records the failed log epoch and original follower NodeIds, -maps the new owner endpoint back to its replacement Pod, and requires that -successor's stable NodeId to be one of those original followers. This proves -the production deployment exercises follower-affine takeover rather than only -object-root restoration. -That case is not evidence until its -signed provider receipt exists, and it does not replace the remaining partition -and commit-window faults. Browser E2E is intentionally outside this gate. - -## Emit and verify qualification receipts - -Release evidence uses the signed, version-3 `QualificationReceipt` contract in -`src/qualification.rs`. A receipt is bound to the source revision, artifact -digest, execution profile, topology, workload seed, bucket-call count, peak -resident set, bounded named measurements, start/finish timestamps, a digest of -the exact fault schedule, every retained raw-artifact digest, and sampled -epoch/published-root ownership watermarks (latency/duration measurements are -recorded by the harness as metrics). The schema is versioned and rejects -unknown fields, dirty worktrees, oversized labels, embedded credentials, and -forged signatures. The runner signs the canonical JSON after recording the -artifact digest; verification recomputes that digest and requires it to appear -in the raw-artifact set before accepting the receipt. - -Consumers call `receipt.verify_for(expected_source, expected_image, artifact)` -after decoding. This rejects a validly signed receipt issued for another -source revision, image, or raw artifact; signature validity alone is not release -eligibility. - -The contract test surface is: - -```bash -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-qualification-receipts \ - cargo test -p crab-cell-runtime qualification --lib --locked -``` - -This proves receipt encoding, signature verification, artifact binding, dirty -source rejection, forged-field rejection, complete matrix membership, and -multi-artifact digest recomputation. It does not claim that a local receipt -proves RustFS, Kubernetes, partition, or capacity behavior. Protected -qualification jobs must attach the emitted receipt to the exact source and -image digest and feed it through a fresh verifier before release consumes it. - -## Prove each primitive through recovery - -Primitive tests require more than procedure-level SQL assertions. - -| Primitive | Recovery evidence | -| --- | --- | -| SQL | Publish a batch, delete local DB, restore, query the same rows | -| KV | Apply checks and writes, restore, preserve versions and TTL behavior | -| Blob | Complete multipart publication, restore, preserve ETag and range bytes | -| Queue | Publish claim, restore, validate lease, reclaim expiry, complete attempt | -| Cron | Publish occurrence effect, restore, preserve next due time and generation | -| Workflow | Pin definition, publish transition, restore, run retained definition | -| Activity | Publish claim, lose owner, take over, reclaim lease, complete next attempt | -| Effect | Publish source intent, deduplicate destination, resolve ambiguity, acknowledge source | - -Also prove output, row, payload, attempt, lease, and retention bounds. - -## Prove product integration - -`crab-http-server` must show a user action, a real durable side effect, and a visible result. - -The repository path needs these cases: - -1. Start the server with real object storage -2. Create or adopt a repository and verify an empty Cell root -3. Exercise issue, comment, label, status, check, pull, release, and settings routes -4. Delete the owning node's local Cell files -5. Route the next request through another node -6. Verify the public repository API reads the restored state -7. Exercise public Git clone, fetch, and push independently of collaboration SQLite -8. Confirm shutdown drains both HTTP/Git work and Cell publication - -The hard cut test must also prove that retired `app/v1` collaboration objects are never read. - -## Run multi-Pod fault qualification - -Unit and in-process integration tests don't prove network ownership behavior. Run at least three Pods against one RustFS origin. - -```mermaid -flowchart TB - Load[Deterministic workload] - P1[Pod A] - P2[Pod B] - P3[Pod C] - Faults[Kill, partition, latency, lost reply] - Origin[(RustFS)] - - Load --> P1 - Load --> P2 - Load --> P3 - Faults --> P1 - Faults --> P2 - Faults --> P3 - P1 --> Origin - P2 --> Origin - P3 --> Origin -``` - -Inject these faults while commands continue: - -- Kill the current owner before and after SQLite commit -- Drop the control-CAS response after the origin accepts it -- Delay immutable uploads -- Partition one management endpoint -- Expire a node advertisement -- Exhaust local disk admission -- Restart with no local SQLite files -- Roll from one compatible image to another -- Enter maintenance with stored incompatible work - -After each fault, verify one authoritative owner, monotonic sequence, stable replay, no false success, and no leaked activity or effect lease. - -## Capacity qualification - -The target workload is 1,000 to 10,000 active databases per node, 100 MB to 5,000 MB per database, and 1,000 aggregate transactions/s per node. - -Qualify each node profile separately: - -| Profile | Process-visible resources | Required matrix | -| --- | --- | --- | -| Small | 1–2 CPU credits, 2–4 GiB memory, 50–100 GiB SSD | Minimum supported workload, admission behavior, drain under pressure | -| Medium | 4–8 CPU credits, 8–16 GiB memory, 100–200 GiB SSD | Mixed repository sizes, sustained command target, sparse takeover | -| Large | 16 CPU credits, 32–64 GiB memory, 500–1,000 GiB SSD | Maximum active-Cell target, 5,000 MB restore, compaction and renewal load | - -Before starting traffic, capture the exact resource-derived envelope from every -node. A profile label or Kubernetes request is not evidence of the resources -visible to the process. The live Kubernetes qualifier rejects a Pod whose -cgroup-aware CPU credits, effective memory limit, or configured and enforced -local-disk capacity falls outside the selected profile. Effective capacity is -the smaller of the backing filesystem and the configured limit. The report -also binds that limit to the Pod's `emptyDir.sizeLimit`; current filesystem -free space remains a separate admission input. - -```bash -kubectl --namespace crab exec POD -- \ - crab-http-server --config /etc/crab/http-server/server.toml \ - cells capacity --json --live > capacity-before.json -kubectl --namespace crab exec POD -- \ - crab-http-server --config /etc/crab/http-server/server.toml \ - cells metrics > metrics-before.prom -``` - -Run the bounded HTTP harness from a dedicated load generator. Each read -`--target` has the form `NAME=CONCURRENCY@/PATH`. A mutation has the form -`NAME=CONCURRENCY@/PATH|BODY_FILE`; all targets run simultaneously. -Redirect stdout to retain its versioned JSON receipt. Put private cookies or -authorization values in a mode-0600 header file, never in command arguments. - -Use a disposable repository for mutation qualification because every successful -request creates durable state. The JSON template must contain the exact -top-level marker `"request_id":"{{request_id}}"`; the harness replaces it with -a new UUIDv7 for every request. - -The release Kubernetes qualifier builds this harness from the exact tagged -source and drives 1,000 aggregate mutation requests/s through each ready Pod -for 60 seconds. The checked-in profile uses 64 distinct commit-status targets -distributed across eight repository Cells (24 commits per Cell); this is the -node-wide aggregate capacity proof while retaining the same per-commit -submission history. Each Pod receives eight targets per Cell, and the receipt -records a configured 125 target requests/s per Cell. Retain a one-Cell run as a -separately labelled hot-Cell limit test. Every attested receipt retains each Pod -UID, p50/p95/p99 latency, success and admission counts, plus capacity envelopes -before and after load. - -```json -{"request_id":"{{request_id}}","title":"load qualification","body":"durable command"} -``` - -```bash -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-load-generator \ - cargo run -p crab-http-server --release --example qualify_http_load --locked -- \ - --base-url https://git.example.com \ - --target 'refs=4@/api/repos/team/project/refs' \ - --target 'commits=8@/api/repos/team/project/commits?rev=main&limit=20' \ - --target 'readme=4@/api/repos/team/project/file?rev=main&path_hex=524541444d452e6d64' \ - --mutation 'issues=16@/api/repos/team/disposable-load/issues|/secure/new-issue.json' \ - --aggregate-requests-per-second 1000 \ - --duration-seconds 300 \ - --warmup-seconds 15 \ - --header-file /secure/load-headers \ - > http-load.json -``` - -The harness fully consumes each body and reports the method, 2xx responses, admission -rejections, unexpected responses, transport/body-limit failures, bytes, -throughput, and all-response plus successful-response latency percentiles. It -checks `/livez` before and after traffic. HTTP 429 is an expected overload -signal; any other non-2xx response, transport failure, oversized body, or -unhealthy liveness check makes the command fail after writing the receipt. A -fixed aggregate-rate run also fails when successful responses fall below 95% -of its configured request count, so 429 responses cannot satisfy the 1,000 TPS -capacity target. - -The report separates CPU-bounded blocking jobs, dirty-memory-bounded jobs, and -the two-slot full-recovery ceiling. Store the report with the immutable image -digest, profile, workload parameters, and live measurements. Reject a receipt -when its observed active-Cell, resident-byte, retained-byte, or local-disk capacity differs -from the corresponding private metrics sample taken before traffic, or when a -live usage gauge exceeds its advertised capacity. - -The Kubernetes qualification receipt records this report for every original -Pod, every Pod after the zero-unavailable rollout, and every Pod after forced -owner loss. Each entry is bound to the Pod UID so a replacement cannot inherit -another process's startup envelope. - -Measure: - -- Resident set size per active Cell and per workload class -- File descriptors per Cell and under HTTP/Git load -- Local bytes for main DB, WAL, retained LTX, sparse pages, and scratch -- Command p50, p95, and p99 latency -- Object-store requests and bytes per command -- Renewal updates per second -- Scheduler pass duration and due lag -- Takeover time with cold and warm directory caches -- Full restore and sparse first-read time -- Compaction throughput and peak scratch -- Graceful shutdown duration - -Reject overload before SQLite or remote download starts. Qualification fails if the process swaps, exceeds its file-descriptor reserve, violates a five-second scheduler pass, or admits work beyond its shared disk budget. - -## Apply the production gate - -Production readiness requires all of these gates: - -- Contract validation passes -- Runtime, LTX, and server tests pass -- Clippy and architecture guardrails pass -- Real RustFS source-loss recovery passes -- Three-Pod fault qualification passes -- Every supported node profile has measured admission envelopes -- Repository API and Git operations pass end to end -- Backup and same-bucket isolated-prefix restore pass -- Hard cutover rehearsal confirms no legacy collaboration reads - -Until the last gate passes, describe the implementation as functionally complete but not production-qualified. diff --git a/crates/crab-cell-runtime/docs/deployment.md b/crates/crab-cell-runtime/docs/deployment.md deleted file mode 100644 index ce3b40938..000000000 --- a/crates/crab-cell-runtime/docs/deployment.md +++ /dev/null @@ -1,489 +0,0 @@ -# Deploy and operate a Cell-enabled Crab fleet - -Run one `crab-http-server` process per Kubernetes Pod or virtual machine. Nodes share one object-store origin, forward private requests over mutual TLS (mTLS), and advertise release and capacity state in the origin. - -| Document intent | Value | -| --- | --- | -| Content type | How-to and operations reference | -| Audience | Crab operators and release engineers | -| Goal | Configure, size, roll out, drain, restore, and observe a Cell fleet | - -[Back to the Cell runtime index](README.md) - -The deployment enables follower fsync as one response-durability proof. It -creates the local follower store, exposes authenticated append/seal/tail -handling on the private mTLS listener, and starts recovery-only before serving -public traffic. Object-root publication remains a valid alternative proof, and -all recovery still goes through the existing claim, witness, pin, control-CAS, -and fresh-database activation gates described in -[Follower durability and warm failover](failover-and-followers.md). -Owner queries remain the default. In object durability mode, the server -reconciles a desired-reader policy and serves explicit authenticated reads -from admitted, verified snapshots. The issue-detail route below is the first -public product consumer. This is local implementation evidence for -[Plan 036](../../../advisor-plans/036-cell-read-replicas-and-fenced-promotion.md), -not a qualified production deployment. Fleet-only acknowledgements do not -provide the plan's all-secondary-loss guarantee. - -An administrator can set a repository Cell's desired read-replica count with -`PUT /api/repos/{owner}/{name}/settings/read-replicas`, sending -`{"expected_revision":0,"desired_readers":1}` for the first policy and the -returned revision for later changes. `GET` on the same path returns the current -target and revision. A successful update means the S3 policy CAS completed; -the update's `convergence` field is `pending`. `GET` probes the selected nodes -and reports selected, proven-ready, and unverified reader counts plus the -lowest proven sequence. Readiness probes use at most 16 concurrent requests -inside a five-second budget and do not activate views. Failed or timed-out -probes remain unverified; a selected-node count below the target is a placement -`shortfall`. The server sends an authenticated owner hint after the -CAS and also reconciles owned Cells periodically. A stale revision returns -HTTP 409. This API is available only in the object durability profile. - -For repository issue detail, `GET /api/repos/{owner}/{name}/issues/{number}?read=replica` -selects an admitted reader and returns `x-crab-cell-reader`, -`x-crab-cell-incarnation`, and `x-crab-cell-sequence` headers naming the -serving node and the issue query's observed position. Supply -both `after_incarnation` (32 lowercase hex digits) and `after_sequence` on a -later replica request to require at least that position. The route reports -`replica_behind` (409) or `replica_unavailable` (503) and never runs the issue -query on the owner as a fallback. The default issue route still reads from the -owner. Issue and label metadata use one typed query against the same snapshot; assignee metadata comes from the authorized repository configuration. - -After the owner session expires, verified snapshots remain warm but cannot -answer queries. Recovery probes the selected live nodes and prefers a warm -reader, after any mandatory durability-log successor. The destination closes -reader admission, proves predecessor death, wins the normal Cell-control CAS, -and opens a fresh writable database from the authoritative root. It never -turns a read-only connection into a writer. A missing warm reader only removes -the placement preference; the existing cold recovery path retains every -session, old-log, and authority gate. - -On shutdown, the host cancels work producers and closes reader admission, -then drains accepted runtime work and seals the covered node log. This includes -admitted replica SQL and snapshot-open jobs whose callers were cancelled. -Heartbeat maintenance remains live through that barrier. Session withdrawal follows -runtime drain; withdrawing earlier fences the log authority and prevents a -clean fleet-to-object transition. Every phase uses the same absolute shutdown -deadline. A failed drain must not authorize a durability-mode change. - -## Configure one process per node - -The existing HTTP server owns Cell runtime construction. Configuration supplies the authoritative object store, local volume, public listener, management listener, and peer identity. - -```mermaid -flowchart TB - Public[Public listener
HTTP and Git] - Management[Management listener
peer mTLS and admin] - Server[One crab-http-server process] - Volume[(Local SSD cache)] - Origin[(Object-store authority)] - - Public --> Server - Management --> Server - Server --> Volume - Server --> Origin -``` - -Startup fails before readiness when any required boundary is invalid: - -- Object storage cannot prove strict create and conditional update behavior -- Root identity differs from configured tenant or application -- Compiled registry differs from the selected release descriptor -- Local memory, disk, or file-descriptor floor is unavailable -- Peer certificate, fleet, release, or node advertisement is invalid -- The first complete scheduler scan has not finished - -The server doesn't open a second public primitive listener. - -## Route through any healthy node - -The external load balancer doesn't need Cell affinity. Any node can receive a public request. - -```mermaid -flowchart LR - LB[Layer 7 load balancer] - A[Node A
request receiver] - Catalog[(Catalog + control)] - B[Node B
Cell owner] - - LB --> A - A --> Catalog - A -->|private mTLS forward| B -``` - -The receiving node resolves the exact owner from `control.json` and its signed live advertisement. It dispatches locally when it owns the Cell, otherwise it forwards once. - -A stale endpoint retry is allowed only when the first attempt definitely did not start. An ambiguous mutation returns evidence for `Resolve`. - -## Advertise node health and capacity - -Each node publishes a signed advertisement every three seconds. The record binds: - -- Fleet and session IDs -- Certificate subject public key info -- Image and release digests -- Compiled registry inventory -- Publicly routable peer endpoint -- Free memory, disk, and job credits -- Scheduler progress - -Advertisements expire after 15 seconds. A node with no scheduler progress for 15 seconds or zero advertised capacity leaves rendezvous assignment until it reports progress and capacity again. - -Local admission remains authoritative. An advertisement cannot force a node to accept work after its measured budget is exhausted. - -## Size node profiles and admission - -The runtime derives active-Cell limits from measured resources. Profile names are operator guidance, not fixed performance claims. - -| Profile | vCPU | Memory | Local SSD | Intended use | -| --- | ---: | ---: | ---: | --- | -| Small | 1 to 2 | 2 to 4 GiB | 50 to 100 GB | Development and low-traffic fleets | -| Medium | 4 to 8 | 8 to 16 GiB | 100 to 200 GB | General production nodes | -| Large | 16 | 32 to 64 GiB | 500 GB to 1 TB | Dense ownership and high aggregate throughput | - -Startup rejects less than 2 GiB memory or 20 GiB usable disk. - -Resource admission accounts these consumers: - -```mermaid -flowchart TD - Resources[Measured node resources] - Cells[Active Cell slots] - Bytes[Shared byte budget] - Jobs[Dirty and full-job credits] - FDs[File descriptor reserve] - IO[Object I/O permits] - - Resources --> Cells - Resources --> Bytes - Resources --> Jobs - Resources --> FDs - Resources --> IO -``` - -Each open `Db` charges: - -- Three SQLite connection page caches at 64 KiB each -- Eight file descriptors -- Runtime actor and mailbox bytes -- Sparse-page cache allowance -- Native handler allowance - -The shared byte budget covers SQLite files, WAL, retained LTX, sparse pages, -and Git/LFS/Release staging. Full restore and compaction use a separate weighted -scratch budget; after admission, the server remeasures actual free space against -both budgets and the node reserve before any remote body download. - -Large jobs reserve two database sizes plus 64 MiB scratch and 64 MiB memory. Concurrent large jobs cap at the smaller of vCPU count and two, then apply byte admission. - -The 1,000 to 10,000 open-Cell and 1,000 command/s aggregate targets require [capacity qualification](delivery.md#capacity-qualification). A profile does not guarantee either number before measurement. - -## Build one canonical release - -The deployable unit is the complete `crab-http-server` image. - -```text -Rust modules + migrations + Cargo.lock + React assets - | - v - crab-http-server image - | - v - canonical registry descriptor -``` - -The descriptor includes: - -| Field | Contract | -| --- | --- | -| `version` | Integer `1` | -| `runtime` | `crab-http-server` | -| `peer_versions` | Sorted unique versions; V1 supports `1` | -| `modules` | At most 128 canonical module entries | -| `namespaces` | At most 128 stable namespace entries | -| `build` | Source revision and `Cargo.lock` digest | - -The descriptor has a 256 KiB limit. Its BLAKE3 digest identifies the compiled release. The operator records the OCI image SHA-256 digest in release state. - -## Activate compatible releases - -The administrative commands operate through the existing server binary: - -```text -crab-http-server --config config.toml cells release inspect --json -crab-http-server --config config.toml cells capacity --json --live -crab-http-server --config config.toml cells release bootstrap --image sha256:1234567890 -crab-http-server --config config.toml cells release prepare \ - --expected-revision 7 --image sha256:1234567890 -crab-http-server --config config.toml cells release activate \ - --expected-revision 8 --strategy compatible \ - --minimum-eligible-nodes 3 -crab-http-server --config config.toml cells release status -crab-http-server --config config.toml cells status --owner team --name repository -``` - -The repository status command reads the durable control object without opening -the Cell or changing ownership. Use its versioned JSON to map a serving endpoint -to a fleet member during takeover qualification. - -The capacity command with `--live` reads the startup envelope retained by the -running server: process memory limit, free local disk, file descriptor limit, -the configured local-disk limit, CPU-derived job credits, and the resulting -admission budgets. `disk_capacity_bytes` is the smaller of that limit and the -backing filesystem total. The limit is Crab's admission ceiling and, in the -Helm deployment, the same byte count configures `emptyDir.sizeLimit`. Without -`--live`, the command calculates a preflight envelope for its short-lived -process instead. Neither mode claims a throughput result. Capture the live -report before every capacity run and compare it with the node-wide gauges -during the workload. - -```json -{ - "version": 1, - "resources": { - "memory_bytes": 2147483648, - "disk_limit_bytes": 64424509440, - "disk_capacity_bytes": 64424509440, - "free_disk_bytes": 53687091200, - "available_file_descriptors": 1048570, - "job_credits": 2 - }, - "admission": { - "active_cells": 2457, - "retained_bytes": 80530636, - "blocking_jobs": 2, - "dirty_jobs": 2, - "recovery_jobs": 2, - "scratch_bytes": 14316208128, - "local_disk_bytes": 28632416256, - "disk_reserve_bytes": 10737418240 - }, - "reservations": { - "active_cell_page_cache_bytes": 196608, - "active_cell_native_bytes": 65536, - "active_cell_file_descriptors": 8, - "dirty_job_memory_bytes": 67108864, - "maximum_recovery_jobs": 2 - } -} -``` - -Compatible activation follows this state machine: - -```mermaid -stateDiagram-v2 - [*] --> Prepared: prepare descriptor and image - Prepared --> Activating: CAS expected revision - Activating --> Activating: migrate assigned Cells - Activating --> Ready: quorum and all Cells current - Prepared --> Maintenance: incompatible activation - Maintenance --> Ready: fleet drained and all Cells transformed - Prepared --> Failed: explicit operator failure -``` - -Before `Ready`, activation verifies: - -1. Candidate descriptor bytes equal the running binary -2. The requested number of eligible nodes advertises the exact release -3. Every catalog shard remains revision-stable during the scan -4. Every non-tombstoned Cell uses current code and maximum schema -5. The eligible-node quorum still holds immediately before the final CAS - -Prepare never changes the current release. - -## Use maintenance for incompatible changes - -Maintenance activation stops normal serving, drains the fleet, and runs one signed zero-capacity maintenance executor. - -The executor owns one SQL worker and one active-Cell slot. It walks the catalog sequentially through the same exact-root migration path used by normal nodes. - -Before final `Ready`, it checks persisted work that may reference removed behavior: - -- `sys_requests` -- `sys_inbox` -- `sys_effects` -- Queue messages and producer dedup rows -- Workflow runs -- Blob objects -- Cron schedules - -Any matching row keeps the release in maintenance. The runtime doesn't guess payload compatibility. Operators must drain retention or compile a purpose-built transform. - -The same fence can reclaim unreachable immutable Cell objects after migration: - -```bash -crab-http-server --config config.toml cells release activate \ - --expected-revision 8 \ - --strategy maintenance \ - --retention-grace-hours 168 \ - --retention-max-deletes 10000 -``` - -Omit both retention flags to run migration only. `--retention-max-deletes` -requires a nonzero grace and accepts 1 through 100,000; omitting the limit while -supplying a grace uses 10,000. The grace is measured from each object's provider -modification time. - -Collection runs only after the executor proves it is the sole advertised -session and every current control is unowned. It verifies live controls and all -retained backup pins into a disk-backed mark set before streaming the object -inventory. Unknown layouts are skipped. The structured completion log records -listed, candidate, reachable, grace, eligible, and deleted counts. - -If eligible objects exceed the selected deletion bound, the command returns an -incomplete-retention error and deliberately leaves the release in -`Maintenance`. Repeat the identical activation command and expected revision; -the operation re-marks authority before deleting the next bounded batch. Each -retry selects the same next unused session identity as competing executors; -conditional creation admits only one. It advances past permanently retired -identities and never revives a withdrawn session. After advertising, it checks -the exact maintenance release again before opening any Cell. The executor -closes its runtime through its owning `CellNode` before collecting objects; -final host cleanup is idempotent. Start -the fleet only after the activation returns a `Ready` release. - -## Deploy on Kubernetes - -Use a `Deployment` for stateless process identity and a per-Pod local volume for cache data. Object storage remains authoritative. - -```yaml -apiVersion: apps/v1 -kind: Deployment -metadata: - name: crab-http-server -spec: - replicas: 3 - template: - spec: - terminationGracePeriodSeconds: 180 - containers: - - name: crab - image: registry.example/crab@sha256:1234567890 - args: - - --config - - /etc/crab/server.toml - - --peer-advertise-host - - $(CRAB_POD_IP) - readinessProbe: - exec: - command: [crab-http-server, --config, /etc/crab/server.toml, healthcheck] - volumeMounts: - - { name: cell-cache, mountPath: /var/lib/crab } -``` - -The snippet shows topology, not a complete production manifest. The shipped -chart adds a Downward API Pod IP, a stable peer TLS server name, peer Secret -mounts, exec readiness, TCP liveness, a three-replica floor, PDB `minAvailable: -2`, and same-selector management ingress. Supply object-store identity, -resource requests, edge ingress, and provider-specific placement through the -deployment environment. - -Readiness requires: - -- Runtime startup completed -- Release permits this compiled registry -- Node advertisement is live -- First scheduler pass completed -- Terminal drain has not started - -Liveness should report process health, not temporary capacity exhaustion. - -## Drain without split ownership - -On SIGTERM or release exclusion, the server: - -1. Closes readiness and new HTTP, Git, and Cell admission -2. Advertises zero capacity while draining -3. Finishes accepted HTTP and Git work -4. Stops schedulers and cooperatively cancels activity supervisors -5. Publishes every accepted Cell command -6. Closes SQLite handles -7. Releases exact owned controls to `Idle` -8. Joins SQL and blocking worker pools -9. Withdraws its exact node advertisement - -Set the Kubernetes grace period above the server's bounded drain budget. A forced kill remains recoverable through stale-owner takeover, but it increases unavailable time. - -## Back up immutable roots and release metadata - -A backup records object-store data, not local cache volumes. - -Include: - -- Root identity -- Release records and immutable descriptors -- Catalog heads and immutable pages -- Cell control records -- Every immutable object reachable from pinned roots -- Node-independent application configuration needed to recreate the fleet - -Backup traversal pins its start revisions and strict-creates the pin only after -every referenced object verifies. Restore verifies that pin before copying. - -Create a nonzero 16-byte pin ID and verify it independently: - -```bash -crab-http-server --config /etc/crab/server.toml cells backup create \ - --pin 11112222333344445555666677778888 -crab-http-server --config /etc/crab/server.toml cells backup verify \ - --pin 11112222333344445555666677778888 -crab-http-server --config /etc/crab/server.toml cells backup restore \ - --pin 11112222333344445555666677778888 \ - --destination-prefix recovery/restore-2026-09-16 -``` - -Both commands print versioned JSON with the application and pin IDs, creation -time, control count, nonempty catalog-shard count, release-snapshot digest, and -`verified: true`. Creation is idempotent by pin ID. Verification rereads the -release metadata, catalog pages, canonical controls, and every immutable LTX -dependency; it does not trust local SQLite files or caches. - -Restore accepts only a canonical prefix different from the configured source -root and only a pin whose selected release was `Ready`. The command verifies -the source graph, uses same-bucket conditional copies, re-verifies every -destination root, removes captured owners from restored controls, and publishes -the destination release and pin pointers last. Run it while the destination is -offline; an exact interrupted attempt is resumable, but a destination used by a -fleet has intentionally diverged and is rejected. - -Start the restored fleet with the compiled release named by the pin. Its first -request acquires each `Idle` Cell and rebuilds disposable SQLite files from the -exact root. Repository catalog configuration, Git/Xet/LFS objects, release -assets, and other product data outside `cells/v1` are not Cell backup contents; -restore or reference those through their owning runbooks. Cross-provider -archive export still requires a separate transport step. - -## Apply the repository hard cut - -There is no legacy application-data importer. - -```mermaid -flowchart LR - Stop[Stop legacy writers] - Delete[Delete retired app/v1 collaboration data] - Keep[Keep Git, Xet, LFS, and asset objects] - Adopt[Adopt each repository] - Verify[Verify empty Cell root] - Serve[Enable Cell-backed routes] - - Stop --> Delete --> Keep --> Adopt --> Verify --> Serve -``` - -Do not enable dual-read, dual-write, or fallback behavior. Repository adoption creates and verifies new empty collaboration state. - -## Monitor the fleet - -At minimum, export these metric groups: - -| Group | Signals | -| --- | --- | -| Ownership | Active Cells, renewals, self-fences, takeover duration | -| Commands | Accepted, committed, rejected, unknown, resolved, deadline exceeded | -| Publication | Prepare latency, CAS conflicts, ambiguous CAS adoption, compaction | -| Storage | Object requests, verified bytes, sparse faults, cache hits, disk reservations | -| Scheduler | Pass duration, due lag, progress counter, excluded sessions | -| Activities | Claims, lease loss, heartbeat, retry, panic, duration | -| Capacity | Free memory, free disk, job credits, file descriptors, rejected admission | -| Releases | State, eligible nodes, pending migrations, terminal migration failures | - -Alert on stalled scheduler progress, repeated owner fencing, publication backlog, control renewal delay, disk reserve breaches, and release-state exclusion. diff --git a/crates/crab-cell-runtime/docs/diagram/system-architecture.svg b/crates/crab-cell-runtime/docs/diagram/system-architecture.svg deleted file mode 100644 index ad73bf800..000000000 --- a/crates/crab-cell-runtime/docs/diagram/system-architecture.svg +++ /dev/null @@ -1,155 +0,0 @@ - - Crab embedded Cell runtime architecture - A client enters any Crab HTTP server. The repository Cell router resolves the authoritative owner and runs locally or forwards over mutual TLS. The owning node executes a typed Rust handler through one Cell actor and SQLite writer, then publishes a checksum-verified LTX root and fenced control record to object storage. - - - - - - - - - - - - - - - - - - - - - - - Crab embedded Cell runtime - One product binary · one active SQLite writer per Cell · object storage is the durable authority - - - ANY CRAB NODE · PUBLIC INGRESS - - - AUTHORITATIVE OWNER NODE - - - SHARED OBJECT STORE - - - - - - - - - - - - - - - - HTTP / Git - local owner - remote owner - private mTLS - resolve / CAS - immutable publish - one writer - - - - - Client - browser · Git · API - - - - - crab-http-server - public routes · auth · repository policy - - - - Repository Cell router - read authority · admit · route once - - - - Authenticated peer client - bounded request · current membership - - - - - Private peer endpoint - mTLS · release-aware dispatch - - - - Compiled Rust registry - typed command · codec · schema - - - - Cell actor - FIFO admission · fencing · replay - - - - SQLite + WAL - runtime + application transaction - - - - crab-ltx - capture · verify - - - - - control.json - owner · epoch · exact root · ETag CAS - - - - Immutable LTX graph - roots · frames · page directories · bundles - - - - Checksum-verifiable bytes - durable origin · node disks are caches - - - - ROUTE - - Any node may receive traffic; authority decides local - execution or one authenticated peer hop. - - - - COMMIT - - A successful mutation names the exact SQLite root - durably published by its owner. - - - - RECOVER - - A fenced successor restores the authoritative root; - peer disks are never durability. - - diff --git a/crates/crab-cell-runtime/docs/diagram/system-architecture@2x.png b/crates/crab-cell-runtime/docs/diagram/system-architecture@2x.png deleted file mode 100644 index e2ee9176b..000000000 Binary files a/crates/crab-cell-runtime/docs/diagram/system-architecture@2x.png and /dev/null differ diff --git a/crates/crab-cell-runtime/docs/failover-and-followers.md b/crates/crab-cell-runtime/docs/failover-and-followers.md deleted file mode 100644 index 0e58d238b..000000000 --- a/crates/crab-cell-runtime/docs/failover-and-followers.md +++ /dev/null @@ -1,1715 +0,0 @@ -# Follower durability and warm failover - -Crab implements a Celld-style replicated node log around the existing per-Cell -SQLite/LTX runtime. The implementation keeps exactly one Cell owner, lets one -or two other nodes durably retain the owner's recent LTX cuts, and recovers -those cuts before a successor opens SQLite. - -| Document intent | Value | -| --- | --- | -| Content type | Low-level target design | -| Audience | `crab-ltx`, `crab-cell-runtime`, and `crab-http-server` implementers | -| Goal | Define the persistence, wire, gating, recovery, lifecycle, and proof contracts needed for Celld-style follower durability | -| Status | Follower durability, bounded recovery, follower-affine takeover, local fast paths, and digest-bound qualification implemented; protected scale, provider, and release runs remain | -| Reference | Celld commit `10cb1303dac710dcb3b557e318e08c855261f68b` | - -[Back to the Cell runtime index](README.md) - -## Decide what “warm” means - -The follower tier is a **durability log**, not a second SQLite owner. - -| Kind of warmth | Target behavior | Is a follower an owner? | -| --- | --- | --- | -| Durable tail | One or two peers fsync recent LTX cuts | No | -| Fast activation | Successor opens the exact root with sparse paging | No, until ownership CAS succeeds | -| Page cache | Immutable pages may already exist in a verified local cache | No | -| Hot SQL standby | Another node keeps a writable SQLite connection open | Not supported | - -This distinction preserves one writer while removing object-store upload -latency from the common response path. It does not create read replicas, allow -follower reads, or permit a secondary to accept writes. -The separate read-only exact-root query capability in -[Plan 036](../../../advisor-plans/036-cell-read-replicas-and-fenced-promotion.md) -has a private peer query path and an explicit public issue-detail route in object -durability mode. Other product reads still use the owner. It is independent of follower durability. A durability-log -follower still cannot answer SQL queries or promote without the Cell control CAS. - -```mermaid -flowchart LR - Client[Client] - Owner[One Cell owner
SQLite + actor] - Gate{Durability gate} - F1[Follower A
fsynced node-log tail] - F2[Follower B
fsynced node-log tail] - Bucket[(Object store
exact roots + recovery bundles)] - Reply[Release response] - - Client --> Owner - Owner --> Gate - Gate --> F1 - Gate --> F2 - Gate --> Bucket - F1 -->|all selected followers ack| Reply - F2 -->|all selected followers ack| Reply - Bucket -->|root CAS proven| Reply -``` - -Crab targets Celld's public behavior: a write can complete after a fleet proof -or an object-store proof; a takeover must recover an earlier fleet proof before -restore. Crab retains its own verified manifests, BLAKE3 identities, exact-root -controls, `crab-storage` adapters, and Rust-native runtime. - -## Preserve these guarantees - -The implementation is acceptable only when all of these statements remain -true: - -1. Exactly one owner session and Cell epoch can serve a Cell. -2. A successful mutation is recoverable after loss of its owner process and - owner-local disk when at least one complete selected follower survives, or - when the exact root reached object storage. -3. A response cannot reveal a SQLite state newer than the durability proof that - released it. This includes successful mutations, durable business errors, - reads performed after a mutation in the same serialized actor, and every - chunk emitted by `CellStateStream`. -4. A successor cannot open SQLite until the predecessor node-log session is - absent-with-proof or sealed and every recovered tail is pinned by Cell - control. -5. A stale owner can upload immutable bytes, but it cannot change Cell control, - complete a new fleet proof after sealing, or release an ungated response. -6. Every recovered segment is checked for framing, BLAKE3, LTX structure, - transaction continuity, pre/post database checksum, Cell identity, - incarnation, Cell epoch, and commit sequence. -7. Loss of all durability evidence fails closed. The runtime never advances a - root, seals a log, or reports success by assuming missing data was empty. -8. Followers persist bounded recent tails. They do not mirror 100 MB to 5 GB - databases or multiply one SQLite process per Cell. - -The availability failure model is one owner-node loss. Simultaneous loss of the -owner and every selected follower can make a Cell unavailable until one copy -returns. No protocol can claim RPO=0 after every acknowledged copy is destroyed. - -The proof obligations can be reviewed as four implications. `covers` includes -the command's ledger outcome, not just application pages. - -```text -response(commit) => object_root_covers(commit) OR fleet_covers(commit) - -fleet_covers(commit) => session.log.active - AND every selected member durable_through >= ticket(commit) - -session.log.sealed => every fleet-covered frame is object-covered - OR pinned by an exact Cell recovery overlay - -cell.serving => cell.root consumes every attached recovery overlay - AND the serving owner/session/epoch CAS is current -``` - -Tests and model checks should assert these implications directly instead of -inferring them from task completion or log messages. - -## Compare current and target behavior - -| Boundary | Current implementation | Target extension | -| --- | --- | --- | -| Owner authority | Per-Cell owner, epoch, and root CAS | Same Cell fence, backed by an authoritative node-session lease | -| Success response | Exact immutable root uploaded and CASed into control | First valid proof wins: exact root CAS or all-follower fsync | -| Recent data | Owner-local retained cuts plus object-store graph | Also retained in a multiplexed follower node log | -| Takeover | Wait for unchanged control, increment epoch, restore `control.root` | Prove predecessor session dead, seal/recover its node log, attach recovery overlays, then acquire and restore | -| Activation | Sparse writable root and background hydration | Same; recovered overlay becomes part of the exact root first | -| Clean handoff | Publish, close, release | Publish all outstanding cuts, seal or advance the node log, close, release | -| Local restart | Fresh exact-root restore | Same correctness path; verified page cache may reduce reads | - -`crab-ltx` already supplies capture, checksum-bearing segments, bundle extents, -cross-epoch continuation, exact roots, sparse writable activation, and hydration. -The missing capability is the node-level durability protocol around those -mechanics. - -## Place responsibilities at the right layer - -```text -crab-ltx - inspect and stream one captured segment - validate recovered segment metadata and bytes - build an exact successor root from a verified recovery overlay - expose immutable bundle extents without cluster policy - -crab-cell-runtime - node-session lease and terminal self-fence - node-log state machine, follower store, shipper, and recovery coordinator - durability tickets and response gate - Cell recovery-overlay attachment and consumption - bounded admission, shutdown, metrics, and fault injection - -crab-http-server - construct the runtime from existing storage and peer configuration - expose private mTLS append, seal, and tail routes - order startup/readiness/drain and map typed errors to public HTTP - never expose a public follower API -``` - -`crab-ltx` must not select nodes, own leases, authorize peers, or decide when an -HTTP response is safe. `crab-http-server` must not parse LTX or create a second -recovery path. - -### Implementation checkpoint - -| Working now | Remaining target gaps | -| --- | --- | -| Strict frame codec plus capacity- and failure-domain-aware deterministic selection, retrying automatic enrollment, activation, coverage, recovery claims, object-covered epoch rotation, and clean log close | Signed small/medium/large live runs and the extended fault/telemetry matrix | -| Crash-safe, node-budgeted follower store under a persisted physical `NodeId`, authenticated remote append/seal/tail/retire transport, a bounded node-wide batched shipper, a recovery-first management-listener lifecycle, startup lane scrub/quarantine, and crash-rebuildable bounded lane indexes | Protected large-tail, large-catalog, and mixed-failure evidence remains; the index is derived acceleration data and every selected record is reread and digest-checked | -| Authoritative create and refresh drive a terminal monotonic node-lease guard; admission, actor dispatch, Cell-control CAS, durability proof, and output acceptance all check it | None for the current non-streaming Cell API | -| Write-all durability gate, first-fsynced-batch activation, bounded dual-watermark command continuation, ordered object publication, object fallback, schema-migration barriers, and contiguous authoritative object watermark | None for this slice | -| Complete-witness grouping, immutable recovery manifests, post-pin session seal CAS, non-forgeable persisted takeover proof, and bounded automatic dead-session recovery with renewable claims | None for this slice | -| Cell control attachment and takeover consumption of overlays; server drain closes a fully object-covered epoch before session withdrawal; grace-aged retired follower lanes are deleted only after authority stops naming their epoch; the Compose qualifier proves a follower-only result survives owner `SIGKILL`, owner-disk deletion, RustFS restoration, takeover, and owner rejoin; the Kubernetes harness exercises each selected node profile and the 1,000 aggregate mutation schedule against every Pod across eight load Cells, and its owner-loss receipt binds the successor stable NodeId to the failed log's original follower set | Signed live runs across small/medium/large profiles plus the extended fault/telemetry matrix | -| Bounded command/query responses and the typed `CellStateStream` bind every emitted chunk to the actor's proven logical head | Extended live fault and profile qualification only | - -The session record now owns one CAS-protected log epoch, its exact sorted member -set, activation bit, contiguous object watermark, and renewable recovery claim. -The private mTLS transport implements enrolled append plus claimant-authorized, -page-bounded seal and tail operations. The follower store admits every append -and seal against the same node-level disk budget used by Cell work, reserves -existing bytes on restart, and NACKs before writing when capacity is exhausted. -Each data directory strict-creates one durable `node-id`; boot sessions remain -ephemeral. Log membership records stable physical node IDs, and each request -resolves that ID to exactly one current live session. Two overlapping live -sessions for one physical node fail closed. -`NodeLogShipper` reserves encoded bytes before assigning a sequence, multiplexes -accepted cuts in submission order, batches for at most one millisecond or 64 -frames, sends each batch to every member concurrently, and advances the gate -only after all receipts cover the batch. Encoding, transport, or receipt -failure stops fleet issuance for that epoch while its tickets remain eligible -for object proof and covered rotation. -The directory now filters live peers by protocol, pressure, and the exact -shared-disk capacity advertised by their follower stores. It greedily maximizes -proven zone separation, then proven host separation, then applies the owner- -session/physical-node rendezvous rank for the full one- or two-member ensemble -before its CAS enrollment. Unknown topology labels receive no separation credit -instead of being assumed independent. Rotation closes -the old gate only after every issued sequence is object-covered, best-effort -retires old lanes behind durable append fences, and CASes a fresh inactive -epoch. Recruitment retries while the node remains healthy and leaves a -one-node fleet on the object path. -The preferred -shard-zero scanner now inventories expired active node -logs with at most 32 concurrent record reads, shares a verified one-second -directory snapshot across cloned schedulers, selects a bounded rotating window, -claims at most two concurrently, scans at most 10,000 affected Cells, renews -each recovery claim while gathering and pinning, refreshes the claim once more -before the final seal, and bounds that seal's object-store CAS so a stalled -store returns a retryable deadline instead of holding an unbounded recovery -task. The snapshot is advisory: each claim reloads the failed session and -revalidates the claimant's live signed recovery admission immediately before -its fencing CAS. Failed sessions use bounded in-memory exponential retry, capped -below the 30-second claim lifetime, so an unavailable object store cannot keep -all recovery workers hot or starve later sessions; the authoritative claim is -still the only ownership record. It leaves a takeover proof that another -request can reload. For -commands, the -actor keeps complete local cuts readable without making their directory entries -durable, then submits those exact bytes before immutable-root preparation. -Every selected follower must fsync the ticket before the shared node-session -authority performs the exact `active=false -> active=true` CAS. Only then can -fleet proof release the command response. The actor advances a logical head -after that proof and may execute the next command while one separate publisher -advances the exact object-backed root in order. A 64-entry queue and a 64 MiB -retained-byte high water apply backpressure; the existing local-disk budget -remains the hard byte admission boundary. Failure of the fleet path falls back -to object proof, while a terminal publication failure fences the Cell and -leaves any already released outcomes recoverable from the node log. -Schema-migration cuts use the same follower/object race and recovery -coverage. A successful fleet proof may release the successor handle before -object publication; its admission is already installed, so requests queue -behind the publication barrier. Migrations, drain, and shutdown do not cross a -command backlog; object-only migrations continue to wait for exact root -publication. - -If immutable-object publication returns a storage error after the fleet proof, -the ordered publisher retains the cut and retries with bounded backoff for a -short grace period. The logical head may serve the fleet-proven result while -the published head catches up. Lease loss, a control conflict, or exhaustion -of that grace period fences the Cell; the unpublished node-log interval stays -owner-pinned so takeover can seal and replay it. -An active predecessor log cannot be converted directly from a session fence -into Cell takeover authority: only the coordinator's successful post-seal -result carries `NodeTakeoverProof`. - -The HTTP node publisher now arms a process-wide monotonic lease guard only -after its session create succeeds and advances it only after an authoritative -refresh. Expiry and refresh failure are terminal: both mark the node unhealthy -and cancel the server, and a late refresh cannot revive the process. The -production Cell runtime stays fenced until that guard is installed. It checks -the same guard before admission, immediately before actor dispatch, around -Cell-control mutation, and before returning any state-observing result. -Heartbeat refresh, log activation, object coverage, and clean close share one -mutex-protected authoritative observation, so their ETag CAS operations cannot -race through stale local state. - -Retired follower lanes keep their durable append-fence marker for ten minutes. -The server then scans at most 64 lanes per minute, requires the exact -node-session record to exist and no longer name that log epoch, rechecks the -unchanged marker and its filesystem timestamp under the lane lock, and only -then deletes it and releases disk admission. Missing authority fails closed. - -Deterministic fault coverage includes the two ambiguous recovery boundaries: a -follower may fsync a frame and lose its ACK without authorizing a fleet proof, -and an expired recovery claim may move to a new live claimant while permanently -fencing the old claimant's renewal. It also closes and reopens every follower -in an ensemble before gathering the witness, and discards a recovery -coordinator after overlay attachment before a new coordinator resumes sealing. - -Correctness boundaries exercised by regression tests: - -| Race or fault | Required behavior | -| --- | --- | -| Two Cells finish object publication concurrently | Serialize coverage preview, authority CAS, and local confirmation; persisted coverage cannot trail an acknowledged local truncation watermark | -| A queued batch contains an object-covered prefix | Accept an ACK retaining every uncovered suffix frame; reject an empty retained range for an uncovered ticket | -| An append requests truncation ahead of authority | Reject before follower storage changes; stale, lower watermarks remain safe | -| Shipping stops before the frame-count threshold | Trigger the same object-coverage barrier and follower re-enrollment used by normal rotation | -| Close clears the old log before re-enrollment | Allocate the new epoch from the monotonically increasing authoritative node generation; never reset it to one within the same session | -| Another Cell holds back global object coverage | Remove each Cell's already-rooted prefix using its exact commit sequence and LTX position; reject contradictory positions/checksums | -| All retained frames are already rooted | Seal without creating an empty recovery manifest; accept a sealed empty lane when every skipped frame is object-covered | -| Two witnesses return different valid bytes for one sequence | Fail closed, including overlapping evidence from shorter or partially readable witnesses | -| A valid frame names another session or epoch | Reject it before building a recovery overlay | -| A recovered suffix is awaiting immutable pinning | Retain its recovery admission until the result is pinned or discarded | -| Recovery must discover affected Cells | Derive authenticated Cell scopes from the sealed tail, read only affected catalog shards once, and revalidate each current Cell control | -| The claim CAS commits but its response never returns | Bound the storage wait, then resume the same persisted claim idempotently on the next scheduler scan | - -## Use one multiplexed log per owner session - -A node can own 1,000 to 10,000 active databases. Full per-repository standbys -would multiply SQLite memory, file descriptors, hydration, and checkpoint work. -Instead, every owner session assigns one increasing sequence across captured -cuts from all of its Cells. - -```mermaid -flowchart TB - subgraph Leader[Owner session S7] - C1[Cell A cuts] - C2[Cell B cuts] - C3[Cell C cuts] - M[Ordered multiplexer
sequence 101, 102, 103] - end - subgraph Follower1[Follower session F1] - L1[One S7 log
many Cells] - end - subgraph Follower2[Follower session F2] - L2[One S7 log
many Cells] - end - - C1 & C2 & C3 --> M - M --> L1 - M --> L2 -``` - -Ordering is by submission to the node-log shipper, not task polling order. -Each Cell still has its own contiguous LTX transaction chain and commit -sequence. The node sequence supplies follower replay, truncation, and recovery -coverage across interleaved Cells. - -## Evolve the unshipped formats in place - -The Cell storage contract is still under active development and has no released -data to preserve. The implementation therefore keeps the existing `cells/v1` -prefix and evolves the current control, session, and reader/writer structures -together. It does not create a `cells/v2` namespace merely because fields or -state transitions change. - -There is one canonical format at every commit. Development environments may be -discarded and recreated when that format changes. There is no dual write, -fallback reader, compatibility branch, or data migration until Crab ships a -persistent Cell format that explicitly requires those guarantees. - -This applies to both the `cells/v1` path and the `version: 1` fields inside its -documents. During development those values remain stable while the only reader, -writer, validation rules, fixtures, and diagrams change together. They are -format identity guards, not counters to increment for each structural edit. - -```text -/cells/v1/ - identity.json - sessions/.json - node-logs/// - recovery/.json - bundles/.bundle - apps// - catalog/... - releases/... - cells// - control.json - inc//objects/... -``` - -Content-addressed Cell objects remain under the application and incarnation. -Node-log recovery bundles are cross-Cell and therefore live outside an -application prefix. Every row inside a recovery bundle carries its application -ID so recovery can route it to the correct `CellStorageLayout`. - -## Make the node session authoritative - -The current signed node advertisement is mainly a routing and capacity record. -Fleet proofs require a node-session lease that a recoverer can atomically move -out of `live`; otherwise a paused owner could continue obtaining follower acks -while takeover reads its Cells. - -The evolved session object separates signed immutable identity from -CAS-protected mutable authority: - -```json -{ - "version": 1, - "identity": { - "fleet": "32-byte-hex", - "node": "16-byte-hex", - "session": "16-byte-hex", - "endpoint": "https://node-a.internal:8081", - "certificate": "32-byte-hex", - "public_key": "32-byte-hex", - "image": "32-byte-hex", - "release": "32-byte-hex", - "failure_domain": { - "zone": "us-west-2a", - "host": "worker-17" - }, - "peer_versions": [1], - "signature": "64-byte-hex" - }, - "lease": { - "state": "live", - "generation": 41, - "expires_at_ms": 1789600000000 - }, - "log": { - "state": "open", - "epoch": 3, - "members": ["physical-node-a", "physical-node-b"], - "active": true, - "tiered_through": 9001, - "recovery": null - }, - "capacity": { - "sampled_at_ms": 1789599997000, - "active_cells": 2048, - "follower_free_bytes": 21474836480, - "log_protocol": 1 - } -} -``` - -`node` identifies the durable local data directory; `session` identifies only -one boot generation. The identity signature covers the canonical `identity` -fields and the top-level capacity snapshot. Heartbeats re-sign changed -capacity without changing the boot identity; lease and log state remain -CAS-protected mutable authority. The whole object is still protected by its -object-store ETag. The owner may renew only a `live` record with the exact -session and generation. A recoverer may change only recovery-owned fields after -expiry. Every transition validates all unchanged fields before conditional -overwrite. - -| Session state | May route application work? | May append to its follower log? | May a peer recover it? | -| --- | --- | --- | --- | -| `live` before published expiry | Yes | Yes | No | -| `live` after published expiry | No | No new proof | Yes, by CAS claim | -| `recovering` | No | No | Only the live claimant | -| `sealed` | No | No | Recovery already complete | -| `retired` | No | No | No; overlays are already consumed | - -The node renews at three seconds and uses a ten-second published lease by -default. The session watchdog closes all Cell admission and terminates the -process when it cannot prove a valid lease before expiry. A late renewal cannot -revive a fenced process. Kubernetes or another supervisor starts a new session. - -Per-Cell `progress` remains a monotonic publication field, but takeover no -longer infers owner death from a quiet Cell. A quiet repository can be healthy -for months. Only the exact owner session's lease, or a graceful release, permits -takeover. - -### Open the node log before its first fleet proof - -A fresh session is strict-created with `log=null`. That is a durable statement -that the session has never acknowledged beyond object storage. Recruitment then -CASes a complete `open` log with `active=false`, a nonzero log epoch, and its -member set before sending any frame. - -```mermaid -stateDiagram-v2 - [*] --> Absent: fresh session, log is null - Absent --> OpenInactive: CAS recruited members - OpenInactive --> OpenActive: followers fsync and active CAS wins - OpenInactive --> Sealed: no fleet proof was ever credited - OpenActive --> OpenActive: append or advance tiered-through - OpenActive --> Reconfiguring: stop new fleet tickets - Reconfiguring --> OpenInactive: all old sequences object-covered, next log epoch - OpenInactive --> Recovering: session expired - OpenActive --> Recovering: session expired - Recovering --> Sealed: tails pinned to Cell controls - Sealed --> Retired: every affected Cell covers its overlay -``` - -For the first candidate fleet proof, followers fsync the batch, then the owner -CASes `active=false` to `active=true`, then it credits the proof. If the active -CAS is ambiguous, it reloads and accepts only that exact successor. If it fails, -the batch waits for object proof. Consequently, `active=true` means recovery -must find a complete witness or fail closed; `active=false` permits sealing -without one because no fleet proof could have escaped. - -`tiered_through` is the largest **contiguous node sequence** for which every -earlier frame has an object-store proof. Per-Cell publishers may finish out of -order; the node-log manager holds those completions until the contiguous prefix -advances, then CASes the session record and tells followers what they may -truncate. A single later Cell root cannot create a hole in this watermark. - -## Extend Cell control with a recovery overlay - -An object-store root can lag a fleet-durable write. The successor therefore -needs a durable pointer to the recovered tail before the node log can be sealed. - -```rust,ignore -struct RecoveryOverlayRef { - leader_session: SessionId, - log_epoch: u64, - manifest_digest: Digest, - first_node_sequence: u64, - last_node_sequence: u64, - predecessor: RootRef, - final_txid: u64, - final_checksum: u64, - final_commit_sequence: u64, -} - -struct Control { - // Existing Cell, incarnation, epoch, revision, owner, root, code, - // schema, state, and next_due fields remain. - recovery: Option, -} -``` - -Only the recovery coordinator can attach an overlay. The transition requires: - -- The control still names the dead leader session and captured Cell epoch -- The current root exactly equals the overlay's declared predecessor -- The node-log recovery claim names that leader session and log epoch -- The overlay object exists and passes bounded structural verification -- The successor changes only `revision`, `progress`, and `recovery` - -Takeover carries the overlay unchanged into `Recovering`. The new owner asks -`crab-ltx` to verify and materialize a successor root, then CASes that exact -root while clearing `recovery`. A Cell cannot become `Serving` while an overlay -is present. - -This extra pointer solves two problems at once: retention can see recovered -bytes before activation, and exact-root restore never depends on listing an -epoch prefix. - -The target transition table is explicit: - -| Transition | Required predecessor | Protected effect | -| --- | --- | --- | -| `AttachRecovery` | Dead owner session, same Cell epoch/root, no different overlay | Pins one exact recovered tail without changing owner | -| `Takeover` | Predecessor session sealed or `log=null`; overlay unchanged | Advances Cell epoch and installs the successor in `Recovering` | -| `PublishRecovery` | Successor owns `Recovering`; exact overlay predecessor/final position | Advances root and clears the overlay | -| `Activate` | Successor owns `Recovering`; root exists; no overlay | Makes the verified local database externally serving | -| `Publish` | Live owner session; no overlay; exact root successor | Advances the normal published root | -| `Release` | Live owner, drained SQL, logical head equals published head | Removes owner and enters `Idle` | - -Normal publication is forbidden while `recovery` is present. A stale owner's -already-issued Cell CAS can win only before the recovery coordinator changes -that Cell control; recovery then reloads the newer root and deterministically -filters or rebases the overlay. Once `AttachRecovery` or `Takeover` wins, every -older Cell ETag is invalid and the stale publisher fences on reconciliation. - -## Define the node-log frame - -The transport sends a strict binary envelope followed by the unchanged LTX -bytes. Integer fields are little-endian. Variable byte fields are length -prefixed. Decoding rejects unknown versions, duplicate fields, noncanonical -lengths, oversized payloads, and trailing bytes. - -```rust,ignore -struct NodeLogFrameV1 { - leader_session: [u8; 16], - log_epoch: u64, - node_sequence: u64, - application: [u8; 16], - cell: [u8; 32], - incarnation: [u8; 16], - cell_epoch: u64, - commit_sequence: u64, - segment: crab_ltx::SegmentInfo, - body_len: u64, - body_blake3: [u8; 32], - body: Bytes, -} -``` - -One SQLite command may produce multiple cuts at checkpoint boundaries. Its -durability ticket names the last node sequence in that command's consecutive -frame range. A follower acknowledgement covers the whole contiguous range, not -individual Cells. - -Before writing, a follower verifies: - -1. The mTLS peer and signed session identity match `leader_session` -2. The log epoch and member set match the session record it joined -3. `body_len` fits the 64 MiB captured-segment limit -4. BLAKE3 matches the body -5. `crab-ltx` inspection matches every declared `SegmentInfo` field -6. The sequence is the expected next value or an exact duplicate - -Full per-Cell predecessor validation happens when building the recovery -overlay. A follower cannot cheaply hold every Cell root just to validate an -interleaved append. - -## Store follower fragments crash-safely - -Follower storage is local SSD cache with a durability obligation. It is not an -evictable read cache until the session record proves the bytes are covered. - -```text -/node-id -/followers/// - retired - chunks/ - 00000000000000000001-00000000000000004096.log - open.log -/followers-quarantine/.bad -``` - -Each record in a chunk contains magic, sequence, encoded-frame length, the -canonical frame digest, and frame bytes. Startup scans the active chunk, -re-verifies the frame digest and LTX body, and truncates only an invalid suffix -after the last completely verified record. - -Append handling is ordered per leader/log epoch: - -```mermaid -sequenceDiagram - participant L as Leader shipper - participant F as Follower store - participant D as Local SSD - - L->>F: Append batch [101..108], covered_through=96 - F->>F: Authenticate + validate every frame - F->>D: Append canonical records - F->>D: sync_data active chunk - F->>D: Atomically persist base/end when rotating - F-->>L: durable_through=108, base=97 -``` - -The acknowledgement is sent only after `sync_data` succeeds. Directory sync is -also required when creating, rotating, renaming, or removing a chunk. A disk -error, short write, checksum mismatch, gap, or sync error returns a typed NACK -and never advances `durable_through`. - -`FollowerStore::open` walks every retained lane before the management listener -starts. It verifies directory shape, closed-chunk names and records, frame -scope and digest, sequence continuity, and seal/retire watermarks. A torn or -invalid suffix in `open.log` is truncated to its last fully verified record and -synced. Any other invalid lane is atomically renamed into -`followers-quarantine`; the server reports the persisted quarantine count and -keeps those bytes charged to the same disk budget. Quarantine is diagnostic -and has no automatic deletion path. Because the corrupt lane is no longer a -recovery witness, an active leader log with no other complete member remains -unavailable rather than treating the damage as an empty tail. - -Exact duplicates return the existing durable end after comparing the stored -digest. A duplicate sequence with different bytes is corruption and -quarantines that leader lane. A future sequence returns `expected_sequence` -without filling the gap. - -Followers delete only chunks at or below `covered_through`. Whole-epoch -retirement first fsyncs the eight-byte `retired` watermark and its directory, -then removes the chunks. The marker permanently rejects old-epoch appends and -lets recovery prove that the now-empty lane was fully object-covered even if -the leader crashes before its rotation CAS. `covered_through` comes from the -authoritative session record, never from leader memory. - -## Use bounded ordered peer streams - -The existing private mTLS listener gains four node-log operations: - -| Operation | Direction | Purpose | -| --- | --- | --- | -| `OpenAppendStream` | Leader to selected follower | Long-lived ordered batches and ordered acknowledgements | -| `SealFragment` | Recoverer to follower | Stop appends for one leader/log epoch and return retained range | -| `ReadTail` | Recoverer from follower | Stream the sealed retained range with checksums | -| `RetireFragment` | Live leader to follower | Persist an append fence and delete one fully object-covered epoch | - -The peer descriptor remains message-only; no public service is generated. The -HTTP server maps messages onto private routes such as -`/internal/cells/v1/node-log/.../append`, `/seal`, `/tail`, and `/retire`. - -Initial protocol bounds are compile-time contracts: - -| Bound | Value | -| --- | ---: | -| One LTX frame body | 64 MiB | -| One append batch | 64 frames or 64 MiB, whichever comes first | -| Outstanding batches per follower lane | 8 | -| Tail page | 1 MiB or 4,096 frames; one individually bounded frame may exceed 1 MiB | -| Append/follower request deadline | Remaining caller deadline, at most 30s | -| Recovery claim heartbeat | 10s | -| Recovery claim expiry | 30s | - -The leader applies backpressure before the window fills. It does not spawn one -task or connection per Cell. One lane per selected follower carries all Cells -for that owner session. - -`tail_page` is the recovery path used by the current implementation. Its -default transport adapter can page a legacy `tail` result in memory. The HTTP -transport and `LocalFollowerTransport` expose bounded pages. The follower disk -reader scans one frame at a time, retaining verified file locations and digests, -then materializes only the requested page. It rechecks the digest after seeking. -Metadata remains proportional to retained frame count, and each page still -performs a full validation scan: payload memory is bounded, not total scan I/O. -A sealed recovery witness is reduced to one authenticated generation per -affected Cell before catalog lookup, so scope-validation memory is bounded by -the affected-Cell admission limit rather than the retained tail length. -A page with -one frame may be larger than the 1 MiB network target, but that frame is still -bounded by the configured capture limit; a multi-frame page may not exceed the -target. Recovery rejects oversized, non-contiguous, or unverifiable pages -before attaching any overlay. The large-single-frame regression is covered by -`node_log_recovery::tests::active_lane_requires_and_returns_a_complete_follower_tail`. - -## Select and change the follower ensemble - -The owner chooses followers from live, release-compatible physical nodes whose -current boot sessions advertise the node-log protocol and available follower -bytes. - -Selection rules, in order: - -1. Exclude the owner's physical node -2. Exclude draining, pressured, stale, protocol-incompatible, or ambiguously - advertised nodes -3. Prefer a different zone, host, and local-disk failure domain -4. Rank by rendezvous hash of owner session and candidate physical node ID -5. Select one follower in a two-node fleet and two in a fleet of three or more - -Every selected member must fsync a batch for a fleet proof. This is write-all, -ack-all. Quorum acknowledgement is not safe because recovery is designed to use -one complete surviving witness, not merge partially acknowledged quorums. - -Zone and host are optional signed boot-identity labels. A missing label cannot -prove separation and therefore receives no preference over a known unequal -label. The persisted physical `NodeId` identifies the local-disk domain; live -inventory rejects duplicate physical IDs, and the selector never chooses the -owner or the same physical node twice. - -Changing members uses a barrier: - -1. Stop assigning new fleet tickets to the old shipper -2. Wait until every submitted sequence is covered by an exact object-store - proof -3. While the old authority is still verifiable, ask reachable old followers to - persist the exact covered watermark and remove that lane -4. CAS the session record to the next log epoch and new member set -5. Open new follower lanes with sequence one - -If a member fails, in-flight writes can still complete through object-store -publication. The owner must not silently shrink the current write-all set while -uncovered entries exist. An unreachable old follower does not block rotation: -it retains inert data, rejects future appends after the authority CAS, and is -collected later. A successful retirement response with any other watermark is -a protocol error and blocks the CAS. - -The long-lived HTTP runtime applies the same barrier when shipping stops or the -current epoch reaches `1_000_000` issued node-log frames. A five-second controller observes -the active binding, closes it through the idempotent -`NodeDurability::shutdown`, and retries `PendingPublication` until object -coverage is contiguous. It then recruits the next epoch and atomically -replaces the runtime binding. New Cell submissions read the current binding at -the start of each durability attempt; a replacement therefore cannot create a -second SQLite writer or a second Cell-control CAS owner. If recruitment is -temporarily unavailable, the server keeps serving through the object proof -path and retries while the node lease remains healthy. A shutdown or lease -fence cancels the controller and closes whichever binding is current. - -## Release responses through one gate - -Each SQLite commit produces a monotonic local `CommitTicket`. The ticket binds -the Cell commit sequence, captured cuts, result ledger row, and the last assigned -node-log sequence. - -```rust,ignore -enum DurabilityProof { - ObjectStore { - root: crab_ltx::RootRef, - control_revision: u64, - }, - Fleet { - leader_session: SessionId, - log_epoch: u64, - durable_through: u64, - members: Vec, - }, -} - -struct DurableCommit { - value: T, - receipt: Receipt, - proof: DurabilityProof, -} -``` - -After capture, the actor submits the same cuts to two concurrent paths: - -```mermaid -sequenceDiagram - participant H as HTTP handler - participant A as Cell actor - participant S as SQLite - participant F as Followers - participant O as Object store - - H->>A: Typed command + request ID - A->>S: Commit outcome ledger and application rows - S-->>A: Capture batch + commit ticket - par Fleet path - A->>F: Ordered frames - F-->>A: All members fsynced through ticket - and Object path - A->>O: Upload verified objects - A->>O: CAS exact Cell root - O-->>A: Published root proof - end - A-->>H: First valid proof releases DurableCommit -``` - -The slower path continues as owned background work. If the fleet wins, object -tiering must eventually advance `control.root`; dropping that task would turn a -short follower tail into permanent primary storage. - -Fleet proof does not perform a per-write object-store owner read. Its safety -comes from the session interlock: the log was made active before the first -proof, takeover changes the expired session to `recovering`, and every follower -serializes append with seal. A local dispatch still checks the process-wide -lease guard before entering an actor, and the output gate checks it before -crediting a newly completed proof. A process pause cannot use an expired cached -deadline to admit more work. - -The actor keeps one SQLite writer and one object publisher. Fleet proof may -release a command and advance the local logical head before object upload -finishes, but the actor retains exclusive ownership and the publisher performs -every root CAS in commit-sequence order. Reaching a backlog high water pauses -new commands until publication catches up; it never creates another writer or -weakens durability. - -### Use a bounded dual-watermark pipeline for hot Cells - -Fleet durability removes the object-store round trip from response latency, but -the baseline still leaves that round trip between two commands on the same -Cell. That is acceptable for a fleet whose traffic is spread across many -repositories, but it imposes an unnecessary per-repository throughput ceiling. -The target therefore separates two positions without creating a second owner. -The shorter name **dual-head** refers only to these publication watermarks; it -does not mean dual primary, two SQLite writers, or two control authorities. - -| Position | Meaning | May accept a new command? | -| --- | --- | --- | -| `logical_head` | Latest local SQLite commit covered by fleet or object proof | Yes | -| `published_head` | Exact immutable root named by Cell control | Yes, while backlog admission remains available | - -```mermaid -flowchart LR - C1[Commit N] --> P1[Fleet or object proof N] - P1 --> R1[Release response N] - P1 --> C2[Commit N+1] - C1 --> Q[Bounded publication queue] - C2 --> Q - Q --> U[One ordered object publisher] - U --> CAS[Exact-root control CAS] - CAS --> Q -``` - -The implementation is a bounded queue behind one actor and one object -publisher, not two SQLite writers and not parallel control CAS operations: - -1. Execute and capture one command on the existing SQL worker. -2. Submit its cuts to the node log and object path. -3. Release its result only after its own durability ticket is proven. -4. After proof, advance `logical_head` and allow the actor to execute the next - queued command. -5. Append the captured cut to an ordered publication queue. One publisher - advances one contiguous queued cut at a time onto `published_head`, performs - the canonical control CAS, advances that ticket's object coverage, and then - removes the cut. - -Queue admission is bounded by both entry count and retained LTX bytes and is -also charged to the existing local-disk budget. A Cell stops starting commands -at 64 pending cuts or once already-retained cuts reach the 64 MiB high water. -Because capture size is known only after SQLite commits, the one command that -crosses the byte high water is retained and published rather than discarded; -the configured capture limit plus the node local-disk budget form the hard -ceiling. Reaching either high water lets the publisher catch up; it does not -drop cuts, shrink the follower ensemble, or acknowledge through a weaker -proof. Schema migrations, graceful handoff, and shutdown remain publication -barriers and must drain the queue completely. - -The queue preserves these ordering rules: - -- Command `N+1` never executes until command `N` has a durability proof. -- Results are released in actor order; a later proof cannot pass an unresolved - earlier ticket. -- Root preparation consumes only a contiguous queue prefix whose predecessor - is the current `published_head`. -- Only the single publisher mutates Cell control. A lost CAS response reloads - and accepts only the exact proposed successor. -- A terminal lease or publication failure fences new execution. Already - released fleet-proven outcomes remain recoverable from the node log. -- A fenced executor may release Cell ownership only when every submitted - node-log cut is covered by the published root. Otherwise it discards local - SQLite but preserves the owner record, so session-expiry takeover seals and - replays the log instead of misclassifying the Cell as cleanly `Idle`. - -Object coverage may advance while an older frame is still waiting in the -node-wide shipper. Followers therefore verify every received frame and reject -conflicting local duplicates, but treat a locally absent prefix at or below -the authoritative `covered_through` watermark as a no-op. They must fsync every -later frame in the same batch. Without this rule, an object-first proof for -sequence `N` could make a queued `[N, N+1]` append look like a sequence gap and -silently disable fleet durability for the valid `N+1` suffix. - -This model is narrower than a general asynchronous publication graph. It adds -one ordered queue and two monotonic positions because they directly remove the -hot-Cell object-store stall. It does not add configurable queue policies, -parallel root writers, speculative branch heads, or compatibility paths. - -This is an intentional throughput-versus-complexity decision: - -| Design | Benefit | Cost or risk | Decision | -| --- | --- | --- | --- | -| One head; wait for every object CAS before the next command | Smallest lifecycle | One slow object round trip caps each hot Cell even after fleet durability succeeds | Keep only as the natural behavior when no fleet proof wins | -| Bounded `logical_head` plus `published_head` | Removes object latency between consecutive commands while preserving one writer and one ordered CAS owner | Retains proven cuts until publication and needs explicit drain/backpressure rules | Chosen and implemented | -| Multiple publishers, branch heads, or an unbounded publication queue | More speculative concurrency | Reordering, unbounded recovery state, and ambiguous CAS ownership | Rejected | -| Let a stream follow the moving logical head without per-chunk gates | Low-latency live output | Bytes could escape after lease loss or observe state newer than the stream's proof | Rejected | - -#### Revisit result: keep the bounded pipeline - -The dual-watermark pipeline remains the best Crab trade-off and is implemented, -so it is not a remaining delivery item. It pays for one extra monotonic -watermark and one bounded queue to remove object-store latency from consecutive -commands. It deliberately stops before a general publication graph: one actor, -one SQLite writer, one ordered publisher, and one Cell-control CAS owner remain. - -The shorter name **dual-head** refers only to these two publication watermarks; -it does not imply two independent root writers. A true dual-head publication -graph is deliberately deferred: it would add another CAS owner, reordering -state, and recovery surface without improving the one-writer contract. The next -durability work is qualification across signed node profiles and the extended -fault matrix, not a second publication head. - -State-observing streaming is delivered separately from publication. The first -Rust API reads mutable Cell state through `CellStateStream`; it adds no stream -scheduler, second writer, or publication head. Any future body adapter must -delegate to this gate rather than bypassing its receipt and lease checks. - -The dual-watermark model earns its extra state only because it changes current -command throughput. It is the narrowest design that gives Crab all three of -these properties: - -1. The next command does not wait for object-store latency after fleet fsync. -2. SQLite and Cell-control mutation still have one serial owner. -3. Recovery has one ordered interval, `(published_head, logical_head]`, rather - than speculative branches to reconcile. - -It is safe for failover because `logical_head` advances only after a -non-forgeable fleet or object proof, every unpublished cut remains in the -predecessor node log, and takeover seals and replays that log before opening -the successor SQLite database. `published_head` remains the compact, -long-term object-store authority; it is not weakened or replaced. - -```text -normal: published_head == logical_head -fleet win: published_head < logical_head # bounded recoverable interval -drained: published_head == logical_head -fenced: stop output; preserve the interval for takeover -``` - -This choice fits Crab because object stores have materially higher and more -variable latency than an in-fleet fsync, while the exact-root CAS must remain -serial. A general multi-publisher graph would add conflict resolution without -improving the one-writer SQLite execution path. - -The same owner-retention rule covers migration cuts. If fleet proof releases a -successor handle and object publication then fails, the actor fences both -admissions and leaves the old owner record for recovery. It never writes -`Idle` while the acknowledged migration exists only in the node log. - -The actor serializes bounded outputs against the logical-head proof. A command -waits for its own ticket, and a query or resolution starts only after the -preceding command has advanced the logical head. Therefore: - -- A mutation result cannot escape before its ledger row is durable -- A durable business rejection follows the same rule -- A query that observes a just-committed row waits for that row's proof -- An error generated after reading Cell state is gated - -The current Cell command and query APIs return bounded replies rather than -state-observing streams. Actor ordering proves that a query can observe only a -`logical_head` covered by an earlier durability ticket. - -### Deliver state-observing streaming as a separate output gate - -Streaming is an output gate, not another publication head and not an extension -of the object-publication queue. The important distinction is what -the producer can observe: - -| Stream kind | Required gate | -| --- | --- | -| Immutable blob or object already authorized by digest/root | Pin that immutable identity before the response head; later byte reads cannot reveal newer Cell state | -| Materialized result fixed at stream open | Prove the captured logical watermark before the response head and retain the materialization until close | -| Producer that can read Cell state between chunks | Take a fresh output ticket for the response head and every chunk | - -Celld uses the third rule: one response release is insufficient because the -producer continues after the head and a later chunk can reveal a later commit. -`CellStateStream` applies that same rule to Crab's Rust state-observing API. - -```mermaid -sequenceDiagram - participant P as Rust stream producer - participant G as Cell output gate - participant D as Durability proof - participant H as HTTP body - - P->>G: emit(stream, observed_sequence, chunk) - G->>G: verify owner epoch and node lease - alt observed_sequence is already proven - G-->>H: release chunk - else proof is pending - G->>D: await fleet or object proof - D-->>G: proven through observed_sequence - G-->>H: release chunk - else fenced, expired, or unprovable - G-->>H: terminate body without releasing chunk - end -``` - -The first implementation must satisfy this contract: - -1. `open_state_stream` allocates a stream ID and binds it to the current Cell, - incarnation, expected owner description, deadline, and latest observed - commit sequence. The owner lease/session is rechecked by the local actor or - authenticated peer on every query. -2. `EmitChunk` carries the highest commit sequence the chunk may reveal. The - gate releases it only when the same epoch has a fleet or object proof - covering that sequence. -3. The response head and each chunk use the same gate. Exactly one chunk per - stream may wait or flush, so held data cannot be overtaken. -4. The terminal node-session lease is checked immediately before every flush. - Lease loss, ownership change, cancellation, deadline, or an unprovable - watermark closes the stream and releases all admission permits. -5. Buffered chunk bytes, stream count, and any pinned materialization are - bounded by runtime admission. The producer cannot build an unbounded queue - behind a slow client or slow proof. -6. A stream never reads a moving logical head implicitly. A producer that - performs another state observation must obtain a new observed sequence and - a new chunk ticket. - -```rust,ignore -let mut stream = client.open_state_stream::(&target, deadline).await?; -let first = stream.emit(first_input).await?; -send_chunk(first.output).await?; -let next = stream.emit(next_input).await?; // waits for the next proof -send_chunk(next.output).await?; -stream.finish(); -``` - -The stream-gate proof matrix is small and specific: - -- A response head and first chunk wait for the commit they reveal. -- A mutation between two chunks makes only the later chunk wait for the newer - watermark. -- Lease expiry or takeover between chunks releases no further bytes. -- A later proven chunk cannot overtake an earlier held chunk. -- Client cancellation, deadline, proof failure, and producer failure return - every buffer, snapshot, and admission permit. - -The streaming gate is now the first Rust state-observing stream API. `CellClient` -opens a typed `CellStateStream`; its mutable `emit` operation serializes chunks, -passes the previous `Receipt` as the next minimum watermark, and closes on -deadline, cancellation, fencing, or a non-monotonic receipt. The existing actor -query path performs the final node-lease check immediately before returning the -observed value. The stream uses the query's declared output limit as its one- -chunk byte bound and never creates a second publication queue or writer. - -The current server stream audit explains that boundary: - -| Current or future output | State source after response head | Decision | -| --- | --- | --- | -| Release asset and LFS download | Immutable object selected by digest and size | Pin identity before the head; no per-chunk Cell gate | -| Repository archive and Git pack | One fixed repository snapshot or fetch plan | Keep snapshot/operation lifetime through the body | -| SQL, KV, Queue, and Workflow response | None; runtime returns one bounded value | Existing actor proof gates the complete value | -| Future SSE, live query, or incremental Cell renderer | May read a newer Cell head for each chunk | Use `CellStateStream`; a custom body must preserve the same per-output gate | - -This narrow API avoids a second speculative queue or stream scheduler. It does -not weaken the contract: introducing a state-observing body without the phase 8 -gate is a correctness regression, not an optional optimization. - -The server now exposes `crab_http_server::state_observing_body` as the narrow -HTTP adapter. It consumes one input only after the previous body chunk has -completed, invokes `CellStateStream::emit` before encoding each chunk, maps -stream errors to body I/O errors, cancels on body drop, and owns no queue or -scheduler of its own. Product routes still choose their media type (SSE or a -custom chunk format) and must set the corresponding response headers. - -Authentication, routing, and malformed-request errors produced before Cell -execution do not need a Cell durability proof. - -## Keep object-store proof as the fallback - -Fleet mode does not make the object store optional. It remains the authority, -long-term store, compaction target, and fallback durability path. - -| Fleet condition | Response behavior | -| --- | --- | -| All selected followers fsync first | Fleet proof releases response | -| Exact root CAS completes first | Object proof releases response | -| No eligible follower | Wait for object proof | -| One member NACKs or times out | Degrade current batch to object proof | -| Object store is slow but lease remains valid | Fleet proof may continue temporarily | -| Object store outage reaches lease expiry | Terminal self-fence; no further responses | - -The default mode is `fleet`, with transparent object-proof fallback. An -explicit `object` mode disables follower recruitment and preserves today's -response path. A one-node fleet stays correct through object proofs; it cannot -claim follower redundancy. - -## Recover a failed owner before Cell restore - -The first request for any Cell owned by an expired session starts or joins one -node-session recovery. Recovery is single-flight per dead session across the -process and elected across the fleet by session-record CAS. - -```mermaid -sequenceDiagram - participant N as Successor node - participant S as Dead session record - participant F1 as Follower A - participant F2 as Follower B - participant O as Object store - participant C as Cell control - - N->>S: Read expired live/open record - N->>S: CAS Open -> Recovering(claimant, expiry) - par Seal members - N->>F1: SealFragment(session, log_epoch) - N->>F2: SealFragment(session, log_epoch) - end - F1-->>N: base/end + tail - F2-->>N: base/end + tail - N->>N: Verify and merge by node sequence - N->>O: Upload recovery bundles/manifests - loop Every affected Cell - N->>C: CAS attach exact RecoveryOverlayRef - end - N->>S: CAS Recovering -> Sealed(manifest set) - N->>C: CAS Cell takeover to new epoch - N->>O: Build exact successor root from overlay - N->>C: CAS root and clear overlay - N->>N: Sparse-open exact root - N->>C: CAS Recovering -> Serving -``` - -### Elect recovery - -The recoverer rereads the session record and current time immediately before -claiming. It refuses a live lease. `Open -> Recovering` records claimant -session, claim generation, and claim expiry. The claimant refreshes that field -every ten seconds. Each refresh has a five-second deadline; a stalled object -store therefore fails recovery closed instead of allowing a claimant to keep -gathering after its 30-second claim expires. Before sealing, recovery renews -the claim again and gives the final `Recovering -> Sealed` CAS its own -five-second deadline; a timeout leaves the claim resumable by the scheduler. - -A second node waits behind a live claim. After the 30-second claim expiry, it -may CAS takeover of recovery. All later operations are content-addressed, -idempotent, or Cell-control CASes, so repeated work converges. - -The request path may take over an expired session only when its node log is -already inactive. An active log returns `PendingPublication` without creating -a claim; the follower scheduler remains the sole path that can reserve and -recover that log, so a cold request cannot strand the preferred follower behind -an arbitrary 30-second claim. - -### Seal and gather followers - -Each follower serializes `SealFragment` with append handling. An append wholly -before the seal is included; an append after the seal is rejected. A fleet proof -needs every member's acknowledgement, so any fleet-acknowledged frame exists on -each complete member. - -The recoverer accepts a member as a complete witness only when: - -- It reports the expected leader session and log epoch -- Its local fragment is sealed -- Its returned tail covers its declared retained `[base, end]` range -- Every record and LTX frame verifies - -One complete witness is sufficient under write-all, ack-all. Additional -witnesses are compared by sequence and digest. Equal sequences with unequal -bytes fail recovery as corruption. The union may contain an unacknowledged -suffix; exposing such a suffix is allowed because a missing response never -promises that a transaction was rolled back. - -If `log.active` is true and no complete witness is available, recovery remains -unavailable. It does not seal the record or declare the tail empty. An operator -can restore a missing follower disk and retry. Any future data-loss override -must be a separate audited administrative operation that writes a permanent -loss record; it is not an automatic runtime branch. - -### Build recovery manifests - -The recoverer filters entries at or below the session's object-covered -watermark, then groups the remaining verified frames by application, Cell, -incarnation, and Cell epoch. Because different Cells publish independently, -the shared watermark may lag a Cell's exact root. Within each group, recovery -also removes frames already covered by that root's commit sequence and LTX -position, checks the checksum at an equal TXID, and requires the uncovered -suffix to begin at the next commit. Fully rooted groups need no overlay. -The recovery reservation stays owned through immutable pinning. -It writes shared immutable bundles and one small -manifest per affected Cell. - -```json -{ - "version": 1, - "leader_session": "16-byte-hex", - "log_epoch": 3, - "application": "16-byte-hex", - "cell": "32-byte-hex", - "incarnation": "16-byte-hex", - "cell_epoch": 12, - "predecessor": { - "root": "32-byte-hex", - "txid": 500, - "checksum": 9223372036854776000, - "commit_sequence": 700 - }, - "entries": [ - { - "node_sequence": 9002, - "bundle": "32-byte-hex", - "offset": 4096, - "length": 8192, - "ltx": { - "min_txid": 501, - "max_txid": 501, - "post_checksum": 9223372036854777000, - "commit_sequence": 701, - "blake3": "32-byte-hex" - } - } - ] -} -``` - -The final implementation uses canonical strict JSON or the existing canonical -binary manifest codec; it must not use floating-point numbers or permissive -unknown fields. The manifest digest covers its canonical bytes. Every bundle -extent is range-readable and independently BLAKE3-bound. - -### Pin every affected Cell - -Before sealing the node session, recovery attaches each manifest digest to the -matching dead-owner Cell control. That CAS makes recovered objects reachable to -backup and retention. A lost response is reconciled by exact overlay equality. - -If control already names a newer root that covers the manifest's final -position, recovery records the Cell as covered. Any other owner, incarnation, -epoch, or predecessor mismatch stops recovery. It is unsafe to seal while one -acknowledged tail is neither rooted nor attached. - -### Consume the overlay - -After acquiring the Cell, the new owner loads the pinned overlay, prepares its -exact successor, and publishes that successor through the control CAS: - -```rust,ignore -let observed = /* latest VersionedControl loaded from authority */; -let control = observed.value(); -let recovery_ref = control - .recovery - .as_ref() - .ok_or(Error::Control("recovery overlay is not pinned"))?; -let overlay = manifests - .load_overlay(control.cell, control.incarnation, recovery_ref) - .await?; -let prepared = replica - .prepare_recovered_overlay(&overlay, control.schema) - .await?; -let successor = control.publish_recovery(&prepared, control.next_due_ms)?; -authority - .transition(&observed, successor, Transition::PublishRecovery) - .await?; -``` - -`prepare_recovered_overlay` performs no ownership decision. It verifies the -manifest and bundle extents, checks the predecessor, folds every LTX segment in -order, builds the authenticated directory, and returns a `PreparedRoot`. The -runtime verifies the returned final position and commit sequence against the -overlay before CAS. - -Only then does the successor sparse-open the exact new root. First-page faults -use current authenticated range reads; background hydration remains bounded -owner maintenance. - -### Retire recovery data only after every Cell covers it - -Retention treats all of these as roots: - -- Current Cell roots -- Attached `RecoveryOverlayRef` manifests and their bundle extents -- Backup pins captured while an overlay is attached -- Recovery manifests named by an `open`, `recovering`, or `sealed` session - -A sealed session keeps a compact affected-Cell index. A maintenance pass may -CAS it to `retired` only after every listed Cell is tombstoned or its current -root covers the overlay's final transaction, checksum, and commit sequence, and -no control or backup pin still names the overlay. The small retired session -record remains as an audit tombstone; recovery bundles become ordinary -grace-period collection candidates. - -Followers may discard their sealed local fragments after the sealed record and -all referenced recovery objects can be reread and verified from object storage. -They do not wait for every Cell to activate because the control-pinned overlays -already preserve reachability. - -## Handle absence and sealed records precisely - -These observations have different meanings: - -| Observation | Meaning | Action | -| --- | --- | --- | -| Expired session exists, `log=null` | Session never released a fleet proof | Current Cell root is complete; takeover may proceed | -| Session log is `sealed` | Recovery overlays were durably pinned | Consume matching overlay, if any | -| Session log is `open` | Fleet-only tail may exist | Recover before Cell takeover | -| Session log is `recovering` | One claimant is gathering/pinning | Join or take over an expired claim | -| Session log is `retired` | Every affected Cell root covers the tail | No recovery data remains live through this session | -| Entire owner session record missing | Authority evidence is missing | Fail closed; do not infer bucket completeness | -| Cell is `Idle` with no owner | Clean release already forced object coverage | Acquire normally | - -The session record is strict-created before a node can own a Cell. Its absence -is therefore corruption or operator deletion, not proof that no follower -acknowledgement occurred. - -## Keep graceful movement cheaper than crash recovery - -A clean per-Cell handoff does not seal the whole owner node log. - -1. Stop new commands for the Cell -2. Wait for accepted commands to reach either proof -3. Force all Cell cuts through exact object publication -4. Verify the exact published root covers the final local commit -5. Close SQLite and CAS Cell control to `Idle` -6. Let the successor acquire a new Cell epoch and sparse-open that root - -Because `Idle` proves complete object coverage, no node-log recovery is needed. -An optional successor hint may prefetch the authenticated root directory and -hot pages, but it grants no authority. - -Clean node shutdown is broader: - -1. Withdraw public admission and mark the session draining -2. Drain accepted Cell work -3. Stop issuing fleet proofs -4. Advance every outstanding node-log sequence to object coverage -5. Seal the node log with no recovery overlays -6. Release owned Cells after each SQL worker closes -7. Withdraw the session lease and close follower services - -A crash at any step leaves `open` or `recovering` state and returns to the same -dead-session recovery path. - -The runtime's shutdown path calls `close_node_log` for steps 3 through 5. It -stops ticket issuance only after every issued sequence is object-covered, -best-effort writes exact retire fences to reachable followers, and CAS-clears -the session log. That clear makes later appends unauthorized and allows exact -session withdrawal. - -## Start and stop in recovery-safe order - -Startup order is: - -1. Load or strict-create the durable physical node ID, then validate local - follower directories and quarantine corrupt lanes -2. Start the private mTLS listener in **recovery-only** mode -3. Serve `SealFragment` and `ReadTail` for surviving peer fragments -4. Probe object-store conditional writes and range reads -5. Strict-create a fresh session record -6. Start the lease watchdog and renewal task -7. Start node-log recovery sweeps, shipper, actors, and schedulers -8. Advertise application readiness - -Serving follower recovery before application readiness allows a fleet-wide -restart to recover from surviving disks without circularly waiting for every -node to become fully ready. - -The server now starts the mTLS management listener before the object-store -probe and session publication. Middleware returns `503` from append, retire, -and ordinary peer-forward routes until the session lease, watchdog, recovery -sweep, shipper, actors, and schedulers are installed. Seal and tail remain -available to an authenticated live claimant: the dead leader's recovery claim -authorizes that claimant against this node's persisted physical `NodeId`, so a -fresh local boot-session advertisement is not a recovery prerequisite. - -Shutdown closes public and application-peer admission first. It then drains -accepted work, closes the Cell runtime and node-log epoch, withdraws the node -heartbeat, and only then cancels the recovery listener. A follower must not -discard the only surviving fragment merely because its own application runtime -is draining. - -## Bound resources and backpressure - -Follower capacity is accounted separately from Cell residency. It covers -fragment bytes, open leader lanes, recovery readers, and sync work. - -| Resource | Required bound | -| --- | --- | -| Selected followers per leader | 1 or 2 | -| Append streams per remote leader | 1 | -| Global follower append bytes | Derived from local SSD budget | -| Per-lane outstanding bytes | 8 batches and 64 MiB maximum | -| Concurrent node-log recoveries | CPU/job-credit bounded, maximum 2 initially | -| Concurrent tail reads | Shared object/network I/O semaphore | -| Recovery scratch | Reserved before reading tail bodies | -| Uncovered owner bytes | Hard bound; reaching it forces object proof/backpressure | - -At 1,000 aggregate transactions/s, the design must group fsync work. The -leader batches frames across Cells for up to the smaller of one millisecond, 64 -frames, or 64 MiB. Each follower appends that batch and performs one `sync_data`. -The exact interval is a measured runtime constant, not a per-deployment tuning -surface until qualification proves one is needed. - -The write-amplification envelope is: - -```text -owner local WAL/LTX -+ N follower fragment writes, where N is 1 or 2 -+ one eventual object-store upload -+ bounded compaction and recovery rewrites -``` - -Follower bytes are temporary. Object coverage advances truncation so retained -space tracks upload delay, not total database size. When follower space reaches -its reserve, it NACKs new batches before writing and the owner uses object -proofs. It never evicts uncovered fragments. - -## Define failure behavior before implementation - -| Fault | Required result | -| --- | --- | -| Owner dies before either proof | No success was promised; recovery may include or omit the cut | -| Owner dies after all-follower fsync | Recovery attaches and consumes the cut before serving | -| Owner dies after object root CAS | Successor restores the exact root; duplicate follower tail is ignored by position/digest | -| One follower fails mid-batch | No fleet proof; object path may win | -| Follower fsync succeeds but ACK is lost | The request may remain unresolved; recovery may expose the cut and request replay resolves it | -| Follower returns a gap | Leader retransmits only exact retained frames or degrades to object proof | -| Follower returns conflicting duplicate | Quarantine lane and fail fleet proof | -| Recoverer dies after uploading bundles | Next claimant reuses content-addressed objects and resumes Cell CASes | -| Recoverer dies after some overlays attach | Next claimant reconciles exact overlays; session remains unsealed | -| Recovery claim stalls | Another live session takes over after claim expiry | -| Owner is partitioned from object store | Lease expires and process self-fences even if followers are reachable | -| Owner is partitioned from followers | Object proofs preserve correctness | -| Stale owner resumes after takeover | Session is non-live, followers are sealed, and Cell CAS rejects it | -| All selected follower disks are unavailable and no object proof exists | Cell remains unavailable; no automatic loss declaration | -| Recovery bundle or LTX is corrupt | Fail closed with the exact digest/sequence category, never partial restore | -| Local successor disk fills | Release reservation, keep control `Recovering`, and retry elsewhere | -| Object CAS result is lost | Reload and accept only the exact proposed session/control successor | - -## Expose narrow Rust APIs - -The names below mirror the current public ownership and data-flow boundaries; -private fields and imports are omitted. Changes must not collapse these layers. - -### `crab-ltx` - -```rust,ignore -pub struct VerifiedNodeFrame { /* private validated fields */ } - -pub fn inspect_node_frame(bytes: Bytes, limits: Limits) - -> Result; - -impl CellReplica { - pub async fn prepare_recovered_overlay( - &self, - overlay: &RecoveryOverlay, - schema: u32, - ) -> Result; -} -``` - -The verified frame exposes immutable metadata and bounded, already-verified body -bytes. It does not expose unchecked constructors for server code. - -### `crab-cell-runtime` - -```rust,ignore -pub trait NodeLogTransport: Send + Sync { - fn append(&self, member: NodeId, request: AppendRequest) - -> BoxFuture<'_, Result>; - fn seal(&self, member: NodeId, request: SealRequest) - -> BoxFuture<'_, Result>; - fn retire(&self, member: NodeId, request: RetireRequest) - -> BoxFuture<'_, Result>; - fn tail(&self, member: NodeId, request: TailRequest) - -> BoxFuture<'_, Result>>; - fn tail_page(&self, member: NodeId, request: TailRequest) - -> BoxFuture<'_, Result>; -} - -pub struct DurabilityGate; -pub struct NodeLogShipper; -pub struct NodeLogSubmission; - -impl DurabilityGate { - pub async fn prove(&self, ticket: CommitTicket) - -> Result; -} - -impl NodeLogShipper { - pub async fn submit(&self, submission: NodeLogSubmission) - -> Result; - pub async fn shutdown(&self) -> Result<()>; -} - -pub struct NodeLogRecovery; - -impl NodeLogRecovery { - pub async fn ensure_sealed(&self) -> Result; -} - -pub struct NodeTakeoverProof { /* private validated fields */ } -``` - -`CommitTicket` is created only by the actor after capture. `DurabilityProof` -has private fields or validated constructors so application handlers cannot -forge a release token. `FencedNodeSession` converts directly to -`NodeTakeoverProof` only when fleet durability was never active. Otherwise the -proof is emitted only after every recovered overlay is pinned and the session -log is CASed to `sealed`. - -### `crab-http-server` - -The server installs one transport and follower store into `CellRuntimeBuilder`. -Private route handlers authenticate and decode, then call those objects. They do -not access Cell actors or execute application commands. - -## Report actionable status without Cell-label metrics - -Prometheus metrics must remain bounded in cardinality: - -```text -crab_cell_durability_proofs_total{source="fleet|object"} -crab_cell_durability_submissions_total{outcome="fleet|unsupported|unavailable|rejected"} -crab_cell_durability_wait_seconds{source="fleet|object"} -crab_cell_command_responses_total{source="recorded|fleet|object"} -crab_cell_command_response_seconds{source="recorded|fleet|object"} -crab_cell_command_confirmation_seconds{source="recorded|fleet|object"} -crab_cell_node_log_append_bytes_total{result="acked|nacked"} -crab_cell_node_log_uncovered_bytes -crab_cell_node_log_lanes{state="open|degraded|sealed"} -crab_cell_node_log_recoveries{state="running|waiting"} -crab_cell_node_log_recovery_seconds -crab_cell_node_log_recovery_phase_seconds{phase="claim|witness|scope_validation|pin_attach|seal"} -crab_cell_node_log_recovery_failures_total{reason} -crab_cell_node_log_recovery_work_total{kind="candidate_count|affected_cells|catalog_shards|catalog_pages|control_reads|follower_pages|follower_frames|follower_bytes|peer_requests|bundle_bytes|object_reads|object_writes"} -crab_cell_node_log_rotations_total{result="started|pending|failed|completed"} -crab_cell_follower_retained_bytes -crab_cell_session_lease_seconds -crab_cell_self_fences_total{reason} -``` - -Cell ID, repository name, request ID, session ID, and object digest belong in -structured logs or bounded administrative queries, never metric labels. - -Command responses count once at the final runtime reply boundary. `fleet` or -`object` records the proof that released a new commit; `recorded` means a -previously durable result was replayed. Later object publication can increment -the proof counters without adding another response. Runtime errors and dropped -receivers add no response; a durable application rejection is still a returned -outcome. The same boundary covers effect delivery. Queries and migrations use -separate paths and are excluded. - -Response duration starts at admitted enqueue and includes actor queueing, -execution, capture, proof, and final worker confirmation. The confirmation -histogram isolates that last worker wait and is zero for recorded results. -These metrics exclude HTTP/peer transport and cannot alone establish public -action latency or sustained publisher drain. - -The current server wiring emits durability-proof and follower-append events -through `CellTelemetry`; it samples the signed node-log phase and session-lease -remaining time, and records recovery duration and bounded failure class from the -scheduler. Recovery `waiting` counts candidates in the bounded retry delay; it -does not include sessions that have not yet been observed by this scheduler and -must not be inferred from a saturated worker count. - -Every captured commit also reports how its node-log submission resolved. -`fleet` means an enrolled lane accepted the commit for shipping, while -`unsupported` (this host installs no provider), `unavailable` (a provider exists -without an enrolled lane), and `rejected` (the enrolled lane fenced or refused -the submission) all describe commits that still succeed through object coverage. -Those three outcomes are the only signal that a node intended fleet durability -and silently fell back, so alert on them instead of inferring durability from -commit success. - -`cells status --owner OWNER --name REPOSITORY --json` reports from persistent -control and signed node-session state: - -- Current owner session and Cell epoch -- Owner-session lease state and expiry -- Pending recovery overlay, if any - -The published commit sequence is the control root. Logical commit sequence and -last durability source are live actor state; they require a future -owner-introspection channel and must not be guessed by an out-of-process CLI or -derived from the node-log sequence. - -`cells node --session SESSION --json` reports the signed advertisement's log -state, epoch, stable member node IDs, activation bit, tiered sequence, follower -free/retained byte estimates, recovery claimant, and recovery-manifest digest. -It does not return frame bodies or credentials. Expired sessions return -`live=false` without treating stale advertisement contents as current state. - -## Qualify the complete contract - -Unit tests alone cannot establish the guarantee. Delivery requires all layers -below. - -### Deterministic protocol tests - -- Model session `Live -> Recovering -> Sealed` CAS transitions -- Explore append/seal ordering at every message boundary -- Prove a fleet proof implies every selected member stored the covered range -- Prove a sealed active log has one complete witness and every affected Cell is - rooted or carries an exact overlay -- Prove no Cell reaches `Serving` with `recovery != None` -- Prove member reconfiguration waits for object coverage -- Prove lease expiry is terminal for the old process - -### Filesystem tests - -- Kill between append, `sync_data`, ACK, rotation rename, and directory sync -- Tear every byte of the active record trailer and recover only the valid prefix -- Replay exact duplicates and reject conflicting duplicates -- Fill the follower disk before append and during recovery materialization -- Restart a follower before the application runtime is ready and serve tail -- Verify truncation never removes a sequence above object coverage - -### Runtime integration tests - -- Return a fleet proof while blocking every object upload, kill the owner, delete - its disk, take over, and resolve the exact request outcome -- Race object proof and fleet proof in both orders -- Advance object coverage while its frame remains queued, then retain and - recover the uncovered suffix -- Gate reads and business-error outputs behind an earlier unproven commit -- Kill recovery after each overlay attachment and resume from another node -- Recover a multi-cut transaction and interleaved cuts from 1,000 Cells -- Include a valid but unacknowledged suffix and preserve request idempotency -- Corrupt bundle bytes, index bytes, and manifest metadata independently -- Exercise cross-epoch continuation, truncate/regrow, sparse faults, hydration, - compaction, backup pins, and retention after recovery - -### Three-node live qualification - -Run three independent processes with separate SSD directories and real RustFS: - -1. Force every response through two follower fsyncs while object upload is - delayed -2. Sustain the declared aggregate 1,000 TPS workload and record fsync grouping, - network bytes, object lag, p50/p95/p99 latency, RSS, and retained bytes -3. `SIGKILL` the owner and delete its local Cell and log directories -4. Keep one follower, restart the other, and require exact recovery -5. Verify higher Cell epoch, identical request outcomes, and monotonic root -6. Partition the old owner, let its session expire, recover elsewhere, then - reconnect it and prove terminal fencing -7. Repeat with follower disk full, slow follower, lost ACK, object-store 429, - recovery crash, and simultaneous fleet restart - -Capacity claims must cover small, medium, and large node profiles separately. -The 1,000 TPS target is aggregate per node, not per repository. A result must -state transaction size, changed pages, follower count, database distribution, -object-store latency, and failure injection. - -The qualification profile is an envelope, not a required machine shape. The -same binary and protocol run on every profile; only admission limits and the -load schedule change: - -| Profile | CPU | Memory | Local SSD | Active Cells | Throughput target | -| --- | --- | --- | --- | ---: | ---: | -| Small | 1–2 vCPU | 2–4 GiB | 50–100 GiB | 1,000–10,000 | 1,000 mutations/s per node | -| Medium | 4–8 vCPU | 8–16 GiB | 100–200 GiB | 1,000–10,000 | 1,000 mutations/s per node | -| Large | 16 vCPU | 32–64 GiB | 500–1,000 GiB | 1,000–10,000 | 1,000 mutations/s per node | - -The active-Cell range is the node admission envelope, not a promise that every -workload reaches the upper bound. A report must include the actual count, -retained follower bytes, queue depth, and p50/p95/p99 latency. The 1,000 -mutations/s figure is always node-aggregate; a hot repository is qualified -separately with the one-Cell schedule described below. - -Use `qualify_http_load --aggregate-requests-per-second 1000` for the bounded -request schedule. Its schema-v2 receipt rejects a run whose successful response -count is below 95% of the configured aggregate rate; 429 responses remain -visible admission evidence but do not count toward the target throughput. -The Kubernetes qualifier runs the schedule separately through each Pod, using -64 bounded status-mutation targets backed by distinct commits per run and -distributed across eight repository Cells (24 commits per Cell). Each Pod gets -eight targets per Cell, so every report exercises the complete Cell set. It -binds the three reports to Pod UIDs, captures capacity again after load, and -rejects server, transport, body-limit, latency-over-60-second, or target-rate -failures. The eight-Cell schedule is the node-level aggregate profile (the -receipt records a configured 125 target requests/s per Cell); retain a separate -one-Cell run when measuring the hot-Cell admission limit. -The version-6 typed cluster-receipt validator maps every failed and successor -session to stable NodeIds, requires the first two successors to be present in -their failed log's original follower set, and requires the third successor to -be a live non-member after every original follower is unavailable. It requires -a successful observation for each fixed recovery phase and binds bounded work -counters to all three loss cycles. Prometheus labels remain fixed; the receipt -keeps the raw metric text only as evidence and rejects identifier-bearing -labels. - -## Deliver in dependency order - -Each phase has a usable exit criterion. Do not enable the fleet response path -until the recovery gate is complete. - -| Phase | Implementation | Exit proof | -| --- | --- | --- | -| 1 | Session lease, watchdog, terminal self-fence, Cell takeover based on session state | Pause/partition owner; no response or renewal after expiry | -| 2 | `crab-ltx` verified frame and recovered-overlay API | Golden, corruption, cross-epoch, and exact-root tests | -| 3 | Follower disk format and private append/seal/tail protocol | Crash matrix proves fsync and torn-tail behavior | -| 4 | Node-log recovery claim, bundles, Cell overlay attachment, retention roots | Kill recovery at every boundary and converge | -| 5 | Dual object/fleet durability gate with one in-flight publication per Cell | Fleet-first response survives owner and disk loss | -| 6 | Ensemble rotation, graceful drain, startup recovery-only listener, GC | Member loss and rolling restart matrix | -| 7 | Bounded logical/published-head pipeline for hot Cells | Consecutive commands no longer wait for object publication; queue bounds and crash recovery hold | -| 8 | `CellClient::open_state_stream` and `CellStateStream` per-output watermark gate | Typed stream tests prove monotonic receipts, serial emission, deadline, cancellation, and fencing behavior | -| 9 | Real RustFS and Kubernetes qualification at target load | Signed receipts with zero lost acknowledged outcomes | - -Phases 1 through 4 may ship with object-only responses. Phase 5 is the first -point at which follower fsync may release a public result. - -## Use Celld as a behavioral reference, not an inherited proof - -The pinned Celld design establishes the pattern used here: - -- [Celld guarantees](https://github.com/denoland/celld/blob/10cb1303dac710dcb3b557e318e08c855261f68b/docs/guarantees.md) - defines one owner, the dual durability proof, node-log recovery, epoch-chain - restore, and terminal self-fencing. -- [Celld node log](https://github.com/denoland/celld/blob/10cb1303dac710dcb3b557e318e08c855261f68b/crates/celld/node_log.rs) - implements follower append, write-all/ack-all, sealing, gathering, and - recovery claims. -- [Celld LTX replication](https://github.com/denoland/celld/blob/10cb1303dac710dcb3b557e318e08c855261f68b/crates/celld/ltx_repl.rs) - races bucket and fleet proofs and multiplexes Cell cuts. -- [Celld output gate](https://github.com/denoland/celld/blob/10cb1303dac710dcb3b557e318e08c855261f68b/crates/logic/output_gate.rs) - gates the response head and each later state-observing stream chunk. - -Crab must prove its own version because its control model differs. Celld can -restore discoverable epoch prefixes. Crab restores one authenticated root, so -it additionally needs the control-pinned recovery overlay described above. -That difference is intentional: it preserves Crab's verified manifests, -checksums, exact-root backup, and existing storage dependencies while matching -Celld's follower durability and takeover behavior. diff --git a/crates/crab-cell-runtime/docs/ltx-performance-audit.md b/crates/crab-cell-runtime/docs/ltx-performance-audit.md deleted file mode 100644 index a4785f406..000000000 --- a/crates/crab-cell-runtime/docs/ltx-performance-audit.md +++ /dev/null @@ -1,3501 +0,0 @@ -# Audit LTX latency and sustained publication capacity - -| Document intent | Value | -| --- | --- | -| Content type | Design audit and acceptance gates | -| Audience | LTX, runtime, storage, and qualification contributors | -| Scope | Initial baseline `0f3f4f7617a`; each follow-up diagnostic identifies its source below. Historical comparisons use `origin/main` snapshot `de0bb234abc`; the follow-up integrates read replicas from `396e0ab1b40`. The latest completed 3/5/10/20-node rate sweep uses `e9e238b17b3` and pinned RustFS 1.0 GA. It predates shared scheduler discovery, clean-resume preservation and the later peer/fixture fixes through `a62df00a644`. | -| Status | Hydration fetch, sparse registration, persistent-cache construction isolation, conditional issue enrichment, recovery receipt preservation, range-proportional compaction, bounded asynchronous cache fills and local checksum read/merge improvements are implemented. Loaded scale-out cannot assume idle ownership transfer. Demand faults, installation latency, recovery storms, sustained publication and fleet performance remain open. | - -[Scaling plan](vfs-ltx-scale-plan.md) · [Recorded measurements](../../crab-ltx/perf/README.md) - -The highest-value next experiments remove provider round trips and cache -bookkeeping from request paths. Keep the existing SQLite VFS, exact-root -verification, fencing, stable command receipts, and follower recovery model. -The recorded sub-millisecond sparse capture and roughly 87–139 ms small-root -RustFS preparation come from different harnesses and revisions. They identify -where to investigate; they cannot be subtracted to explain a public action. -Baseline descriptions retain the original finding; implementation paragraphs -identify changes already made. Each measurement record names its source and -harness separately. - -## Current-head audit: remaining work and evidence - -The design retains the right authority boundary: one fenced -SQLite writer, authenticated immutable roots, ordered publication and explicit -durability proofs. Current evidence does not establish a supported high-throughput, -low-latency profile. An earlier fleet run exposed concentrated ownership; -the corrected placement now distributes execution across every node in the -fixed-load fleet. Latency still increases at twenty nodes on the shared host. -Checksum allocation, synchronous demand I/O and ordered publication retain -their separate performance gates. - -| Priority | Gap | Next decision and proof | -| --- | --- | --- | -| Corrected; GA published-root recovery passes | An inactive-log node claim blocked other successors restoring that node's Cells (30) | Permanent fencing evidence now permits independent Cell CAS; active-tail recovery stays exclusive. The two-Cell RustFS regression and all sixteen later GA published-root recovery points pass. The separate unpublished-tail fault remains unqualified. | -| High, observed setup race | A newly provisioned repository reached an entry node before its peer's catalog refresh; an earlier fleet stopped with HTTP 403 before timing (32) | The later GA sweep provisions fixtures before serving nodes. Dynamic creation still needs a version-aware readiness contract; do not convert authorization errors into retries. | -| Corrected local publication race | A delayed import could overwrite a newer catalog snapshot and restore revoked membership (34) | Materialized indexes now retain the durable revision and publish monotonically. Import and polling share Cell-readiness validation. HTTP revocation regresses before the fix and remains denied after it, including on RustFS GA. Cross-node creation readiness remains open. | -| High, measured application round trip | Covered repository mutations pay an archive-state query before their command; saturated ten-node p99 is 1,448.518 ms (31) | Evaluate checking writable state in the same command transaction. Cover archive/unarchive ordering, retries, all mutation siblings and external-write policy before removing the HTTP check. | -| Implemented and fixed-load verified | Rounding the receiver margin up prevented donation at a one-Cell target; a batch could also overfill its preferred receiver (26) | Whole-Cell margins and projected receiver room pass both regressions. The latest 3/5/10/20-node run uses every execution owner; skew, sustained load and the combined source remain unqualified. | -| High, implemented mechanism; latency unqualified | The baseline empty checksum overlay retained its largest allocation and cloned that capacity (25) | Sealed merges now consume the overlay. Compare large-cut → repeated one-page-cut allocation and latency for both bases, retaining failure fencing and recovery-plan clone semantics. | -| High, reproduced latency interference | Demand faults block a resident sibling on the same SQL worker; installation, confirmation and cleanup also use that worker (9–10, 19) | Local RustFS release diagnostics show 43–45 ms median sibling delay without injected latency and 414–416 ms with 20 ms per GET. Qualify public actions on one vCPU before selecting bounded executor scheduling or prefetch. An active SQLite callback cannot yield its connection. | -| High, recovery scaling | Writable activation still walks all authenticated checksum leaves (3, 14) | Measure first query and first mutation during concurrent recovery. Prototype demand-loaded existing directory leaves only with an aggregate/truncation integrity design and bounded old-checksum availability for capture. | -| High, measured publication delay | Ten-node publication queue/completion p99 reaches 899/1,460 ms; compaction shares the ordered publisher (4–5, 17) | Split preparation, provider attempts, worker binding and confirmation before selecting an optimization. Qualify sustained logical commit progress and bounded uncovered age/bytes; evaluate consecutive-root coalescing only with receipt and effect-order proof. | -| Medium, resource interference | Optional cache fills share blocking jobs with required work; read-ahead may fetch pages never used (6, 8, 18, 21) | Pause cache syncs while another Cell activates or publishes. Measure useful/fetched bytes and unused prefetch eviction before adding priority or changing cache policy. | -| High, replica read path | Warm snapshots still require routing and response authority reads; refresh uses a new demand-cache view identity (27) | Count provider operations per successful replica read and bytes fetched after refresh. Preserve fencing while evaluating coalesced metadata observations and verified immutable-frame reuse. | -| High, implemented and range-verified; service latency open | A fragmented root's demand read-ahead fetched 55 pages already cached in that same view (28) | Demand misses now stop before a cached suffix. Exact-range regressions pass at 512/4096-byte pages in memory and RustFS; cross-view reuse and public-action latency benefit remain unmeasured. | -| Release gate | Current-source saturation, recovery under arrivals and independent-host evidence are incomplete (12, 16, 23) | Run fixed-workload then offered-rate curves at 3/5/10/20 nodes, with actual owner distribution, cgroup/host resources and every acknowledged result checked after failure. | -| Verifier corrected; current-source fleet open | Earlier fleet acknowledgement checks ignored issue bodies (29) | All sixteen later GA rate points verify complete acknowledged payloads before and after published-root owner loss. Earlier receipts remain ID/title evidence. The separate unpublished-tail fault never reached final payload verification. | - -The [one-vCPU GA activation-burst probe](../../crab-ltx/perf/README.md#concurrent-ga-activation-under-one-vcpu-2026-09-27) -first checked four distinct Cell graphs at 32 and 256 MiB each, reading the -payload before its first write. All 32 first-write -object-root restores match every expected payload. Four-way activation helped -the smaller case (82–107 ms versus 137–139 ms serial) but hurt the larger case -(393–542 ms versus 244–290 ms). At 256 MiB, checksum preparation alone reads -about 5.8 MB of metadata per Cell. These LTX measurements use shared Host -admission and explicit container limits; they do not use `CellNode`, runtime -SQL-worker sharding or application acknowledgement. Keep the public recovery -gate open and qualify larger roots before changing concurrency defaults. - -The [write-first follow-up](../../crab-ltx/perf/README.md#write-first-ga-activation-under-one-vcpu-2026-09-27) -removes that warming query and records origin reads during every first update. -All 32 additional outcomes and 4,608 restored payloads pass under the same -container limits. Whole-round preparation ranges from 109 to 697 ms at 32 MiB -and 253 to 368 ms at 256 MiB; the slow round is retained. Random source data -differs between processes, so this is not a controlled speedup comparison. -Public-action and worker-assignment qualification remain open. - -### Latest completed RustFS GA rate evidence - -The [complete GA rate report](../../crab-http-server/deploy/cell-issue-fleet/qualification/2026-09-27-ga-rate-curves.md) -records runtime/harness `e9e238b17b3`, the pinned image/provider identities, -all sixteen points and independently recomputed raw-sample results. The sweep -finished at 3/5/10/20 nodes, each capped at one vCPU and 1 GiB, on one four-CPU -host. Ten offered-load points passed; six failed. All sixteen passed exact -acknowledged-payload readback, published-root owner loss and restart checks, -covering 16,906 successful create/read pairs out of 19,200 scheduled. - -At twenty nodes and twenty pairs/s, write p99 was 8,236.080 ms; at fifty -pairs/s only 1,456 of 3,000 scheduled pairs completed successfully. Actor queue, -archive check and durability-proof wait remain measured investigation targets: -their p99 values at twenty pairs/s were 3,929.068, 4,634.662 and 1,245.374 ms, -while WAL/LTX capture p99 was 10.006 ms. These overlapping distributions cannot -be added or subtracted. They do not isolate CPU, provider or background-work -contention, and this shared host cannot establish independent-node capacity. - -The following unpublished-tail fault rejected its owner/state/epoch guard -before SIGKILL or disk deletion. It therefore provides no acknowledged-tail -recovery proof. The later fault-harness change retains the rejected control -observations; the original artifact cannot identify which field changed. - -**Major qualification gaps remain.** Repeat the rate and fault gates on the -combined source, resolve the retained availability failures, and qualify hot -Cells, recovery storms and independent failure domains before assigning -supported throughput or latency limits. The -[twelve subsequent GA activation-history runs](../../crab-ltx/perf/README.md#rustfs-ga-verification-2026-09-27) -verify fresh/sparse/hydrated/resumed payload recovery separately; they do not -replace application load or fleet recovery evidence. - -### Earlier offered-rate evidence: recovery and capacity gates failed - -[Run 36265830657](https://github.com/crabbuild/crab/actions/runs/36265830657) -finished with failure. Its image and stage runner use `8e61bcf3ad2`; the job -and fault harness use `f3c4d56416f`. The imported ARM64 image digest is -`sha256:c2b47dfab9b0e6db4cedc411a68ea8f8d106add757e13edacb1c8b8dfb38d79e`. -Each node has one-vCPU/1-GiB limits; all nodes, RustFS and the load generator -share four CPUs and 16,722,006,016 bytes of host memory. This measures shared-host -contention, not independently provisioned node capacity. - -Each point offers create/read pairs for sixty seconds across twenty Cells. -Every acknowledged write joins to an execution owner. All expected owners -execute work. Load schema 7 checks the issue number, title and complete body. - -| Nodes | Offered pairs/s | Acknowledged / scheduled | Write p99 ms | Read p99 ms | Gate result | -| --- | --- | --- | --- | --- | --- | -| 3 | 5 | 300 / 300 | 42.243 | 15.240 | Pass | -| 3 | 20 | 1,200 / 1,200 | 53.343 | 17.590 | Pass | -| 3 | 50 | 2,997 / 3,000 | 123.177 | 38.983 | 3 late arrivals; recovery passes | -| 3 | 5 repeat | 300 / 300 | 44.790 | 14.696 | Pass | -| 5 | 5 | 300 / 300 | 58.828 | 23.932 | Pass | -| 5 | 20 | 1,200 / 1,200 | 76.092 | 34.492 | Pass | -| 5 | 50 | 2,989 / 3,000 | 413.899 | 171.747 | 11 late arrivals; recovery passes | -| 5 | 5 repeat | 300 / 300 | 61.325 | 25.580 | Pass | -| 10 | 5 | 300 / 300 | 94.603 | 51.798 | Pass | -| 10 | 20 | 1,200 / 1,200 | 233.176 | 121.692 | Pass | -| 10 | 50 | 2,121 / 3,000 | 2,587.981 | 1,509.477 | 834 capacity and 45 late arrivals; recovery passes | -| 10 | 5 repeat | 300 / 300 | 74.793 | 45.221 | Timed arrivals pass; post-loss readback fails | - -These are observed points, not supported service limits. The ten-node 50-pair/s -point reached the client's 64-pair concurrency cap; successful admitted requests -do not erase the 879 missed arrivals. The later five-pair/s point returned to -lower latency, but its recovery check failed. The twenty-node stage and the -separate unpublished-tail fault were **not run**. The job did not time out. - -The final point drained publication and verified all 300 acknowledgements -before killing `node-10`. `work-20` recovered on another session in 10.883 seconds -with the same root. Reading `work-01` from that same dead owner then failed six -times on different entry nodes with `node recovery is already claimed`. -This is failed post-loss availability and incomplete recovery proof; the report -does not establish data loss. See finding 30 for the authority-boundary defect. - -Action durations at the saturated ten-node point identify where to investigate: - -| Measured phase | p50 ms | p99 ms | -| --- | --- | --- | -| HTTP write | 1,159.720 | 2,587.981 | -| Archive-state check | 477.083 | 1,448.518 | -| Cell invocation | 618.742 | 1,612.772 | -| Actor queue | 0.068 | 933.306 | -| Durability proof wait | 329.956 | 698.034 | -| Publication queue | 0 | 899 | -| Publication completion lag | 371 | 1,460 | -| SQL worker queue | 0.040 | 4.640 | -| SQL worker execution | 2.750 | 21.136 | -| WAL/LTX capture | 0.528 | 7.932 | - -Durations overlap; their percentiles must not be added or subtracted. The -archive check is a separate pre-command Cell query (finding 31). LTX capture -and SQL worker execution are small compared with end-to-end latency in this -workload. That does not rule out cold-demand or large-database checksum costs. -Of 2,121 writes, 1,235 use object proof and 886 use follower proof. Forwarded -writes number 1,930, with p99 2,597.073 ms versus 1,851.544 ms for 191 local -writes; this observational split does not isolate the causal forwarding cost. - -The collector now joins publication events by execution node, Cell, -incarnation and commit sequence. Replaying all twelve raw trace sets matched -**13,507 acknowledged writes to successful publication completions**. Existing -response/proof/owner joins remain byte-for-byte equivalent as JSON values after -excluding the new publication fields. This does not change the failed -post-loss recovery result. - -At ten nodes and 50 pairs/s, publication queue p95/p99 was 505/899 ms and -completion lag p95/p99 was 1,121/1,460 ms. Follower-proof winners alone had -publication lag p99 1,633 ms. The per-action difference between completion lag -and queue wait has p99 1,271 ms across all writes; it includes root preparation, -worker binding, authority CAS and confirmation, so it is not provider latency. -These producer measurements have millisecond resolution; a reported zero queue -wait means less than one millisecond. They support measuring publication work -alongside pre-command queueing before choosing batching or provider changes. - -Each joined action retains publication status and observed queue/retained-work -fields. Summaries count `not_observed`, `started`, `failed`, `completed` and -`recorded` populations explicitly; only completed publications enter publication -latency percentiles. A killed publisher can remain `started` while recovery -preserves its follower acknowledgement. A recorded retry must not inherit the -original command's publication timing. Missing completion is never fabricated -as zero latency. Duplicate, mismatched-scope or impossible event pairs cannot -be used as a successful timing sample. The load and fault acceptance gates -remain unchanged; incomplete timing populations cannot establish a supported -performance profile. - -Artifact `cell-fleet-qualification-36265830657` retains all twelve reports, -arrival samples, action joins, node metrics and Compose logs. SHA256 of -`load-10-04.json` is -`b74c6ff2f84dab4d9763668235cae82ef95c5c35cafe8cdc10e8dd842f826ecd`. -Rebuild joins from the retained raw samples and node logs with -[the trace joiner](../../crab-http-server/deploy/cell-issue-fleet/action_traces.py) -before summarizing older artifacts; previously joined files lack publication -status. For example, from the downloaded artifact directory: - -```sh -python3 /path/to/Crab/crates/crab-http-server/deploy/cell-issue-fleet/action_traces.py \ - --samples load-3-03.samples.jsonl \ - --node-log node-01=load-3-03.traces/node-01.log \ - --node-log node-02=load-3-03.traces/node-02.log \ - --node-log node-03=load-3-03.traces/node-03.log \ - --output publication-actions.jsonl -``` - -The command writes new action and `.summary.json` files and refuses to replace -existing artifacts. All 62 harness tests pass, including unfinished/failed -publisher, recorded reply, scope, ambiguity and timing-order regressions. -A corrected-source fleet rerun must retain the same arrival, body verification, -and recovery gates before changing the performance verdict. - -### Earlier fixed-rate fleet evidence - -[Run 36255479387](https://github.com/crabbuild/crab/actions/runs/36255479387) -passed all stages and the unpublished-owner fault. The image and stage runner -use `c248fcaad78`; the fault harness uses `15b608452d9`. Each container is -limited to one vCPU and 1 GiB, but all containers and RustFS share a **four-CPU, -16,722,006,016-byte ARM64 host**. Twenty container limits do not provide twenty -independent CPUs. This qualifies a shared-host fixture, not scale-out capacity. - -Every stage offered five create/read pairs per second for sixty seconds across -the same twenty Cells. All 1,200 pairs succeeded without retries. Every write -joined to an execution owner and every acknowledgement survived the subsequent -published-root owner-loss check. Latencies below are milliseconds. - -| Nodes | Write p50 / p99 | Read p50 / p99 | Executed writes per owner | -| --- | --- | --- | --- | -| 3 | 23.145 / 41.127 | 5.879 / 12.934 | 90–105 across all 3 | -| 5 | 29.755 / 60.734 | 11.473 / 19.100 | 60 on each of 5 | -| 10 | 29.267 / 68.222 | 13.279 / 35.782 | 30 on each of 10 | -| 20 | 65.484 / 160.765 | 64.709 / 120.615 | 15 on each of 20 | - -Only two of the 1,200 steady-stage writes used follower proof; the others used -object proof. Peak in-flight pairs were one or two. These are low-offered-rate -latency observations, not maximum throughput. Placement is now balanced at -this workload, but the host and per-request/control-plane work still require -separate attribution before explaining the twenty-node latency increase. - -The 180-second fault run offered 900 pairs. It denied immutable object writes -for the target Cell, verified a received follower-proof acknowledgement at -sequence 289 while the root remained at 288, then removed `node-16` and its -local Cell volume. Recovery on `node-06` took 9.572 seconds from the recorded -kill. All **898 acknowledged results** were verified before restarting the old -owner, with no duplicate application results; publication drained afterward. -Two writes to the affected Cell exhausted their retry budget after roughly -3.2 seconds with 503/502 responses. The passing recovery gate does not mean -uninterrupted availability. Retry/resolve behavior and a recovery-time objective -must be qualified together; two failed submissions are retained in the report. - -Artifact `cell-fleet-qualification-36255479387` retains stage samples, node -resource observations, logs, action joins and fault evidence. SHA256 of -`load-20-stage.json` is -`62839410bd9f09a90b64bbd9650ba956934f0e07ef6f0b98f3bb363a2195ff5a`; -`fault-20/report.json` is -`cdbeacbb375fe2265efa6a51a6a5535fcd2c69563f2d5171b818b893f4601c49`. -This closes the earlier image's balanced fixed-load and acknowledged-tail -recovery gates. It does not qualify the newly integrated read-replica source, -offered-rate saturation, prolonged update/delete churn or independent hosts. - -The current-runtime offered-rate attempt -[36263054379](https://github.com/crabbuild/crab/actions/runs/36263054379) -imported the qualified image from run 36261394085, but both attempts stopped -before traffic while Compose pulled auxiliary images. The first identifies -`bucket-init`'s pinned public ECR AWS CLI image with `toomanyrequests: Data -limit exceeded`; the retry reports `toomanyrequests: Rate exceeded` without -identifying a service. Neither attempt supplies a latency or throughput sample. -The retained attempt logs separate this infrastructure failure from runtime -behavior. These attempts used the earlier schema-6 load harness. - -The 1,200 joined actions from successful run 36255479387 also provide matched -phase evidence. The table reports p99 in milliseconds over 300 acknowledged -writes per stage; its columns overlap and must not be added. - -| Nodes | HTTP write | SQL worker queue | SQL worker execution | Proof wait | HTTP outside typed invocation | -| ---: | ---: | ---: | ---: | ---: | ---: | -| 3 | 41.127 | 0.129 | 3.699 | 22.427 | 13.804 | -| 5 | 60.734 | 0.138 | 4.421 | 31.488 | 17.774 | -| 10 | 68.222 | 0.127 | 4.084 | 31.362 | 25.989 | -| 20 | 160.765 | 1.148 | 5.551 | 101.063 | 67.578 | - -At twenty nodes, capture p99 was 2.964 ms, inside worker execution. This -low-offered-rate workload therefore points toward proof/publication and the -HTTP request path for the next measurements. It does not support increasing -SQL worker count as the first remedy. The HTTP remainder is computed per -matched action on one entry node before taking percentiles; it does not isolate -routing, enrichment, authentication or network cost. Those older logs lack -the newer boundary events, and cannot supply their missing durations. - -The common [trace summarizer](../../crab-http-server/deploy/cell-issue-fleet/action_traces.py) -now emits these distributions automatically in stage and fault reports, grouped -by local/forwarded route and winning proof. Two focused cases protect matched -subtraction and absent recorded-response observations; all 58 harness tests pass. -Reprocessing the retained actions reproduced 72 independently computed phase -percentiles. Source/summary SHA256 receipts and the summaries are retained in -`fleet-36255479387/` beneath this checkout's external target. This is a new -analysis of the earlier image, not current-runtime throughput qualification. - -### Measurement contract - -The fleet workflow budgets orchestration separately from application latency: -ten minutes for image import, 200 for all sixteen rate points, fifteen for the -unpublished-tail fault, and five each for evidence collection and upload, within -a 240-minute job. Each point includes placement convergence, complete -acknowledgement readback and owner recovery in addition to its timed arrivals. -Explicit step limits leave artifact time if a phase stalls; a phase timeout -fails qualification, and partial reports cannot establish completed capacity. -The existing arrival durations, placement and recovery deadlines, integrity -checks and overload classification remain the acceptance gates. GitHub defines -[step timeouts as process limits and job timeouts as job cancellation](https://docs.github.com/en/actions/reference/workflows-and-actions/workflow-syntax#jobsjob_idstepstimeout-minutes). - -Report separate latency distributions for resident local actions, forwarded -actions, cold first query, cold first mutation, and recovery. Hold the database, -changed pages, payload entropy and durability contract fixed. The replica-cost -runner previously used immediate capture for fresh databases and deferred -capture for sparse ones, confounding the comparison. The follow-up to -`3ced0777a6f` uses the same instrumented host, deferred capture and exact-cut -pruning for both. Six local RustFS runs independently restored every payload -from the final root. Their [matched measurements](../../crab-ltx/perf/README.md#matched-capture-and-provider-restore-2026-09-26) -are small-database diagnostics, not a service latency or representation-only -performance claim. - -The current public-host follow-up ran three release processes against local -RustFS `1.0.0-beta.8-glibc`, each with 100 generated-client local actions, -100 forwarded actions and one owner takeover. Every action's typed query -verified the expected count at its committed receipt; takeover verified the -final count of 200 in each process. This aggregate check is narrower than the -fleet runner's individual acknowledgement readback. One unconstrained macOS process hosted the three -`CellNode` instances with static signed loopback routing. This measures the -application boundary but excludes HTTP ingress, dynamic placement, independent -hosts and the one-vCPU/one-GiB profile. - -| Run | Local action p50 / p95 / p99, ms | Forwarded action p50 / p95 / p99, ms | First verified read after takeover, ms | -| ---: | ---: | ---: | ---: | -| 1 | 13.423 / 18.520 / 140.954 | 14.712 / 22.992 / 115.160 | 77.693 | -| 2 | 13.623 / 18.000 / 73.027 | 15.059 / 21.583 / 82.120 | 66.413 | -| 3 | 12.798 / 15.059 / 78.377 | 14.832 / 21.213 / 84.782 | 72.769 | - -Local object-proof wait p50 was 11.653–12.432 ms and its p99 was -71.814–139.729 ms; forwarded proof-wait p50 was 11.815–11.894 ms and its p99 -was 78.895–112.634 ms. This interval includes publication queueing and proof -work. Its long tails justify attributing those phases next; the retained -percentiles cannot isolate provider latency or be subtracted from action -percentiles. These serial lanes do not establish saturation throughput. - -Measured source: `19d7ba3fa6d` plus the separately staged reference-suite -relocation, effective index tree `4a152bac8a494fb16ae617a4889c988cb32d53b7`. -The staged patch, binary/host/provider identities, all six action distributions -per run and log hashes are retained in `public-host-rustfs-20260926/` beneath -this checkout's external target. These results do not qualify an unmodified PR -checkout or resolve the relocation's inventory approval. The earlier -[2026-09-25 measurements](../../crab-cell-app/performance/2026-09-25-public-host-rustfs.md) -used a different source and RustFS environment; this is not a controlled -before/after comparison. Recovery times cover fixture fencing and takeover, -not production failure detection or lease-expiry latency. - -**Is this the best fix, rather than only a plausible one?** Bounding demand -read-ahead at a cached suffix removes reproduced duplicate work using the -existing hydration rule and exact-view identity. It needs no new format or -cross-root cache contract. Worker isolation and publication drain have larger -architectural consequences; choose their implementation using action traces -and interference measurements. Keep the current VFS/LTX authority model while -resolving these specific costs. - -The [AMD64 Compose run](https://github.com/crabbuild/crab/actions/runs/36246017568) -and [deep property run](https://github.com/crabbuild/crab/actions/runs/36246048251) -passed at `e1d4052b28e`. They exclude the subsequent checksum change. The -[ARM64 run](https://github.com/crabbuild/crab/actions/runs/36244732114) at -`1a8c4ad670c` failed its elected-successor membership assertion. Its diagnosis -remains open; the successful AMD run neither reproduces nor explains that -failure. The [deep property run at `64c66f20200`](https://github.com/crabbuild/crab/actions/runs/36248053477) -also passed, including the sealed-overlay change. None of these runs supplies a -service saturation curve. - -The follow-up recovery audit reproduced a qualification gap shared with the -compared main snapshot. The first-fault collector and both receipt selection -checks used the startup log's members even though the receipt already contains -the active log observed after the follower-only acknowledgement. The -[durability supervisor](../../crab-cell-host/src/durability.rs) can retire a -fully covered log and recruit different members before that write. Two public -receipt regressions demonstrated rejection of a valid later cohort and -acceptance of an inactive acknowledging log. Selection now uses the active -post-write log; the collector also requires unchanged Cell ownership across the -write and unchanged log epoch/members through its last pre-kill observation. -Failure output retains all three log observations and the successor control. -The second-loss path already checks its replacement member after the write; -the fallback path deliberately verifies a nonmember after object coverage. -This corrects the reproduced evidence gap. The earlier ARM log omitted the -later cohort and successor, so attributing that failure to this gap still -requires a live rerun. The five public receipt tests, eight private validator -tests and six recovery-metric collector tests pass. Runtime all-target Clippy, -Bash parsing and ShellCheck also pass. No runtime election, durability mode or -qualification threshold changed. - -The [fault-during-traffic driver](../../crab-http-server/deploy/cell-issue-fleet/FAULTS.md) -now binds its kill trigger to a received acknowledgement and an unpublished -fleet-proof action. It rechecks the active cohort, removes the owner's process -and named Cell volume, verifies every acknowledged issue before restarting that -owner, scans public issue lists for duplicate effects, and requires publication -to drain afterward. Cleanup failures retain the original failure and cannot -produce a pass. The 35 load, trace and fault tests pass. The pinned AWS CLI and -local RustFS also pass the isolated target-prefix denial/cleanup probe. Those -checks qualify the injector; the complete live fault and stage curves still -need their own retained report. - -Both exact-source container gates now pass at `a3638ef7e55`: -[AMD64](https://github.com/crabbuild/crab/actions/runs/36249416611) and -[ARM64](https://github.com/crabbuild/crab/actions/runs/36249443776). They include -the corrected acknowledging-cohort receipt check. They neither establish the -cause of the earlier ARM failure nor provide a saturation curve. The separate -[deep property run](https://github.com/crabbuild/crab/actions/runs/36249445675) -also passed at that revision. - -The local exact-source fleet attempt at the same revision stopped before -serving traffic: the Docker volume had 29,724,612 KiB free and the node reported -`Cell runtime requires at least 20 GiB usable local disk`. The existing reserve -means that raw free space was insufficient. The failed report and node logs -remain retained; no stage or latency result was produced. The container workflow -now also accepts a successful image run for a fresh-worker fleet qualification, -using the image source's stage runner and separately identified fault harness. -This provides a reproducible environment for the pending live proof without -lowering the runtime's disk admission requirement. - -The fault driver now also retains survivor CPU/memory snapshots and runtime -metrics while the selected owner is absent. The ordinary load path still -requires every node; deliberate absence is limited to the fault driver's owner -after its kill begins. A snapshot racing that transition may retry once with -the cause retained; unrelated loss or missing statistics fails the gate. Every -received write joins to its actual execution owner using surviving logs and the -removed owner's saved log, with before/during/after counts based on client -dispatch time. Read and failed-attempt owner attribution remain open. The four -new regressions cover deliberate versus unexpected loss, snapshot/kill races, -missing statistics and retained removed-owner traces. Live qualification must -use this harness revision before those observations close a measurement gate. -The missing-statistics regression also fails against `6fc1bbc1ceb`: the previous -observer accepted an empty Docker stats response without reporting an error. -All 39 Python harness tests and the documentation checks pass after the change. - -The [fresh-worker fleet run](https://github.com/crabbuild/crab/actions/runs/36251209972) -completed all fixed-load stages at `a3638ef7e55`, then failed the fault driver's -precondition: every Cell was on one owner. The source and per-stage results in -finding 12 show why healthy nodes and balanced ingress do not qualify balanced -execution. Recovery during arrivals remains unqualified; the fault was never -injected. That run used the `6fc1bbc1ceb` fault harness and therefore also -predates the survivor-observation additions at `3ced0777a6f`. - -## Priority after the implemented changes - -The following is the current execution order. Numbered findings below retain -their original baseline and subsequent implementation evidence. - -| Order | Remaining gap | Decision and acceptance gate | -| --- | --- | --- | -| 1 | Corrected-source recovery and capacity evidence (12, 16, 23, 29–30) | Repeat full-body readback after process loss with multiple successor nodes; finish all 3/5/10/20-node stages and the follower-only tail fault. Count missed arrivals and actual owners; preserve shared-host limits in the verdict. | -| 2 | Application preflight adds a serialized Cell query (31) | Seven issue/comment/label commands now check archive state in their transaction; RustFS proof covers the removed query, concurrent archive ordering, duplicate delivery and recovery. Compare matched load and cover the remaining mutation families. | -| 3 | Publication delay grows under saturation (4–5, 17) | Split provider GET/HEAD/PUT, root preparation, worker binding and confirmation. Measure logical commit advance and uncovered age/bytes across repeated debt thresholds before selecting metadata reuse or consecutive-root coalescing. | -| 4 | Replica routing repeats provider work; root refresh loses demand-page reuse (27) | Measure provider operations per response and hot-set bytes fetched after small updates. Compare bounded observation coalescing and verified immutable-frame reuse while retaining fresh response authority. | -| 5 | Synchronous demand faults, installation and cleanup block a shared SQL worker (9–10, 19) | Run cold and resident Cells on the same worker. Separate origin wait, installation and confirmation; evaluate scheduling changes only after that attribution. | -| 6 | Writable activation and checksum residency scale with total page count (3, 14, 20, 22, 25) | Qualify merged-overlay release and the local read/merge change; evaluate lazy authenticated metadata separately. Measure first query, first mutation and recovery storms at fixed changed-page count. | -| 7 | Cache fills return verified reads before persistence; service benefit is unqualified (18) | Measure first mutation, required host-job interference, skipped-cache origin traffic and foreground tails with slow local syncs. | -| 8 | Fixed read-ahead and fragmented hydration amplify object reads (8, 21, 28) | Compare point, random and scan workloads on one fixed root; count useful/fetched bytes, refresh reuse and concurrent duplicate ranges before changing the window. | -| 9 | Checkpoint and full-image tails remain insufficiently sampled (6, 11, 13) | Sustained updates/deletes, pinned readers, large changes and simultaneous maintenance under 1 GiB; preserve checkpoint ordering and exact recovery. | - -These priorities identify code-supported risks and missing evidence. The earlier -trace analysis attributes 13,507 writes at 3/5/10 nodes, including saturation, -but that run failed recovery and never reached twenty nodes. The later GA sweep -completed all four node counts and published-root recovery; its six load-gate -failures and rejected unpublished-tail fault remain failures. Combined-source -scale and saturation proof remain open. The best next fix should remove work -from a measured critical path while retaining the existing authority and -durability contracts. Raising concurrency or queue capacity alone does not -meet that criterion. - -### Design refinements required before the next implementation - -**Is this the best fix, rather than only a plausible one?** Keep the existing -single writer, authenticated directory and ordered publisher. Range compaction -has a reproduced amplification problem and now has selected-range proof. -Choose the next change by the path it improves: cache completion for cold reads, -checksum blocks for activation/capture, or publication work for a hot Cell. -These costs need separate experiments; a queue-size increase addresses none -of their underlying work. - -1. **Make checksum optimization two bounded changes.** - The baseline [capture persistence](../../crab-ltx/src/pages.rs) copied the - full fresh-memory base after each cut and read restored checksums eight bytes - at a time. The local change below reuses the owned array and bounds read - windows without changing the format. Qualify those changes, then - evaluate loading authenticated checksum information from the existing - directory leaves on demand. Each leaf already binds checksums and locators; - a second remote checksum-object graph needs evidence that those leaves are - insufficient. Prove truncate/regrow, aggregate checksums, post-seal failure - fencing and the minimal-feature local path in both steps. Lazy loading must - not merely move the full activation scan into the first mutation or add - unbounded network waits to synchronous capture. -2. **Specify cache work ownership before removing the await.** - The baseline [directory reads](../../crab-ltx/src/replica/directory.rs) - awaited admitted persistence before returning verified origin bytes. The - [cache](../../crab-ltx/src/environment/directory_cache.rs) syncs each fill - and rewrites the membership index; recency updates also scan the entry queue. - A host-owned fill mechanism must bound retained bytes, deduplicate keys and - retain admission through dispatched work. Batch or reconstruct membership - only with restart and disk-accounting proof. Cache persistence is derived - state, so losing an optional fill must cause a verified origin read, never - loss of an acknowledged command. Keep this policy within the LTX host. -3. **Do not treat asynchronous hydration as asynchronous SQLite.** - [Demand reads](../../crab-ltx/src/paged_io.rs) still wait in `recv_timeout` - while the [SQL worker](../src/cell/worker/run.rs) owns the connection. - SQLite's [I/O methods](https://www.sqlite.org/c3ref/io_methods.html) return - synchronously. Prefetch or moving an idle executor can reduce interference; - neither can preempt an in-progress callback. Any worker policy must preserve - connection ownership, per-Cell order, deadlines and fencing. Existing - hydration-fetch isolation tests do not prove demand-read isolation. -4. **Measure publication in logical commits as well as roots.** - [Compaction](../src/publication.rs) publishes a representation change at the - same commit sequence. Counting roots/s as commands/s therefore overstates - progress; future root coalescing would change the ratio in the other - direction. Track the authority's published commit-sequence advance, oldest - uncovered acknowledged command, retained bytes and response-proof winner. - Steady accepted work must drain without growing backlog. Preserve each - stable receipt, effect order and exact predecessor when evaluating coalescing. -5. **Use complete action traces for the architecture decision.** - The latest attributed run offered only five pairs/s and observed one pair - in flight; all 300 writes used object proof. It is a latency probe for that - revision. Require sustained update/delete/skew workloads and repeated - 3/5/10/20-node stages before declaring high throughput. The public - `CellNode`/application-handle suite must retain stable-ID descriptor and - compile-fail proof, duplicate delivery, ambiguous results, compatible - rollout and owner-loss readback. Independent-host qualification follows - the shared-host Compose evidence. - -Review evidence connects the public HTTP/CellNode entry points to actor and -worker ownership, `Db` capture, replica preparation, immutable storage and -authority CAS. Sibling paths include fresh/resumed capture, native/shared-bundle -compaction, cold/resident queries and command/effect confirmation. The detailed -findings name their tests and compared main behavior. Current-source service -percentiles, sustained drain and independent failure domains remain missing; -this audit makes no production latency or throughput claim. - -The [scheduled RustFS measurements at `c12b41ef638`](../../crab-http-server/deploy/cell-issue-fleet/qualification/2026-09-26-scheduled-baseline.md) -now supply a fixed-20-Cell, 3/5/10/20-node series and one 20-node repeat at five -create/read pairs per second. All 1,500 pairs passed without retries. However, -20-node write p95 varied from 347 to 68 ms between runs, execution was spread -across fewer nodes than ingress, and almost every sampled runtime response -used object proof. These observations make placement/phase attribution and -follower-path qualification the next measurement priorities; they do not -establish the dominant bottleneck or qualify later compaction changes. - -### Latest action evidence changes the next experiment - -The [attributed RustFS run at `e50055c48bb`](../../crab-http-server/deploy/cell-issue-fleet/qualification/2026-09-26-action-audit.md) -completed 300 create/read pairs across 20 Cells and three nodes at five pairs/s. -Write p50/p95/p99 was 20.738/35.213/38.973 ms; read p50/p95/p99 was -6.561/11.445/21.322 ms. Every write joined to an object-proof response; -122 were local and 178 forwarded. The later owner restart failed its disk -budget, leaving the overall qualification failed and 5/10/20 stages unrun. - -At this offered rate, SQL-worker queue p99 was 0.068 ms while proof-task wait -p99 was 20.073 ms. Local write p95 was 23.188 ms; forwarded write p95 was -37.750 ms. Within each action, HTTP time outside the typed invocation had -forwarded p50/p95 of 7.482/14.065 ms. That interval includes routing and -response enrichment, not just transport. Twenty captures contained checkpoint -work; their capture p95 was 7.696 ms versus 0.510 ms for the other 280. -These are overlapping measurements and observational groups, not additive -phase percentiles or a controlled experiment. - -The immediate measured-work sequence is therefore: - -1. Measure the removal of the unnecessary label query for empty selections (24), - and add routing/query/enrichment attribution to locate the remaining HTTP - time. Keep the generated typed client and committed receipt as the app - boundary. This is the smallest code-supported optimization of the measured - action; an A/B run must establish its latency benefit. -2. Restore a valid qualification environment and complete fixed-workload scale - stages, then offered-rate curves. All 300 winners were object proof; this - run does not assess the follower response path or publication backlog under - saturation. Queue expansion has no supporting evidence from this load. -3. Qualify the range-proportional compaction change under sustained write debt - (4–5). For recovery-size scaling, use authenticated - checksum blocks (3, 14). Cold-demand worker isolation (9), cache fills (18) - and placement under continuous traffic (12) require their own workloads. - -## Architecture decision after this audit - -Keep one SQLite writer per Cell and immutable, verified LTX roots behind the -authority CAS. The best next change is the smallest ownership change that -removes a measured wait or repeated work while preserving those contracts. -This is not yet a verdict that the current PR meets the performance plan. - -1. Complete peer admission and same-worker interference qualification first. - A low-latency local capture cannot compensate for an ingress rejection or - a worker waiting on another Cell's storage request. -2. Buffered compaction, admitted merge jobs and range-proportional directory - work/scratch are implemented. Qualify sustained drain before adding - publication concurrency. Concurrent work must - never create competing root publishers for one Cell. If publication still - cannot drain, evaluate a - bounded batch of consecutive cuts with one covering root and separate, - stable command receipts. -3. Treat lazy checksum loading as a second design step after measuring the - existing eager walk. It needs authenticated old-checksum lookup and exact - aggregate validation; deferring validation alone is not an optimization. -4. Give hydration an asynchronous fetch stage and a short owner-thread install - stage. Ordinary SQLite demand reads still use synchronous - [VFS callbacks](https://www.sqlite.org/c3ref/io_methods.html); an asynchronous - provider thread does not make an in-progress SQLite statement yield its - SQL worker. Worker reassignment can help idle executors but cannot move a - connection with an active call. Any broader scheduling change needs a - separate ownership design and the one-vCPU interference proof. - -At each step, compare the same public application action, database size, -payload entropy, durability mode, offered arrivals, and resource profile. -Record successful response latency, failed arrivals, publication drain, and -recovery together. The service gate must include Entity, Shard, Workflow, and -read-model operations through public handles, as well as the issue service. - -### Turn the remaining directions into implementation gates - -The next changes need explicit bounds and ownership, in addition to faster -microbenchmarks. These are proposed acceptance gates, not achieved SLOs. - -| Order | Change boundary | Required result | -| --- | --- | --- | -| 1 | HTTP action → runtime receipt → winning proof → client response | Correlate submission ID, attempt ID, Cell, owner, commit sequence, and actual HTTP acknowledgement in traces/raw samples. Keep IDs out of metric labels. Report queue, SQL/capture, proof and confirmation on the same action; do not add unrelated histogram percentiles. | -| 2 | Directory read and cache installation | Release origin admission after bounded transfer/verification. Then evaluate returning verified bytes before a bounded cache fill completes. A slow or canceled cache fill must not retain network admission, escape byte/job/disk accounting, or make cache contents authoritative. | -| 3 | Hydration and SQL ownership | Fetch authenticated pages asynchronously, then install a bounded batch on the owning worker, checking that owner writes have not superseded those pages. Give foreground work priority between batches. An active SQLite demand read still blocks its worker; retain that limit until a separate ownership solution passes same-worker tests. | -| 4 | Compaction and root publication | Reuse unchanged authenticated directory branches. With the selected range held fixed, metadata work and scratch should scale with affected locators/branches rather than all descriptors and database bytes. Preserve newest-wins, truncate/regrow and exact predecessor checks; benchmark repeated 31/32/33-segment crossings. | -| 5 | Checksum state | Use bounded authenticated checksum blocks for old-value lookup and transactional updates. With changed pages held fixed, a fresh Cell must not copy a database-sized array every cut. Lazy activation must retain aggregate verification, including deleted suffixes and later mutations. | -| 6 | Sustained service and recovery | At fixed Cell count and offered arrivals, measure acknowledgement rate, publication rate, retained bytes, oldest unpublished age, rejection rate and latency together. Run long enough to cross repeated compaction/checkpoint cycles. Fault a proven follower-only acknowledged tail during arrivals and verify every acknowledged result after owner/local-data loss. | - -**Action-attribution implementation:** the shared HTTP submission validator -records the stable application input under the existing server request span. -The public typed client records its runtime attempt and committed receipt -before the issue/label/status/check output adapters return their payload. -The owner carries Cell, incarnation, request ID and owner session through actor -and worker queues, capture, winning proof and final successful reply. IDs live -in traces, never metric labels. Effects share worker/actor tracing under their -effect identity; the issue-load join qualifies only command writes. - -The scheduled fleet runner retains every HTTP attempt and the successful -response's `x-request-id`. It collects each node's log before owner loss and -joins every acknowledged write to exactly one matching owner response and -commit sequence. Missing phases, conflicting owners, mismatched receipts or -ambiguous duplicate responses fail attribution. The report records actual -write owners and forwarded writes; the previous pre-load-owner estimate was -removed. A stored runtime result has source `Recorded` and does not invent a -new capture or proof. A new runtime attempt deduplicated by the application can -still commit a new runtime receipt; its stable submission ID remains separate. - -The real RustFS/mTLS HTTP test passed after owner takeover and Git readback; -its actual text-formatted logs also passed the fleet join CLI. One acknowledged -issue write measured 165.890 ms at the client, 164.083 ms to HTTP response -readiness, 72.339 ms in the typed invocation, 4.065 ms on the SQL worker -(including 1.653 ms capture), and 27.227 ms waiting in the proof task. This is -one diagnostic action on a debug build, not a percentile or an isolated -bottleneck measurement. Artifacts beneath the checkout's external target are -`action-trace-rustfs-formatted.log`, `action-trace-rustfs.samples.jsonl`, and -`action-trace-rustfs.actions.jsonl`. - -The timings overlap: capture is inside worker time; worker and proof work sit -inside broader request lifetimes. HTTP response readiness precedes client body -receipt. Proof wait starts when the proof task runs and excludes earlier node- -log submission and actor scheduling. Authentication/routing/activation, query -phases, individual provider attempts, and proof submission need further -attribution before a complete latency decomposition. Replaying a shared-process -test log proves the correlation fields and parser, not cross-container collection. -The later three-node Compose run joined all 300 writes across container logs, -as recorded above. Current-HEAD scale, trace-overhead comparison, query phases -and saturation curves remain required. See the [action trace runbook](../../crab-http-server/REFERENCE.md#attribute-acknowledged-cell-writes). - -Keep checkpoint execution serialized with the managed writer when exploring -background work. The workspace pins `rusqlite` 0.34.0 / `libsqlite3-sys` 0.32.0; -the bundled header reports SQLite 3.49.1. SQLite's -[WAL-reset guidance](https://www.sqlite.org/wal.html) -identifies a write/checkpoint race fixed in 3.51.3 and selected backports. -The current exclusive `Db` contract and serialized worker do not establish -that race is reachable here. Moving checkpoints onto a competing connection -requires dependency qualification and a new capture-order proof before any -performance claim. Offloading file cleanup is a different operation. - -## Findings in execution order - -### 1. Small LTX bodies pay the multipart protocol cost - -**Confirmed at audited revision:** [native segment upload](../../crab-ltx/src/replica/upload.rs) -calls [upload_source](../../crab-ltx/src/replica/compaction/scratch.rs), which -always calls `Store::put_multipart_source_retry`. The -[storage implementation](../../crab-storage/src/store.rs) starts multipart -even when the source fits in one part. The pinned `object_store` 0.14.1 S3 -implementation separately creates, uploads, and completes the upload; 0.14.2 -is pinned by the standalone cost harness. This agrees with the -[S3 multipart protocol](https://docs.aws.amazon.com/AmazonS3/latest/userguide/mpuoverview.html). -The publication counter counts the completed object once, not these requests. - -**Change to evaluate:** read and verify a small, admitted source into a bounded -buffer and use the existing create-only `Store::put`. Keep multipart for larger -sources. Locate the size decision at the storage transfer boundary after -checking its other callers; avoid separate policies in capture and compaction. -Retain source length/digest verification, exact retry bytes, cancellation, -same-content reconciliation, and conflict refusal. - -**Gate:** one successful small body uses one PUT without multipart initiation; -count provider attempts, including retries. Test changed/truncated sources, -lost responses, existing same/different content, and both native and compaction -uploads. Compare public action p95/p99 and publisher drain rate over RustFS. -The current five-object small-write count is not five HTTP requests. - -**Implementation:** native, compacted, and bundled bodies now share a private -LTX transfer function. Sources up to 256 KiB are length/digest verified into -one buffer under the existing host I/O permit, then sent with `Store::put_exact`. -That method retains exact staging paths, create-only writes, same-content -reconciliation after an uncertain response, and different-content refusal. -Larger sources retain multipart streaming. The generic storage multipart API -has other overwrite, staging, and cancellation callers; it is unchanged. -Recovery manifests also retain their existing transfer path. - -Focused tests count transfer calls and restore native, compacted, and bundled -roots byte-for-byte for both sizes. They inject a lost response after a -successful create and reject wrong lengths, changed digests, truncated sources, -and conflicting existing content. These establish transfer semantics, not -public-action latency or sustainable publication capacity. - -The first seven-run RustFS comparison is -[recorded with its raw artifact location](../../crab-ltx/perf/README.md#small-body-transfer-experiment). -Median run p50 moved from 6.64 to 6.16 ms, but p95 rose from 10.95 to -12.14 ms and unchanged local phases slowed. Latency qualification remains -open. The real HTTP/peer RustFS test passed mutations and takeover restore. - -### 2. Persistent directory-cache hits perform durable index writes - -**Confirmed at audited revision:** [DirectoryCache::get](../../crab-ltx/src/environment/directory_cache.rs) -calls `persist_index` after a valid hit and after a missing file. That method -clones and serializes the entire entry map, writes a new file, syncs it, and -renames it. A hit changes in-memory recency, but the serialized index contains -only keys and lengths. The hit therefore persists unchanged content. This -path runs after a memory-cache miss in -[read_node](../../crab-ltx/src/replica/directory.rs); it does not affect a pure -memory hit. The runtime installs this cache during -[Cell acquisition](../src/cell/actor/acquire.rs). - -**Change to evaluate:** eliminate index persistence on pure hits and misses -that remove no indexed entry. Retain durable cache installation and actual -membership updates initially. Keep recency maintenance bounded; its current -queue scan also grows with cache entry count. - -**Gate:** a verified disk-cache hit makes zero writes, renames, and syncs; -an absent unindexed key also makes zero writes. Preserve restart, corruption, -symlink, concurrent-fill, eviction, and disk-accounting tests. Measure with -the memory cache churned, a nearly full disk index, and concurrent Cells. -This removes redundant work without changing cache durability policy. - -**Implementation:** verified hits now update only memory recency; absent keys -persist the index only when indexed membership was actually removed. Durable -installation, invalidation, and eviction retain their writes. A regression -test reproduced the old hit write and now proves neither a hit nor an -unindexed miss creates an index file. Existing restart, corrupt-entry, symlink, -and accounting tests pass. Full-cache concurrent latency remains unmeasured. - -**Remaining cost:** a fill still persists the entire membership index, and a -hit scans the recency queue with `retain` while holding the state mutex. The -entry cap is 16,384. Removing hit fsyncs does not establish constant-time hit -cost or cheap cold fills. Compare hit/fill latency as entry count grows, plus -concurrent activation that churns the process-wide 8 MiB directory cache. -Evaluate bounded recency bookkeeping and batched/reconstructible membership -persistence only with restart, corruption, and disk-accounting proof. - -### 3. Sparse writable activation still reads the complete checksum directory - -**Confirmed:** [prepare_writable](../../crab-ltx/src/replica.rs) awaits -[load_checksums](../../crab-ltx/src/replica/directory/checksums.rs). That function visits -every directory node, authenticates every page entry, writes -eight checksum bytes per database page, and syncs the checksum file before -opening the writer. LTX page bodies are lazy; this metadata walk is eager. -At 4 KiB pages a 10 GiB database alone needs a 20 MiB checksum file, excluding -directory transfer and validation. Tiny bootstrap Cells hide this cost. -The directory format stores 88 bytes per page locator, plus a 32-byte header -per 256-entry leaf. For a dense 1 GiB database at 4 KiB pages, that is about -22 MiB of leaf metadata to read/validate on a cold walk, plus a 2 MiB checksum -file and internal nodes. These are format-derived sizes, not resident-memory -requirements or a supported database-size claim. Measure directory transfer -as well as sidecar bytes when setting recovery targets. -At the audited revision, this async path directly called synchronous filesystem -writes and syncs, including the final checksum barrier, instead of dispatching -them through the host's blocking executor. Slow local storage could therefore -stall its Tokio worker as well as delaying activation; its fleet latency impact -remains unmeasured. - -**Change to evaluate:** first remove finding 2's cache overhead and overlap -independent authenticated node reads within existing host admission, retaining -ordered validation/output. Consider lazy checksum chunks only if this measured -walk still prevents the recovery target. That larger change must preserve the -old checksum for every overwritten/truncated page and the final root checksum. -Move checksum writes and barriers through admitted blocking work, retaining -file and reservation ownership until dispatched work finishes. - -**Gate:** measure first query and first mutation separately for fixed data -sizes, empty/warm caches, and simultaneous owner loss. Count directory requests, -checksum bytes, local syncs, admission wait, RSS, and time to serve. Exact-root -corruption, canceled activation, fresh destination, and checksum-link tests -must remain intact. A fast first page fault does not prove fast activation. -Inject slow filesystem writes/syncs and verify unrelated task progress, bounded -blocking admission, and cleanup when activation is canceled or returns an error. - -**Implementation:** checksum-file creation, bounded 64 KiB writes, final sync, -parent sync, and metadata validation now use `Host::run`. The file owner moves -with each dispatched job and retains dirty admission. Cancellation schedules -cleanup through the same blocking-job ceiling; successful delivery disarms -cleanup only after the validated checksum handle reaches its caller. Normal -errors await cleanup before returning. Cleanup remains best effort when the -filesystem or executor fails, and requires the Tokio runtime to stay alive. -This follows the [runtime task contract](https://docs.rs/tokio/1.53.1/tokio/runtime/struct.Handle.html#method.spawn). - -The regression first failed on the old async-thread filesystem call, then -passed with a 40 MB database, disk directory cache, and one blocking-job slot. -Fault tests pause creation, writes, both sync barriers, and final metadata; -unrelated async work progresses, cancellation retains both admissions through -paused cleanup, and the same destination can be retried and queried. Error -injection preserves pre-existing destinations. Exact-root, sparse publication, -directory, and process-kill recovery tests pass. Activation percentiles and -fleet interference remain open. -The real RustFS HTTP/peer test also passes application mutations, owner -takeover, restored collaboration state, and Git reads with this path. -Full-image restore is a separate sibling: bulk writes/syncs already use host -jobs, but initial file setup and scratch cleanup still need the same audit. -Compaction needs that audit too: its scratch creation, final index-file opens, -and `MergedEntries` iteration perform filesystem work from async preparation. -The iterator is consumed by `directory::initial::build_and_upload`, so moving -only the initial opens would leave synchronous index reads on the async path. - -Sibling leaf reads now overlap through an ordered stream capped at eight, -sharing existing host I/O slots. Internal branches stay depth first so sibling -prefetch cannot reorder coverage or aggregate validation. With ten leaves and -100 ms injected delay per GET, the previous scan took 1,000 ms even with four -slots; the changed scan takes 300 ms. One slot takes 1,000 ms and sixteen slots -still take 200 ms, proving the eight-read ceiling in that fixture. These are -virtual-time scheduling results, not RustFS performance percentiles. - -A 40 MB database with 512-byte pages exercises two parent levels. A separate -late-leaf corruption test refuses activation, releases I/O admission, removes -its checksum file, and permits retry after repairing the object. The 26 -exact-root/sparse tests and four activation tests pass. This change applies to -writable activation only; selected-page lookup and exhaustive retention -inventory retain their existing traversal and verification contracts. - -The [real RustFS activation probe](../../crab-ltx/perf/README.md#sparse-activation-over-real-rustfs) -now separates root open, checksum preparation, writable open, and first query. -On a 256 MiB source, three cold-metadata samples per admission setting measured -checksum preparation medians of 1,057 ms with one I/O slot, 420 ms with four, -and 295 ms with eight. All settings fetched 259 objects and 5,797,768 bytes. -Reused metadata required zero origin reads but still spent 57–119 ms in this -phase. This supports bounded read overlap for cold metadata; it does not -attribute the remaining local cost or qualify end-to-end tails. The probe -also fixes an earlier timer that included compaction and a second restore in -the reported restore duration. - -### 4. Range compaction can do whole-graph metadata work on the publication lane - -**Confirmed:** [compaction::prepare](../../crab-ltx/src/replica/compaction.rs) -spools indexes for **all** descriptors, although it spools bodies only for the -selected range. It then merges the final descriptor set and rebuilds the -complete directory. [CellPublisher](../src/publication.rs) checks compaction -after eight appends during quiet periods and forces debt handling before an -append projected to reach 32 segments. That work retains the serialized -publisher token; it can delay following roots even after follower responses. - -The admission estimate also grows with the whole database: -[compaction_scratch_bytes](../../crab-ltx/src/replica.rs) reserves twice the -logical database size plus 64 MiB, every descriptor index, and the selected -compressed bodies. The first term comes from -[full_job_scratch_bytes](../../crab-ltx/src/recovery.rs). A 1 GiB database thus -requires over 2 GiB of scratch admission even for a small selected range. -This is disk reservation, not resident memory or measured peak disk use. -Optimizing the merge alone leaves this admission floor unchanged. - -**Change to evaluate:** reuse authenticated unchanged directory branches and -update locators only where selected segments still supply the current page. -Measure before making compaction concurrent with publication: any such change -needs an exact predecessor check and must discard or safely rebase stale work. -Derive scratch admission from the bounded range algorithm in the same change; -include codec scratch, worst-case output expansion, indexes, and cancellation -lifetime. Keep the current conservative bound until that proof exists. - -**Gate:** run updates as well as inserts on a large base; cross repeated -8-segment promotion and 31/32/33-segment pressure boundaries. Record all-index -bytes, directory rewrites, scratch peak, root lag, foreground p99, and restored -byte equality. Raw library tests around 96/97 descriptors cover descriptor-page -boundaries; they do not substitute for the runtime's earlier debt threshold. - -**Implementation:** compaction now spools only the selected segment indexes -and bodies. The new -[`directory/relocate.rs`](../../crab-ltx/src/replica/directory/relocate.rs) -streams compacted locators through the authenticated current directory. It -replaces a locator only when its object **and byte range** belong to the -selection, preserving newer cuts even inside the same bundle. It checks the -old/new page checksum, validates affected leaves against the final extents, -and preserves each branch's aggregate. Later truncation discards compacted -suffix pages; unchanged subtrees retain their digests. This has the same -authenticated-predecessor boundary as incremental append, not an exhaustive -integrity scan of unchanged dependencies. - -Traversal retains bounded state per level, one streamed index batch, and -at most eight pending directory uploads. Root/descriptor metadata still -depends on segment count. Full-range compaction still reads the whole -directory, and one Cell still has one ordered publisher. The immutable proposal -retains the exact predecessor, TXID, commit sequence and schema; authority CAS -and local-only compaction are unchanged. - -Scratch admission now sums selected compressed inputs and indexes, a -worst-case encoded output bound, the temporary varint index, the output -sidecar, and fixed footer headroom. A regression pauses cleanup after all five -files reach their final lengths and verifies peak logical bytes fit the held -reservation at 512/65536-byte pages. Cancellation continues to retain file, -job and scratch ownership through cleanup. The former assertion that even a -tiny promotion must exceed 64 MiB was removed; capacity refusal still tests -an input/output set larger than its one-MiB admission, and small ranges now -positively prove admission at that limit. - -The initial regression failed on `65b359d8dc2`: two small updates on a 16 MiB -base fetched 247,175 bytes, despite a warm metadata cache. The expanded real -RustFS test uses cold metadata, 512/4096-byte pages, native/shared-bundle inputs -and multiple directory levels; it completes at one MiB scratch and restores -the same database bytes, including a newer overwrite. Its observed reads were -27,993/28,013 bytes and six uploaded objects for 4096-byte pages; -61,738/61,762 bytes and seven objects for 512-byte pages. -These are operation-work measurements, not before/after service percentiles; -the baseline and expanded fixture differ in cache state and page coverage. -The [executable runbook](../../crab-ltx/README.md#verification) records the real -provider command. Retain `range-compaction-before.log`, -`range-compaction-rustfs.log` and `range-compaction-host.log` under the external -target. Repeated runtime debt boundaries, foreground tails and sustained -publication drain remain open gates. - -Six focused compaction cases, the expanded truncate/regrow case, eight host -failure/admission cases and four runtime dispatcher cases pass. Both the -explicit RustFS range test and the public RustFS HTTP/mTLS collaboration, -owner-loss and Git-readback test pass. Replica and minimal-feature LTX builds -pass all-target Clippy with warnings denied; formatting and documentation -validation pass. The unrelated staged reference-suite move still fails its -old layout-inventory entry and remains outside this change. Current-source -container/property CI and sustained fleet performance remain required. - -### 5. Early follower responses do not establish sustainable write throughput - -**Confirmed at audited revision:** [start_publication](../src/cell/actor/requests.rs) removes one -publisher from the active Cell and publishes one queued command at a time. -`prove_command` races external proofs and discards their source before final -worker confirmation. Completed-proof counters cannot identify the proof that -released each successful action. The existing byte admission bounds backlog; -it does not make an arrival rate above publication capacity sustainable. - -**Change to evaluate:** record one response winner plus confirmation time and -publisher queue age. After findings 1–4, evaluate bounded root coalescing only -if a hot Cell still cannot drain. Preserve each command's receipt, outcome, -effect ordering, and replay coverage even if one root covers several commands. - -**Gate:** offered-rate sweeps must hold long enough for compaction and backlog -to settle. Accepted commands/s, published commands/s, root lag, retained bytes, -and rejections must be reported together. Owner loss with follower-only tails, -duplicate delivery, ambiguous publication, and rollout must still work through -public application handles. Increasing queue limits is not a throughput fix. - -**Instrumentation:** the command/effect reply boundary now reports one source -(`fleet`, `object`, or `recorded`), admitted-enqueue-to-response time, and final -SQL worker confirmation time. It records only a successful runtime delivery; -failed results and abandoned receivers add no winner. A later object proof -does not add another response. Structured traces include Cell and commit -sequence; Prometheus labels contain only the finite source. Queries, -migrations, and transport have separate boundaries and are excluded. This -closes response-source ambiguity; sustained drain and full phase attribution -remain unqualified. - -### 6. Bounded I/O does not yet prove foreground latency isolation - -**Confirmed:** [Host](../../crab-ltx/src/environment/host.rs) shares I/O permits -across page faults, uploads, and compaction. Compaction and restore also share -recovery admission. [Paged I/O](../../crab-ltx/src/paged_io.rs) has 32 active -jobs, a 256-request queue, an 8 MiB cache, and a per-view gate. These provide -bounds; no latency-priority guarantee follows from those bounds. Directory -origin reads now release their I/O permit before admitted disk-cache insertion; -the reader still waits for that insertion and its blocking job remains shared. - -**Change to evaluate:** measure permit wait and background interference before -changing concurrency. Release provider permits when transfer/verification no -longer needs them; if measurements justify scheduling changes, reserve progress -for foreground reads and durability while preventing compaction starvation. - -**Gate:** under 1 vCPU/1 GiB per node, overlap cold activation, hydration, -compaction, and writes. Report foreground p99, queue deadlines/rejections, -provider concurrency, and maintenance debt. More concurrent requests must not -silently exceed memory or provider budgets. Test fragmented page spans as well -as the existing contiguous hydration case. - -The sparse bridge adds another shared execution boundary: `Driver::new` uses a -current-thread Tokio runtime. `read_run` decodes and checksums its fetched span -inline before returning to that runtime. Consequently, 32 active async jobs -are not 32 parallel decoders. Compare decode time and driver scheduling delay -before dispatching bounded decode jobs; full restore already decodes its -windows inside `Host::run`. Tokio's -[current-thread and fairness contracts](https://docs.rs/tokio/1.53.1/tokio/runtime/index.html) -require bounded task polling time. No measured decoder bottleneck is claimed. - -### 7. Qualification needs a less favorable payload and arrival model - -**Confirmed at audited revision:** the [cost harness](../../crab-ltx/perf/replica-cost/src/main.rs) -generates `(command * 131 + index) % 251`, repeating every 251 bytes. It is -compressible despite its source comment. Its loop calls `CellReplica::prepare` -directly: no runtime authority CAS, follower race, or scheduled compaction. -The [Compose load generator](../../crab-http-server/deploy/cell-issue-fleet/load.py) -uses one sequential write/read lane per Cell, with the Cell count equal to -node count. Slow responses reduce offered traffic. Its retry-inclusive logical -latencies are useful, but cannot establish an arrival-rate saturation curve. - -**Change to evaluate:** retain the historical fixture and label its entropy. -Add seeded high-entropy bytes, structured application data, update/delete churn, -and variable database sizes. Hold Cell count and offered rate fixed while -changing nodes, then independently vary skew and load. Add scheduled arrivals -with a bounded outstanding limit; report rejected/late arrivals instead of -silently slowing the generator. Bind source, image, target architecture, -filesystem, SQLite and provider dependency versions to each report. - -**Gate:** retain raw samples and repeated runs with enough completed operations -to study tails. Twenty-eight preparation samples cannot establish p99; that -nearest-rank p99 is the maximum. Keep microbenchmark, public action, Compose, -and multi-host results separately labeled. - -**Instrumentation:** the cost harness now labels the periodic payload, offers -deterministic command-seeded random bytes, and records root-preparation calls, -outcomes, bytes, and durations through the existing storage observer. The -[RustFS smoke evidence](../../crab-ltx/perf/README.md#backend-calls-and-payload-entropy) -shows two HEADs and five PUTs per small prepared root; the larger random body -uses multipart. Provider-internal retries remain opaque. - -The Compose generator now keeps a fixed Cell count across node stages. Its -load runner schedules create/read pairs independently of completion, bounds -in-flight work, and records client-capacity rejections and missed scheduling -intervals. Uniform, hot, and skewed targeting retain arrival indices and stable -write IDs in raw samples. Runtime metrics and container resources are sampled -during load; post-load uncovered bytes must drain on every node. Failed runs -retain reports, and acknowledged readback mismatches stop new arrivals. -Controllable HTTP tests cover slow responses, lost responses, and delayed -scheduling. Sustained offered-rate curves, update/delete churn, and full action -phase attribution still need qualification against the current native image. - -### 8. Sparse point faults and bulk hydration use the same read-ahead window - -**Confirmed:** [paged_io::fetch](../../crab-ltx/src/paged_io.rs) asks for up to -64 pages for every cache miss, for both `Sparse` and `Hydrating` origins. -[CellPagedDatabase::read_run](../../crab-ltx/src/replica.rs) selects the first -same-object span in that window, fetches it, and validates every returned frame -before the requested page is delivered. Extra pages enter the shared 8 MiB -view-keyed cache; only pages demanded by SQLite become materialized VFS pages. - -In the 32 MiB RustFS probe, the first `SELECT length(value) ... WHERE rowid = 1` -made two range reads totaling 524,994 bytes. The whole activation, including -open, materialized just four 4 KiB pages. The 256 MiB probe fetched 265,332 bytes -for that query, reflecting a different segment layout. This is observable -read amplification, not proof that a smaller window universally lowers latency: -read-ahead can avoid later requests during scans and hydration. - -**Change to evaluate:** distinguish demand access from bulk progress using the -existing origin classification. Compare demand-only reads, bounded adaptive -read-ahead after sequential access, and the current window. Keep hydration -coalescing, the shared memory ceiling, deadlines, and authentication intact. -Any asynchronous prefetch must retain admission until completion and must not -publish unverified bytes or occupy all foreground slots. - -**Gate:** compare point lookups, random reads, sequential scans, and hydration -on both contiguous and fragmented roots. Record requested/decoded/consumed -pages, useful prefetch hits, wasted bytes, origin calls, CPU, and p50/p95/p99 -under concurrent Cells. An improvement in point-read bytes must not silently -regress scan throughput or starve durable publication. The existing contiguous -hydration test proves coalescing correctness; it does not establish this tradeoff. - -**Additional source gap at `7d3dd9232b0`:** `read_run` computes all directory -spans for its window, then consumes only `.next()`. With fragmented locators, -a 64-page window can produce 64 one-page spans while the call fetches just -the first. Later faults repeat directory parsing and span allocation for -overlapping windows; the byte cache avoids downloads of directory nodes but -does not cache their parsed, context-validated entries. Full restore already -consumes all spans with bounded concurrent fetches in `read_restore_window`. -Evaluate stopping demand lookup at the first useful span and giving bulk -hydration bounded multi-span progress. Keep verification scoped to the exact -root/extents; a digest-only parsed cache cannot silently reuse validation -against a different root. Count parsed entries, discarded spans, fetch waves -and consumed pages on alternating-object roots before selecting a policy. -This is a source-supported opportunity, not a measured latency contribution. - -### 9. One cold Cell can block unrelated Cells on its SQL worker - -**Confirmed:** [SqlWorkerPool](../src/cell/worker.rs) assigns a Cell to a fixed -worker using its ID modulo worker count. The -[worker loop](../src/cell/worker/run.rs) executes one synchronous operation at -a time. A sparse VFS read waits in `paged_io::receive` for its provider result; -the dedicated I/O thread keeps the provider progressing but cannot run another -Cell's SQLite operation on the blocked SQL worker. CPU-derived sizing can -select one worker for the requested 1-vCPU node profile. - -[Background hydration](../src/cell/actor/lifecycle/background.rs) checks the -selected Cell's queue, then sends up to 64 pages through the same worker and -the same global worker-job semaphore as foreground queries. An idle Cell does -not imply an idle worker. On a multiworker node, jobs queued for one busy shard -can also occupy the global admission permits while another shard has capacity. -`confirm_durable` and `confirm_published` use that worker, so this interference -can extend durable response time as well as query time. - -**Change to evaluate:** measure worker admission, shard queue, SQL execution, -and sparse-provider wait independently. Make maintenance admission aware of -worker foreground demand. Prefer admitted asynchronous fetch followed by short -owner-thread installation for hydration. If demand faults still dominate, -evaluate bounded reassignment of idle Cell executors to available workers; -preserve exclusive connection ownership, per-Cell order, cancellation, and -fencing. Merely adding provider I/O slots cannot resolve a blocked SQL shard. - -**Gate:** one cold/fragmented Cell plus one resident Cell on the same worker, -then different workers; slow origin, slow local sync, canceled waiter, and -background hydration cases. Measure resident p99 and confirmation latency, -including waits before worker dispatch. The existing -`sparse_fault_pool_progresses_under_saturated_sql_workers` test places two -Cells on **different** workers and proves I/O progress, not same-worker latency -isolation. Hydration cancellation tests protect admission but do not establish -foreground latency. This scheduling behavior also exists on the main snapshot. - -**Implementation:** worker-job admission now has one permit for each fixed SQL -worker. A queued job waits for its own worker before reserving a node job slot; -it cannot consume another worker's capacity. Dispatched work retains its permit -and ledger reservation through completion even when its caller is canceled. -Shutdown closes every admission queue. Cell assignment, thread count, actor -ordering, lifecycle messages, and publication-confirmation messages are unchanged. - -A public worker regression holds one native operation, queues another Cell on -that worker, and invokes a third Cell on the idle worker. The old global gate -failed its one-second completion bound twice; the per-worker gate passes, and -canceling the queued request leaves no mutation behind. A second fixture uses -real sparse SQLite with delayed object-store reads during hydration and a fully -materialized Cell on the other worker. It also fails on the previous gate and -passes with per-worker admission. This separates admission from SQLite and -storage progress; these controlled delays are not public-action percentiles. -Seventeen focused worker, hydration, and public three-node application tests -pass, as does runtime all-target Clippy. The real RustFS HTTP regression also -passes through remote execution, owner loss, restored collaboration state, and -Git clone/tag reads. These checks protect behavior; they do not replace the -current-source fleet curves. - -**Asynchronous hydration implementation:** preparation selects at most 64 -missing pages on the owner worker; fetch releases worker admission; installation -returns authenticated bytes and their retained-byte reservation to that worker. -Installation checks activation identity and skips pages superseded by owner -writes or truncation. Dropping a fetched batch cannot advance the cursor. - -The strengthened worker regression fails with the committed synchronous -hydration path and passes with the initial draft. A real RustFS diagnostic adds 500 ms -to each GET and queries fully resident Cells on the same and another worker: - -| Diagnostic | Synchronous runtime | Asynchronous draft | -| --- | ---: | ---: | -| Same-worker resident query | 1,026.685 ms | 0.053 ms | -| Other-worker resident query | 0.163 ms | 0.060 ms | -| Hydration plus queries | 1,027.172 ms | 1,526.698 ms | -| Origin read operations during hydration | 2 | 3 | - -These are single debug-build, injected-delay diagnostics against the local -RustFS fixture, not p95/p99 or a service SLO. The before run temporarily used -the committed executor/worker dispatch with the same current test fixture; -the draft source was restored afterwards. They establish the sibling-Cell -wait and its removal, not faster hydration: the draft makes an additional -origin read in this fixture (finding 21). The workload creates a fresh random -database each run, so byte-for-byte traffic attribution needs a fixed-root -comparison. Raw logs are `worker-interference-rustfs-before-matched.log` and -`audit-hydration-rustfs.log` beneath the checkout's external target directory. - -Demand faults remain synchronous, and installation, confirmation and cleanup -still execute on the worker. The initial split needed additional same-Cell -isolation and cache reuse; findings 19 and 21 record those follow-ups and their -tests. The final RustFS diagnostic (`hydration-rustfs-final.log`) measured -0.129 ms for the same-worker resident query, 0.119 ms on the other worker, -and 1,010.155 ms for hydration plus queries, with two origin reads. This remains -a single injected-delay diagnostic. It does not establish service percentiles -or a throughput change across different random database fixtures. - -**Demand-fault follow-up, 2026-09-26:** the ignored -`rustfs_sparse_reads_report_worker_interference` diagnostic now distinguishes -background hydration from a foreground full-payload read. It uses real RustFS, -three Cells and two SQL workers. Two resident Cells each contain a 256 KiB -payload; one shares the cold Cell's worker. The cold Cell contains a 4 MiB -random BLOB. Each process reopens the same authenticated root into six fresh -sparse destinations, alternating added GET delays `0, 20, 20, 0, 0, 20` ms. -Root metadata and the provider remain warm; activation precedes the measured -interval. The observer wakes the sibling queries on the first actual origin -read, with no stale notification carried between trials. - -Three release processes at production source `c248fcaad78` plus the diagnostic -patch produced the following **per-process medians**, three trials per delay: - -| Process | Same-worker query, no added delay | Same-worker query, +20 ms/GET | Largest same-worker query, +20 ms/GET | Other-worker query medians, no delay / +20 ms | -| --- | ---: | ---: | ---: | ---: | -| 1 | 42.709 ms | 413.709 ms | 414.202 ms | 0.029 / 0.027 ms | -| 2 | 44.896 ms | 416.460 ms | 1,013.702 ms | 0.034 / 0.027 ms | -| 3 | 44.359 ms | 416.378 ms | 418.517 ms | 0.033 / 0.029 ms | - -The same-worker query's SQLite callback took 31–50 microseconds across all -18 trials. Nearly all its delay preceded callback entry, establishing shared -worker interference rather than slow resident SQL. The one-second sample is -retained; the current trace cannot attribute its additional delay to provider -retry, host scheduling or another cause. No p99 is inferred from nine samples -per condition. In the separate 500 ms/GET background-hydration phase, the -same-worker query took 0.031–0.039 ms while hydration took 1.009–1.012 seconds. -The asynchronous hydration improvement therefore does not remove demand stalls. - -Every cold payload query performed 16 observed origin read operations and -transferred 4,008,092 bytes. Its returned payload digest matched the original -source database. Repeating that full-payload query made zero new origin read -requests and returned the same digest in 2.964–3.303 ms. -Those full-payload timings include reading and hashing all 4 MiB; they are not -point-query costs. The locked `object_store` 0.14.1 throttle adds its per-call -delay before GET; it does not replace the RustFS provider. The linked SQLite -version was 3.49.1. - -Run against an existing isolated RustFS bucket with credentials in the process -environment: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ -TMPDIR="$HOME/Workspace/crabbuild-target/crab-8bc8/tmp" RUSTC_WRAPPER= \ -CRAB_LTX_TEST_ENDPOINT=http://127.0.0.1:19010 \ -CRAB_LTX_TEST_BUCKET=crab-cell-issue-fleet \ -cargo test -p crab-cell-runtime --release --locked --lib \ - cell::worker::tests::rustfs_sparse_reads_report_worker_interference \ - -- --ignored --exact --nocapture -``` - -Choose the external target directory for the actual checkout. Evidence lives in -`demand-interference-20260926/` beneath that checkout's external target: -`release-{1,2,3}.log`, the extracted JSON records, `source.patch`, and -`source.json` with parent/source/binary hashes. This was an unconstrained macOS -ARM64 process on a 12-logical-CPU host, using Docker-hosted RustFS. It does not -qualify the requested one-vCPU/one-GiB profile, public application handles, -first activation, confirmation latency, fragmented roots or independent hosts. - -**Next design decision:** compare bounded scheduling of idle Cell executors -with prefetch that reduces demand faults, using the same-worker diagnostic and -public action traces. A scheduler needs available execution capacity; moving -idle executors alone cannot help a one-worker node. Qualify any additional -blocking workers under the one-vCPU memory/CPU envelope. Preserve exclusive -connection ownership and per-Cell order; do not attempt to move an executing -connection or turn a VFS error into a resumable SQL result. SQLite's -[VFS method contract](https://www.sqlite.org/c3ref/io_methods.html) returns an -I/O result synchronously, and Crab's `paged_io::receive` waits for that result. - -**One-vCPU follow-up, 2026-09-27:** the -[Compose worker profile](../qualification/worker-profile.md) runs the same -diagnostic with one and two SQL workers. Both release processes passed against -pinned RustFS 1.0 GA on ARM64 Colima. Kernel counters and Docker inspection -confirm one vCPU (`100000/100000` quota), 1 GiB memory and zero swap for each -process. Rust detected one available CPU. The builder exited before measurement. - -| SQL workers | Same-worker query median, no added delay / +20 ms per GET | Comparison Cell median, no delay / +20 ms per GET | Largest same-worker query, +20 ms per GET | -| --- | ---: | ---: | ---: | -| 1 | 20.773 / 423.385 ms | 20.828 / 423.452 ms (same worker) | 510.334 ms | -| 2 | 23.887 / 427.773 ms | 0.077 / 0.070 ms (other worker) | 445.042 ms | - -These are three samples per condition in each process, not p99. Resident SQL -callbacks took 6–43 microseconds across demand and hydration observations; -the demand interference remains before callback entry. During the separate -500 ms/GET hydration phase, same-worker queries took 0.060 and 0.062 ms while -hydration plus queries took 1.015 and 1.011 seconds. All twelve cold reads -performed 16 observed origin reads and transferred 4,008,092 bytes. Payload -digests matched; repeated payload reads made zero new origin requests. - -Peak cgroup memory was 91.4 / 89.9 MiB for one / two workers; both had zero -CPU throttling and zero OOM events. Charged cgroup memory includes cache and -is not process RSS. Three Cells and this I/O-bound workload cannot establish -per-worker memory cost, sustained CPU capacity or a production worker default. -The shared-worker stall persists even when CPU quota is not exhausted; an -available second worker can serve its resident Cell during the provider wait. -This supports testing bounded extra blocking capacity and idle-Cell scheduling -alongside prefetch, while preserving one owner per SQLite connection. - -Source: production parent `f33f7b94387b` plus the worker diagnostic wrappers; -the separately staged app-to-host relocation was excluded from the build input. -The pinned compiler image reports Rust 1.97.1. Evidence is retained in -`worker-profile-20260927/evidence/` under the checkout's external target: -`source.json`, the adjacent source tree, `single.json`, `paired.json`, full logs, -binary SHA-256, container/image inspection, kernel counters and `summary.json`. -The initial unshared-mount build failure is retained; it ran no tests. Each -process creates a different random root, so this is not a byte-identical A/B -comparison. Public actions, first activation, durable confirmations, sustained -load and independent failure domains remain unqualified by this diagnostic. - -### 10. Published-cut cleanup occupies the SQL worker after durability - -**Confirmed at audited revision:** [CellExecutor::confirm_published](../src/cell/executor.rs) calls -[Db::prune_captured](../../crab-ltx/src/db.rs) on the SQL worker. Its -`prune_retained` implementation reads the entire selected file into a `Vec`, -hashes and decodes it through -[verify_segment](../../crab-ltx/src/recovery.rs), then deletes it and reconciles -disk accounting. The decoder avoids retaining every decoded page, but the -compressed input remains fully buffered. Default library limits permit a -512 MiB LTX file; these limits explicitly are not an RSS quota. - -On the object-only actor path, the publication proof channel is completed -after this cleanup. With node-log durability, an external proof can arrive -earlier, but the final `confirm_durable` still waits for its SQL worker. Thus -cleanup can delay a response or later work after external durability exists. -This is separate from the capture-time reinspection already removed and -[measured](../../crab-ltx/perf/README.md#large-sparse-checkpoint-capture-2026-09-25). -The same cleanup implementation exists on the main snapshot. - -**Change to evaluate:** first replace full-buffer verification with admitted, -streaming verification using the existing decoder contract. Then evaluate -moving exact-file cleanup out of SQL execution. Keep retained disk accounting -and owned file identity until deletion finishes; bound cleanup debt. Separating -durable sequence advancement from file reclamation requires explicit handling -of cleanup failure, shutdown, duplicate confirmation, and canceled waiters. -Do not simply remove verification or release reservations when dispatch starts. - -**Gate:** small cuts, large incompressible cuts, and full-image cuts under the -1 GiB profile. Record cleanup bytes, temporary RSS, worker occupancy, proof-to- -response delay, and sibling-Cell p99. Preserve -`captured_pruning_retains_accounting_after_io_failure`, -`published_deferred_capture_is_pruned_without_a_local_durability_barrier`, and -`published_root_survives_local_prune_failure_without_replaying_sql`. These -already distinguish a published root from failed local reclamation; they do -not bound reclamation latency or memory. - -**Implementation:** cleanup now opens the selected file once and verifies it -through a 64 KiB buffered reader. The canonical stream verifier checks header -limits, full page/index/trailer structure, exact length, metadata, and BLAKE3 -before deletion. Bundle and node-frame callers retain their byte-buffer -length/digest rejection before the same structural verifier. Commands, -migrations, and bootstrap all use this cleanup path; no authority or sequence -transition moved. A read or unlink failure still retains unfinished accounting. - -The new public-`Db` regression fails on the old whole-file transfer and passes -with streaming; truncation, extension, corruption, and substitution with another -valid cut preserve files and reservations until a successful retry. A local -[RustFS comparison](../../crab-ltx/perf/README.md#streaming-published-cut-cleanup-2026-09-26) -observed 194.7 to 153.7 ms median cleanup for the largest 50.8 MB batch across -three processes, with an unchanged pooled 0.161 ms small-cut median. This is -microbenchmark evidence. Cleanup still blocks the SQL worker and the decoder -still allocates indexes; finding 13 prevents interpreting this as constant -memory or completed foreground-latency qualification. - -### 11. Long-run checkpoint tails need their own qualification - -**Confirmed:** [checkpoint_if_needed](../../crab-ltx/src/capture/checkpoint.rs) -runs inside capture. Passive work is triggered by appended frames or elapsed -time; emergency truncation uses the original logical WAL size and a threshold -at least as large as the database for larger databases. A truncate restart -captures a full boundary image. The relative threshold avoids repeated -database-sized captures on every large insert, but the eventual boundary -capture still runs synchronously before that command's cuts are returned. - -**Change to evaluate:** attribute WAL bytes, checkpoint mode/restart, full-image -cut bytes, and their downstream upload/cleanup cost to the triggering action. -Explore scheduling safe checkpoint work during available worker time only after -measuring the remaining tails. Keep the writer barrier, exact sealed boundary, -and WAL restart detection. SQLite's -[checkpoint contract](https://www.sqlite.org/wal.html#performance_considerations) -explains why checkpoints involve extra I/O; Crab's managed checkpoint policy, -rather than SQLite's default autocheckpoint threshold, controls this path. - -**Gate:** repeated updates of a fixed-size database, inserts, deletes/truncation, -and concurrent Cells over multiple checkpoint cycles. Include p99/max and the -largest retained cut, not just steady small-cut medians. Existing sparse -checkpoint measurements use a different harness and short runs; they cannot -establish public action tails or memory headroom during a full-image cut. -This path is unchanged from the main snapshot. - -Before moving checkpoints onto a concurrent connection, check the SQLite -dependency too. The workspace enables `rusqlite`'s bundled feature; the locked -`libsqlite3-sys` 0.32.0 source contains SQLite 3.49.1. SQLite's official -[WAL-reset advisory](https://www.sqlite.org/wal.html) identifies a -write/checkpoint race in that version range and names fixed releases. The -current exclusive worker and managed checkpoint barrier serialize this path; -this audit has not reproduced that upstream race in Crab. Any proposal to -overlap those operations must qualify a fixed SQLite build first, then prove -the capture barrier and restart handling. Record the linked SQLite version in -the evidence; a Rust crate version alone does not identify it. - -The [fixed-working-set runner](../../crab-ltx/perf/README.md#fixed-working-set-updatedelete-churn-2026-09-26) -now supplies executable update/delete/reinsert and exact provider-restore -proof. Six local RustFS release runs completed 300 mutations over 32 rows with -32 KiB payloads, each crossing two checkpoints and restoring the expected -final state. Separate cases end after deletion, retain zero live rows, and -force full-image capture through an 8 KiB incremental bound. Raw per-command -mutation and checkpoint counts accompany phase timings. This closes the -append-only fixture gap for local LTX diagnostics. Pinned readers, scheduled -compaction, concurrent Cells and sustained public-action tails remain open; -these short unconstrained runs do not qualify that broader gate. - -### 12. Even ingress and Cell targeting do not prove even owner execution - -**Confirmed at audited revision:** [run_stage](../../crab-http-server/deploy/cell-issue-fleet/qualify.py) -checks live sessions, reads existing Cells, and records their owners. The -[load runner](../../crab-http-server/deploy/cell-issue-fleet/load.py) checks a -70–130% ingress split and uniform offered Cell targets. It does not enforce -owner balance or record the executing owner for each action. Its forwarded -count compares ingress with a pre-load owner observation, which can become -stale during movement. The older main runner also lacked this execution proof. - -A real RustFS functional rerun grew 20 fixed Cells from three to five nodes. -All five owned Cells, but their counts were **6, 5, 5, 1, 3**, not four each. -This observation demonstrates the measurement distinction; it does not prove -placement cannot converge. It used the earlier local server image -`4ecf6e3e6e83`, whose source revision is unavailable, and cannot qualify the -current runtime's placement or throughput. Raw owner maps and container data -are retained in `five-startup.json` beside the scheduled harness smoke report. - -**Change to evaluate:** record owner/epoch and execution counts over the -measurement interval. For uniform scale comparisons, require a documented -placement settling criterion and report workload-weighted execution imbalance; -retain skew as a separate intentional workload. Preserve fixed data size and -Cell count across stages. One Cell per node is too coarse for meaningful -load-balance behavior; test several Cells per node and a hot Cell separately. - -**Gate:** distinguish healthy containers, ingress distribution, Cell target -distribution, owner distribution, and actual execution distribution. Also record -the Docker VM's physical CPU/memory and host contention: twenty 1-vCPU limits -on an 8-vCPU VM are an oversubscribed topology, not twenty independent CPUs. -Require current-source images, repeated sustained runs, and isolated multi-host -failure domains before choosing supported limits. Run Entity, Shard, Workflow, -and read-model actions through public application handles as well as this issue -service; issue creation alone does not exercise those service compositions. - -At the audited revision there was also a provenance gap: both `qualify.py` and -`load.py` record the harness checkout's `git rev-parse HEAD` independently of -the image ID. `--skip-build` does not verify that the selected image was built -from that revision, and HEAD does not describe dirty build input. Require a -clean build or retained source-tree digest, and verify image provenance before -calling a comparison current-source evidence. Distinguish an OCI manifest -digest from its configuration digest when comparing Docker engines. - -**Implementation:** qualification now requires a clean checkout and builds -from its committed Git archive, so ignored files or edits during the build do -not alter the attributed source. Local and CI source builds set the same -revision label as release builds. The runner rejects a missing or mismatched -label before startup, pins every server service and release bootstrap to the -inspected image ID, retains a tag for that image, and checks each running node. -Schema 3 load reports separate server source/platform/image from harness source. -Labels remain producer metadata; imported images still need their CI source -and checksum receipts. - -The wrong-source regression fails on the previous runner and passes after the -change. Nine scheduler/provenance cases pass. A real Docker fixture builds from -a committed archive while its working file differs, extracts the committed -bytes from the image, moves the build tag, and verifies the retained image and -wrong-source refusal. All Compose profiles validate. The first Docker trial -exposed image-index disappearance after retagging; retaining the qualification -tag fixes that observed failure. This verifies evidence binding, not service -latency; current-source fleet curves remain open. - -**Attribution follow-up:** schema 4 now joins each acknowledged write to the -actual executing owner/session and receipt, replacing the stale pre-load -forwarding estimate. It does not enforce execution balance or attribute read -owners. The shared-process RustFS test and retained-log replay prove the join; -placement settling and cross-container measurements remain open. - -**Loaded scale-out gap at `e50055c48bb`:** the server -[rebalance adapter](../../crab-http-server/src/cells/router.rs) checks every -15 seconds and normally excludes a Cell used within the last 60 seconds. -The [actor](../src/cell/actor/lifecycle/scheduling.rs) refreshes last-used time -for both queries and commands. The [planner](../src/fleet/placement.rs) also -requires two stable observations and 60 seconds from the adapter's first -eligible observation, with at most two transfers per planning batch. Draining -has a separate eligibility path. These rules also exist on the compared main -snapshot; they are not regressions introduced by the recent LTX changes. - -At five uniformly scheduled pairs/s across 20 Cells, a Cell is targeted about -every four seconds. Continued traffic therefore prevents its ordinary idle -eligibility. A healthy new container and evenly distributed ingress cannot -establish that old owners shed this workload. This is a source-derived -explanation to test against the retained execution traces, not a causal -attribution of the historical p99. The current qualifier starts each load -after functional checks without a placement convergence gate. - -**Design correction:** qualify two explicit scenarios. For settled capacity, -stop application actions while ownership converges, observe signed capacity -and authority without invoking Cell handlers, and require stable ownership -and the declared weighted balance before timing. A fixed 60-second sleep is -insufficient: fresh eligibility evidence, bounded movement and current signed -observations all matter. For scale-out during arrivals, keep traffic running -and measure time to redistribute work. Supporting that case requires bounded -quiescence of a busy Cell through the ordinary drain/publication/release and -takeover gates. Preserve the separate hot-Cell case: moving a single writer -cannot parallelize its workload across nodes. - -Existing `fleet_rebalance_donates_ownership_surplus_without_headroom_gain` and -`fleet_rebalance_releases_settled_cell_and_restores_its_result` tests cover -settled transfer. Add continuous-arrival convergence and unaffected-Cell tail -latency to the public service gate; do not reduce production idle guards merely -to obtain an even benchmark chart. - -**Live follow-up:** [CI run 36251209972](https://github.com/crabbuild/crab/actions/runs/36251209972) -used the qualified ARM64 image at `a3638ef7e55` with that revision's clean stage -and load harness. All stages retained the same 20 Cells, scheduled five -create/read pairs per second for 60 seconds, and verified all 300 acknowledged -writes after the stage's post-drain owner loss. Each container was limited to -one vCPU and 1 GiB on one shared GitHub runner. - -| Nodes | Executing owners: acknowledged writes | Write p50 / p99, ms | Read p50 / p99, ms | Post-drain recovery, s | -| ---: | --- | ---: | ---: | ---: | -| 3 | node-01: 105; node-02: 90; node-03: 105 | 24.938 / 49.403 | 7.828 / 17.298 | 9.789 | -| 5 | node-02: 195; node-03: 105 | 31.823 / 65.027 | 12.464 / 24.086 | 8.835 | -| 10 | node-03: 105; node-05: 195 | 33.593 / 91.335 | 14.937 / 41.780 | 10.968 | -| 20 | node-03: 300 | 41.334 / 248.667 | 34.195 / 138.631 | 9.162 | - -The stages killed node-01, node-02, node-05, and node-03 in that order. The -next stage's owner maps show consolidation after those recoveries. No rebalance -completion was recorded in the retained final logs. Continued public checks -and load keep Cells active, and the runner has no idle-convergence phase. -This explains why this experiment cannot establish balanced scale-out; it does -not isolate the causes of the higher p99 from shared-host contention, observer -overhead, peer forwarding or provider work. All but one write used object proof; -peak in-flight pairs were 1, 1, 2, and 3. It was not a saturation experiment. - -**Retained action-phase audit:** the same run's joined write traces separate -the following durations. Each stage contains 300 acknowledgements. Values -below use nearest-rank percentiles of those individual observations; phases -overlap and their percentiles must not be added or subtracted. - -| Measured interval | 3 nodes: p50 / p99, ms | 20 nodes: p50 / p99, ms | -| --- | ---: | ---: | -| Client HTTP write | 24.938 / 49.403 | 41.334 / 248.667 | -| Typed command invocation | 16.723 / 34.959 | 23.578 / 127.108 | -| Worker admission and queue | 0.023 / 0.178 | 0.031 / 1.990 | -| Worker execution, including capture | 1.590 / 3.943 | 1.806 / 7.791 | -| LTX capture within execution | 0.401 / 2.539 | 0.474 / 4.038 | -| Proof task wait | 11.148 / 23.482 | 13.724 / 95.988 | -| Durable confirmation | 0.265 / 1.288 | 0.259 / 7.501 | -| HTTP boundary time outside typed invocation | 5.192 / 14.035 | 13.131 / 73.302 | - -The last row subtracts the nested invocation duration from response-ready -duration **for each request first**, on the same entry node. It includes -unattributed work before and after the invocation, not a measured routing or -authentication phase. Proof wait starts when its task is polled and is not a -complete object-store RPC timer. The worker queue includes admission before -dispatch; the capture interval is nested inside worker execution. - -The slowest 20-node write, submission -`5392fa77-b319-5b36-b0e4-7361d870b4cb`, took 431.292 ms at the client: -429.190 ms to HTTP response readiness, 209.458 ms in the typed invocation, -170.727 ms in proof wait, 4.134 ms executing on the worker, and 0.398 ms in -capture. That request spent 219.732 ms outside the typed invocation. Its -worker queue was 0.013 ms. This trace directs investigation toward the wider -HTTP path and durability wait; it does not support attributing this particular -tail to LTX encoding or the cold-read interference measured in finding 9. - -Retained inputs are `load-{3,5,10,20}-stage.traces/actions.jsonl` under the -external `ci-fleet-36251209972/` evidence directory. The derived -`derived-phase-audit.json` records input hashes, per-phase distributions and -the five slowest writes at each stage. These are observations from the -concentrated, lightly loaded `a3638ef7e55` fleet, not a current-source capacity -result. Add separate timings for authorization/catalog work, client preparation, -owner resolution, response enrichment and provider attempts before choosing -the next service-latency fix; repeat after actual ownership convergence. - -**Boundary-attribution implementation:** HTTP principal resolution and the -pre-mutation archive policy now report separate durations. Repository routing -reports its static action, result and elapsed time; client preparation binds -its duration to the exact Cell, incarnation and mutation request. Issue creation -reports its conditional label/assignee enrichment and response-value construction. -The collector requires these events for each acknowledgement and retains nested -routes separately: lifecycle routing belongs inside the archive-check interval, -not the command route. These changes observe the existing execution order and -add no authority cache or bypass. Provider attempt timing, query execution phases -and tracing-overhead qualification remain separate work. - -The 44 Python harness cases pass. A missing archive phase was accepted by the -previous collector and is now rejected; mismatched preparation identities or -operation IDs, missing phases and failed routes also fail attribution. The -HTTP/mTLS behavior test and isolated RustFS test pass through owner loss and -restored readback. The latter's default text logs and actual received-response -sample pass the updated join CLI. Runtime/server all-target Clippy also passes. - -One debug action at source `15b608452d9` plus the boundary patch measured -31.165 ms at the client, 30.489 ms to response readiness, 6.292 ms in the archive -check, 0.064 ms in the command route, 0.329 ms in preparation, 20.204 ms in -invocation and 0.022 ms in response enrichment. It used local-operator -authentication, which measured below one microsecond. The lifecycle route's -0.066 ms belongs inside the archive-check interval. This proves phase capture -on the real provider path; it is one shared-process diagnostic, not a comparison -with the earlier fleet percentile or an authenticated-user latency target. -`action-boundary-rustfs.log`, `.samples.jsonl`, `.actions.jsonl`, `.source.patch` -and `.source.json` in the external target retain the received acknowledgement, -logs, source and binary binding. The source patch includes the pre-existing -staged test relocation; that relocation remains excluded from these commits. - -The subsequent `6fc1bbc1ceb` fault driver refused to kill an owner because no -Cell remained on an unaffected owner. Its failure is retained in `fault.log`; -no fault report or during-arrivals recovery result was produced. Keep this -refusal. Add non-activating, bounded placement convergence before uniform -capacity measurement and retain an explicit failure report if it cannot -converge. Qualify scale-out under uninterrupted arrivals separately through -the public application boundary. - -**Placement preflight implementation:** the stage qualifier and fault driver -now share a status-only convergence gate. It requires serving roots, distinct -live sessions, Cell counts within each node's floor/ceiling capacity-weighted -share, and agreement with advertised active counts. Session, incarnation and -ownership epoch must stay stable for 30 seconds while every advertisement -generation advances. The gate retains incomplete views and retries for up to -ten minutes; failure stops before the workload or fault and retains samples. -It does not bypass the production idle, settlement or authority gates. Tests -reject concentrated owners, stale signed counts, stale generations and changing -epochs; transient transfer resets the window. A fault-preflight failure now -retains its report without constructing the destructive driver. Live convergence -at 20 nodes remains required; this is a prerequisite for settled capacity, not -an implementation of scale-out under arrivals. - -All 42 Python harness tests pass, including the retained preflight failure. -Runtime documentation validation passes 28 schema/protocol assertions and 194 -local links. The gate's ten-minute retry window is a qualification timeout, -not a supported placement SLO. - -### 13. A streaming decoder still retains avoidable metadata - -**Confirmed at audited revision:** [codec::Decoder](../../crab-ltx/src/codec.rs) accumulates the -decoded page/offset/size index and, whenever `replica` is enabled, a second -`EncodedPage` index with frame hashes and page checksums. The latter is built -even when cleanup or ordinary inspection never asks for it. At close, the -decoder reads the remaining stream into a new `Vec`, then allocates another -decoded index to compare with the first. Streaming the input alone does not -remove these allocations. A corrupt early end-of-pages marker can make the -buffered remainder much larger than a valid footer, up to the admitted input -length. This behavior also exists on the main snapshot. - -**Change to evaluate:** collect replica index entries only for callers that -need page lookup. Stream and compare footer entries against the observed page -index while checking the exact encoded length and CRC; reject excess bytes -without accumulating the whole remainder. Preserve both supported LTX page -encodings, complete page ordering/coverage, exact digest and trailer checks. -If the remaining observed-page index is material, evaluate admitted file-backed -index scratch separately, including cleanup and cancellation ownership. - -**Gate:** measure peak allocations for the same decoded database represented -as compressible and high-entropy cuts, multiple page sizes, and concurrent -verifications under the 1 GiB node profile. Include long invalid tails, early -end markers, truncated varints, and valid CRCs around invalid index entries. -Run the independent Celld/Superfly vectors, bundle and node-frame verification, -exact restore, and sparse publication. A bounded individual `read` does not -prove bounded accumulated decoder memory. - -**Implementation:** ordinary verification now collects only the observed -page/offset/size index. Explicit replica inspection collects the frame hashes -and checksums needed for page lookup and moves that index to its caller instead -of cloning it. Footer entries are compared directly with the observed index -through a 64 KiB buffer; neither the complete footer nor a second decoded copy -is retained. CRC and encoded-length checks use the original varint bytes, -including previously accepted nonminimal encodings. Actual I/O failures retain -their source and classification; clean EOF within the footer is corruption. - -The long-tail regression fails before this change and passes afterward for -both an early page-end marker and bytes appended after a complete valid file. -Each malformed case reads less than 128 KiB rather than draining its 8 MiB -tail. Independent format vectors retain exact digest, CRC, ordering, coverage, -and restore checks. This bounds excess footer buffering, not total decoder -memory: the observed index still grows with page count, explicit replica -inspection needs additional metadata, and the caller may retain input bytes. -Concurrent peak RSS and same-worker response latency remain open gates. - -The [exploratory RustFS comparison](../../crab-ltx/perf/README.md#streaming-footer-and-optional-replica-index-2026-09-26) -observed a 140.5 to 101.9 ms median cleanup for the largest batch across three -processes per implementation. Baseline compilation and unrelated host activity -limit attribution; overlapping whole-process RSS ranges do not prove a memory -improvement. This evidence does not qualify public-action latency or capacity. - -### 14. Checksum bookkeeping depends on activation history and uses tiny file I/O - -**Confirmed:** fresh `Db::open` and `CellReplica::open_new` start with -`PageChecksums::default()` in [capture initialization](../../crab-ltx/src/capture.rs). -Sparse activation and clean resume seed a file-backed base instead. -[PageChecksums::persist](../../crab-ltx/src/pages.rs) is called after each sealed -cut in [WAL capture](../../crab-ltx/src/capture/wal.rs), before capture returns. -The representations have materially different costs: - -| Path | Work at audited revision | Missing qualification | -| --- | --- | --- | -| Fresh Cell, memory base | Allocate a dense array for all database pages and copy the retained base on every cut, even for a small update | Database-size scaling of capture CPU and peak memory with changed pages held fixed | -| Restored/resumed Cell, file base | Read each overwritten old checksum separately; persist each changed checksum with an 8-byte write in hash-map iteration order | Host calls, allocation count, local I/O time, and fragmented versus contiguous changes | -| Clean handoff, file base | `write_dense` buffers output, but obtains each input checksum through a separate 8-byte read | Drain/eviction duration and other Cells waiting on the same worker | - -For a 512 MiB database with 4 KiB pages, the memory base is 1 MiB and a dense -file-backed handoff reads 131,072 checksum entries individually. These are -source-derived counts, not measured latency. The default filesystem implements -each positional read with a seek, a new buffer, and a read. The -[resume writer](../../crab-ltx/src/resume.rs) reaches this loop from -`CellExecutor::close_resumable` on the SQL worker. Both representations and -the handoff loop also exist in the compared main snapshot. Existing sparse -capture results therefore cannot establish the cost of a long-lived fresh Cell. - -**Change to evaluate:** batch dense sidecar reads as well as writes; use a -bounded checksum-block cache for ordered capture reads and coalesce adjacent -changed checksums before persistence. Measure a shared block-based strategy -for fresh and restored sessions after those changes. Keep transactional -candidate isolation: the old checksum remains available until the cut is -sealed, and partial sidecar failure fences the session. Local scratch is never -recovery authority. The minimal-feature library still needs a tested local path. - -**Gate:** fixed-size updates, append, truncate/regrow, sparse activation, clean -resume, and repeated eviction at multiple page sizes. Use `capture_deferred` -in both runtime comparisons: the current replica-cost harness changes capture -durability along with `--sparse`, so its two modes do not isolate this finding. -Record checksum host -reads/writes and bytes separately from SQLite WAL I/O, capture allocation peak, -handoff time, and sibling-Cell p99. Preserve the file-backed overlay tests, -`cell_checksum_write_failure_fences_after_sealing_the_cut`, process-exit -continuation recovery, and same-length local-corruption refusal. Evaluate the -verified whole-file scan in `open_resumed` separately: zero origin reads do not -make warm reactivation constant time, and skipping that scan needs a replacement -integrity proof. - -**Implementation:** file-backed persistence now sorts changed pages and writes -contiguous checksums in blocks capped at 64 KiB. Dense handoff reads also use a -64 KiB buffer, consuming base entries even when an overlay replaces them so -later checksums retain their correct offsets. Aggregate verification, post-seal -failure fencing, and the fresh durable handoff sidecar remain unchanged. -Sorting retains one reference pair per changed page; the bounded output buffer -does not make total capture memory constant. - -The 32 MiB real-SQLite regression fixture first reproduced 8,210 host reads -for its 65,680-byte handoff sidecar. The same fixture now requires at most two -reads and reproduces identical sidecar bytes. Updating that resumed database -first reproduced 8,757 host writes for its roughly 34 MB cut; the changed path -uses fewer than 1,024, including LTX writes, each at most 64 KiB. It re-reads and -folds the persisted checksum blocks before clean handoff. These operation-count -tests use local files; their unused replica transport is in memory. They do not -measure RustFS latency or isolate checksum writes from all capture writes. - -**Local merge and read-window follow-up:** the capture owner now adopts the -isolated candidate only after the LTX file is sealed, retiring its reference to -the old base before merging. Fixed-size updates mutate the uniquely owned -`Vec` instead of allocating and copying the whole array. Retained immutable -memory snapshots still use Rust's -[`Arc::make_mut` copy-on-write contract](https://doc.rust-lang.org/std/sync/struct.Arc.html#method.make_mut). -Growth may reallocate. Truncation releases capacity when it exceeds twice the -live entry count; memory residency still scales with database size. The old -owner remains unchanged on errors before sealing, and errors while merging a -file-backed sidecar still fence `Db` before another capture. - -Restored capture retains one 4 KiB checksum window for each ordered page -application. A new cut starts with an empty window so a prior merge cannot -leave stale checksums. The read ends at the base's page count. Sparse edits -can read up to one block per touched checksum, so this trades extra local -bytes for fewer calls; it adds no remote reads or cross-cut cache. Truncation -and clean handoff keep their existing 64 KiB sequential scans. - -Both performance regressions failed on the preceding implementation. A -one-page update reused the original base allocation at 1,024 and 32,768 pages -after the change. The 32 MiB resumed SQLite fixture produced two cuts, including -checkpoint maintenance: checksum reads fell from **8,208 to 18**, with bytes -changing from **65,664 to 69,776** and a maximum 4 KiB transfer. A point edit can -touch separate checksum blocks for SQLite metadata and its data page; its -bound is based on independently decoded changed-page counts. Repeated point -edits and dense handoff exercise invalidation between cuts. Memory snapshot -isolation and truncate/regrow pass at 512/4096-byte page sizes with and without -`replica`. These are allocation-reuse and I/O-count proofs, not latency SLOs. - -Verification also covers the independent full-database CRC oracle, same-length -database/sidecar corruption refusal, post-seal sidecar failure fencing, exact -sparse publication, truncate/regrow directory selection and process-exit clean -continuation. The public HTTP/mTLS collaboration test passes against real -RustFS through owner takeover and Git readback. Replica and minimal-feature -all-target Clippy, formatting and documentation checks are required for this -change; full-suite CI results must match its source revision. - -Public-action percentiles, allocation peak, fragmented-update tradeoffs and -eviction interference still need qualification. Eager authenticated activation -metadata and work on the shared SQL worker remain separate open gates. - -### 15. Peer verification couples provider latency to scarce CPU admission - -**Confirmed at `3cd0bd1bfe6`:** the HTTP -[forward handler](../../crab-http-server/src/peer.rs) reserves a primitive-job -slot before awaiting `NodeDirectory::verify_peer_request`. That method loads -the current signed node advertisement from storage. One slow enrollment read -therefore holds the one-vCPU fixture's only primitive slot, and unrelated -requests receive 503. The same coupling exists in the compared main snapshot. -The node-log handlers load enrollment before acquiring their codec slot; -their larger frame and follower-durability contracts need separate proof. - -**Follow-up fix and proof:** the public HTTP/mTLS regression sends five -concurrent comments queries after owner-hint expiry. It reproduced 503s with -both memory storage and RustFS; a temporary probe identified codec admission -refusal. The receiver now decodes the bounded envelope once, releases codec -admission during enrollment I/O, and rechecks the signed enrollment's lifetime -before verifying the request. Request bytes remain reserved while waiting. -The shared ingress hint removes the observed entry catalog/control reads on -the warm path; receiving-owner authority checks remain in place. - -Queued codec admission uses the same resource limit as immediate admission. -It explicitly checks the absolute deadline before waiting and after waking: -the pinned Tokio 1.53.1 `Timeout::poll` polls a ready inner future first, so -`timeout_at` alone admitted an already-expired request in the regression. -Nine worker tests now pass, including expired/free-capacity admission, expiration -when a slot opens, cancellation, timeout, shutdown, and resource accounting. -The node-directory test separately rejects an expired enrollment even when its -request signature remains valid. Twelve peer protocol tests preserve strict -payload, unknown-field, duplicate-field, and signature validation. - -The HTTP receiver bounds enrollment, resolution, activation, dispatch waiting, -and response encoding by the received transport budget. Initial structural -admission uses the protocol's 60-second ceiling until the envelope's timeout -is available; elapsed admission time still counts against that timeout. The -budget starts after body ingress, and is distinct from the actor's five-second -native-work deadline and the sending client's transport wait. Cancellation of -an HTTP wait does not roll back accepted commands or change unknown-result -resolution. - -The TLS regression delays enrollment GETs by one second, proves another codec -job can run during that delay, and checks a 10 ms received budget returns 504. -The public HTTP regression also delays only owner resolution by two seconds -with a 500 ms received budget: it returns 504 without invoking the query -handler, then the same signed query succeeds when the delay is removed. -The full public application regression passes against memory storage and real -RustFS: concurrent hint expiry, owner takeover, restored collaboration state, -and Git clone/tag reads. These are functional proofs, not measured p95 gains. - -The same public HTTP/mTLS fixture now separately delays the activation router -after a clean Cell drain, while authentication and owner resolution use their -normal stores. Activation reads receive a two-second injected delay against a -500 ms received budget; the read counter proves that path was entered. The -response is 504, no query handler runs, primitive -job reservations return to zero, and the complete idle control value is -unchanged. Removing the delay lets a fresh activation-enabled request execute -exactly one query; subsequent public reads verify the persisted comment. Both -the in-memory and real local RustFS application tests pass, including the later -owner takeover and Git readback. The original same-envelope resolution retry -case is retained. This closes early activation deadline coverage without -changing production code; it does not prove cancellation midway through an -already-dispatched restore or SQLite operation. - -**Remaining gate:** exercise accepted-command completion under HTTP cancellation; -retain the runtime stable-identity cancellation regression. -Measure admission wait, enrollment I/O, codec time, retries, 503s, and retained -request bytes under the one-vCPU profile. Compare public p95/p99 at the same -offered load before accepting the hint and queue changes as a performance win. - -### 16. Post-load recovery does not qualify acknowledged tails during load - -**Confirmed at `c12b41ef638`:** the scheduled -[load runner](../../crab-http-server/deploy/cell-issue-fleet/load.py) waits for -`drain_publication` before `recover_owner`, which again requires zero uncovered -node-log bytes. It then kills one owner and verifies the latest acknowledged -issue for one selected Cell. Every successful pair has an immediate readback, -but the post-fault check does not revisit all successful request IDs. The -existing README correctly labels this as published-root recovery. The separate -[Compose cluster gate](../../crab-http-server/tests/qualify_compose_cluster.sh) -exercises follower recovery, but it is not the scheduled capacity workload. - -**Acknowledgement follow-up:** the load runner now revisits every recorded -acknowledgement before the fault and again after takeover, while the lost node -is still stopped. It verifies exact issue number and title, limits verification -to eight concurrent reads, and records counts by Cell. A real HTTP fixture -keeps the latest issue intact while deleting or changing an earlier result: -the previous recovery function silently accepts both cases, and the follow-up -rejects each with the original request ID. All ten load/provenance tests pass. -This closes the latest-only verification gap; a current-image fleet run and -failure during sustained arrivals remain required. - -**Impact:** the current runner cannot establish that low response latency -remains sustainable while publication is delayed, or that every earlier -follower-only acknowledgement survives failure during that backlog. This is -a missing proof, not evidence of lost data. Earlier revisions of the runner -have the same post-drain fault shape. - -**Change to evaluate:** add a fault phase while scheduled arrivals continue. -Trigger from observed fleet response proof and a nonzero, position-attributed -unpublished tail, then lose the owner's process and local data. Keep selected -followers available for that case. Run separate follower-loss, delayed-origin, -and ambiguous-publication cases with their declared fault budgets; a combined -fault beyond the durability contract cannot be labeled a supported scenario. - -**Missing attribution:** the HTTP issue handler's `command_output` discards -the typed commit receipt. Its submission UUID is a durable application key; -`mutation_identity` creates a different runtime request ID for each attempt. -The runtime response trace includes Cell, sequence, and proof source, while -`cells status` exposes the published root position. Aggregate response and -uncovered-byte counters cannot join those positions to one HTTP submission. -Before using them to trigger a fault, retain a structured submission/receipt -correlation at the application boundary and match it to the owner proof trace. -The trace alone is not an HTTP acknowledgement: the load generator must also -have received and recorded that request's successful response. - -**Gate:** retain every acknowledged request ID and expected result, then query -or resolve all of them through public handles after takeover. Count duplicate -effects, unrecoverable results, interrupted arrivals, and recovery delay. -Measure healthy Cells' p99 throughout the fault. The run must show that -publication catches up after origin recovers without discarding accepted work. -Repeat for uniform and hot/skewed workloads at 3, 5, 10, and 20 nodes, including -compatible rollout. Record executing owners during load; a pre-load owner map -cannot identify execution after migration. - -### 17. Compaction reads index records individually on the async task - -**Confirmed at `9ec6da5176e`:** -[SpoolCursor::advance](../../crab-ltx/src/replica/compaction/source.rs) -calls `FileIo::read_exact_at` for each 60-byte index entry. The default -[filesystem implementation](../../crab-ltx/src/environment/host.rs) performs -a seek, allocation, and read for each call. Remote index downloads already use -bounded chunks; their subsequent local merge does not retain a read buffer. - -Both consumers run this iteration outside `Host::run`: -[write_compacted](../../crab-ltx/src/replica/compaction/output.rs) obtains the -next entries before dispatching body decoding and encoding, and -[build_and_upload](../../crab-ltx/src/replica/directory/initial.rs) consumes -the final locator merge while constructing directory nodes. Full compaction -therefore reads the selected index entries and then the new compacted index -again. Truncation can make a merge scan many discarded entries before yielding -one live page. Buffering only the output writer does not address these reads. - -**Live diagnostic:** a temporary instrumented run of -`cell_compaction_coalesces_local_output_writes` used the existing 3,000,000-byte -`randomblob` SQLite fixture on a current-thread Tokio runtime. Local RustFS at -port 19010 backed immutable preparation, full compaction, and two restores. -It recorded **1,474 60-byte reads, all on the async thread**, out of 1,480 -local reads during compaction. The original and compacted roots restored -byte-identical databases. An in-memory-provider control recorded the same -counts. Production code and existing test assertions were unchanged; the -temporary instrumentation was removed after both runs. - -The diagnostic patch and logs are retained outside the checkout under -`$HOME/Workspace/crabbuild-target/crab-8bc8/ltx-index-audit/`: -`rustfs-probe.patch`, `rustfs-probe.log`, `probe.patch`, and `probe.log`. -These are debug operation counts, not latency percentiles, a slow-disk fault -test, or a fleet capacity result. The same per-entry read and async iterator -consumption exist in the compared main snapshot. - -**Impact:** source tracing shows that slow scratch reads can hold the Tokio -task between await points, including while the publisher's renewal select is -waiting to regain control. Tokio's -[fairness guarantee](https://docs.rs/tokio/1.53.1/tokio/runtime/index.html#detailed-runtime-behavior) -requires bounded task polling time. The provider concurrency improvements do -not isolate this local work. The existing compaction test bounds output writes; -it does not bound input reads or verify unrelated async task progress. - -**Change to evaluate:** retain sequential buffers for index cursors and consume -bounded batches through `Host::run` in both merge passes. Budget the sum of all -cursor buffers, not just one buffer. Bound discarded-entry work as well as live -output pages; preserve merge state across batches. Let the dispatched job own -its file handles, reservations, and scratch until completion. The initial -append's in-memory index iterator shares locator selection but does not need -a file-I/O adapter. Keep newest-wins selection and truncate/regrow handling -canonical in `LocatorMerge`. - -**Gate:** count reads by index stream and assert block-scale input I/O; inject -slow and failed scratch reads while checking unrelated Tokio progress and -renewal scheduling. Cover partial-range and full compaction, bundles with -nonzero body offsets, later overwrites, truncate/regrow, cancellation, and -byte-identical restore. Measure retained buffer bytes, scratch peak, foreground -p99, and publication drain through repeated compaction boundaries. This is a -bounded first change before finding 4's larger directory-reuse work. - -**Implementation:** both file-backed merge passes now read sequential index -blocks through admitted blocking jobs. All cursors together retain at most -960 KiB of read buffers, with a 60 KiB maximum per cursor. A merge job stops -after 4,096 input entries at the next page-group boundary; a group visits at -most the admitted descriptor count. Discarded entries count toward the budget, -so a long truncated suffix cannot become one unbounded merge job. The shared -newest-wins/truncation resolver remains canonical for in-memory initial -directories and file-backed compaction. - -Scratch creation, file opens, reads, writes, syncs, and removal now use host -jobs. Every open scratch file and upload source retains the scratch owner. -Cancellation cannot remove files or release their dirty/recovery/scratch -admission while dispatched work still uses them. Normal completion waits for -cleanup; canceled work schedules cleanup after its last owner drops, using the -same blocking-job ceiling. Cleanup remains best effort on filesystem/executor -failure and requires the Tokio runtime to remain alive. - -The existing 3 MB fixture first failed the new read bound with 1,480 local -reads; the async-thread refusal regression also failed before the change. -The updated RustFS diagnostic records **8 local reads** and the same 792 -writes, with byte-identical restores of original and compacted roots. This -reduces local read calls by 185 times in that fixture; it does not establish -a latency improvement. The diagnostic delta and log are -`rustfs-after-test.patch` and `rustfs-after.log` in the artifact directory above. - -Seven focused compaction tests pass, including six canceled filesystem stages -with paused cleanup and one job slot. All 26 exact-root/sparse cases pass; -the truncate/regrow fixture now spans multiple merge jobs and verifies both -partial and full compaction. Replica all-target Clippy passes with warnings -denied. Whole-graph metadata work, the conservative scratch admission floor, -and sustained foreground/publication measurements remain open. - -### 18. Cold directory-cache fills retain origin admission and delay readers - -**Confirmed at `7d3dd9232b0`:** -[read_node](../../crab-ltx/src/replica/directory.rs) takes `Host::io_permit`, -downloads and authenticates a directory object, then awaits -`directory_cache_put` before inserting the bytes into memory and returning. -The permit remains in scope during that await. -[Host::directory_cache_put](../../crab-ltx/src/environment/host.rs) dispatches -blocking cache work under a separate job permit. -[DirectoryCache::put](../../crab-ltx/src/environment/directory_cache.rs) -syncs the new file and renames it, then clones, serializes, syncs and renames -the membership index. The filesystem contract also requires durable parent -installation. A slow cache device can therefore delay an already verified -read and occupy origin capacity needed by unrelated requests. - -This affects cold/missed directory reads used by activation, sparse faults -and publication metadata. Memory hits and verified disk-cache hits bypass the -fill; the earlier removal of hit index writes does not close this path. -The same fill ordering exists in the compared main snapshot. - -**Diagnostic:** a temporary public-API host-hook probe captured a real SQLite -row and prepared an immutable root, then reopened it through a distinct store -identity with a cold directory cache and one I/O permit. Pausing cache -`sync_all` left the root read unfinished and the permit count at zero; another -permit waiter timed out after 100 ms. Releasing the filesystem pause returned -the root and restored the permit. The probe reproduced with an in-memory -control and with local RustFS at port 19010. The RustFS root then restored -into a fresh SQLite file and `SELECT count(*) FROM t` returned the captured -row. The 100 ms wait is an injected diagnostic bound, not a service percentile. - -The production source and existing assertions were unchanged. Temporary -instrumentation was removed after the run. Patches and logs are retained under -`$HOME/Workspace/crabbuild-target/crab-8bc8/ltx-design-audit/` as -`cache-admission-memory.patch`, `cache-admission.log`, -`cache-admission-rustfs.patch`, and `cache-admission-rustfs.log`. -The eight existing `cell::roots::directory` tests also pass on the restored -source, including corruption, restart, incremental update and truncate/regrow. - -**Best next fix to evaluate:** first end origin admission once transfer and -bounded authentication finish. This isolates network capacity while retaining -the current cache contract. Then evaluate returning authenticated bytes before -optional cache installation through a bounded, deduplicated fill queue owned -by the host. Account queued bytes, dispatched jobs and disk reservations; -shutdown must drain or cancel undispatched work and await dispatched work. -Do not replace the await with unlimited detached tasks. Tokio's -[blocking-task contract](https://docs.rs/tokio/1.53.1/tokio/task/fn.spawn_blocking.html) -requires dispatched work to retain its resources until completion. - -**Gate:** preserve digest, symlink, restart, eviction and disk-budget proof. -Pause/fail cache writes and index syncs while a second Cell reads or publishes; -the second origin request must progress after the first transfer completes. -Exercise canceled readers and concurrent fills for one key. Measure first -read/first mutation, cache fill queue age, blocking-job wait and foreground -p99 with empty and churned caches under the 1-vCPU/1-GiB profile. Network -permit isolation alone does not establish foreground latency isolation from -the shared blocking executor or cache-index rewrite cost. - -**Implementation follow-up:** origin admission now ends after the bounded -download and digest check. Cache fills take immediate blocking-job admission; -when the pool is busy they skip optional persistence instead of retaining -verified buffers in an unbounded wait queue. Admitted fills share the same -dispatch/ledger/cancellation owner as other host jobs. Cache-hit reads, -invalidation, digest checks, and persistent installation are unchanged. Skips -may increase origin reads after a restart; cache contents remain derived data. - -Both new regressions failed before the change. With a cache fsync paused, a -separate cold origin read now completes using the same one-permit I/O pool; -aborting the first reader retains its blocking slot until the fsync finishes. -The second root restores into a fresh SQLite file with the captured row -visible. The same test passes against real RustFS and is retained as an -explicitly invoked host test, documented in the -[LTX verification runbook](../../crab-ltx/README.md#verification). A private -admission regression proves that busy job slots do not queue fill buffers and -that a subsequent admitted fill persists normally. - -**Bounded asynchronous fill implementation:** verified directory bytes now -return after immediate job admission and dispatch. A host-family tracker -deduplicates in-flight keys; no extra task queue or thread pool was added. -The existing job ceiling and directory-node size cap bound retained buffers. -Each dispatched closure retains blocking-job and disk accounting through its -completion or drop. Derived fills do not retain the reader's recovery/dirty -cohort. Busy disk-cache lookups and per-key fill locks yield a cache miss, -allowing authenticated origin reads to proceed. Failed optional fills cannot -poison subsequent cache access; every returned entry retains shape and digest -validation. - -Cache construction excludes accepted fills before cleaning private temporaries. -`Host::drain_cache_fills` waits for accepted cache work; runtime shutdown calls -it after stopping admission and draining Cells, workers and durability work. -Canceling a drain does not release running jobs; callers can await it again. -This follows the executor's accepted-job ownership contract and Tokio's -[blocking-task lifetime](https://docs.rs/tokio/1.53.1/tokio/task/fn.spawn_blocking.html). -There is no separate permanent cache shutdown state, so an explicitly drained -host can be reused by another runtime. - -The strengthened public fault test first failed because root open waited for -cache fsync. It now returns while the sync remains paused, as do fresh-store -lookups for the same and another Cell sharing that cache. The one/two-job-slot -matrix passes over memory storage and real RustFS, with exact SQLite restore. -Draining and reopening remain pending until the pause is released. Private -tests cover bounded dispatch, duplicate fills, dropped/rejected jobs, and -filesystem failures/panics followed by a successful fill. Existing restart, -corruption, symlink and cancellation tests remain required. - -The restart audit also reproduced an unaccounted cache file: when installed -bytes outlived their membership entry, a later same-length fill reused them -without reserving disk. Reuse now requires indexed membership; otherwise the -normal replace/reserve path applies. Its regression changes disk usage from -zero to the exact eight fixture bytes. An unchanged indexed fill updates only -in-memory recency, sharing the existing cache-hit persistence rule. - -Verification passed: 16 host-environment unit tests, eight authenticated -directory/restart tests, the real RustFS paused-fill matrix, and the public -HTTP/mTLS RustFS collaboration test with owner takeover and Git readback. -The cache-open lifecycle suite and runtime activation/shutdown regressions also -pass. Replica/runtime and minimal-feature LTX all-target Clippy run with -warnings denied. These are correctness and concurrency checks; no controlled -service-latency comparison was performed. - -Fills and cache reads still share the blocking pool with required work, and -membership changes still rewrite its index. Skipping cache access can increase -origin traffic. Public first-read/first-mutation percentiles, cache-hit effects -and sibling-Cell interference under sustained load remain open gates. - -### 19. The initial asynchronous draft still blocked its own Cell and treated every error as stale - -**Confirmed in committed coordination and the hydration draft:** -[`BeginHydration`](../src/coordination.rs) sets `busy = true`, and `Schedule` -will not start queued work until `FinishHydration` clears it. The -[background task](../src/cell/actor/lifecycle/background.rs) holds this state -across preparation, asynchronous fetch and installation. Releasing the SQL -worker therefore helps sibling Cells, while a new request to the hydrating -Cell still waits for remote I/O, even when its required pages are resident. -An empty queue when hydration begins does not bound the latency of later -arrivals. This coordination behavior also exists in the compared main snapshot. - -The same task fences on every returned error, and -[`handle_hydrated`](../src/cell/actor/tasks/activation.rs) maps every error to -`stale: true`. The draft defers retained-byte admission pressure, but a remote -timeout before installation still follows the fencing path. The previous -synchronous VFS path could have partially installed pages before failing; -the new fetch stage has a stronger no-local-mutation boundary. Its error policy -has not yet taken advantage of that distinction. - -**Change to evaluate:** track in-flight hydration separately from exclusive -foreground work in the pure coordination state. Keep the effect alive for -drain, shutdown and ownership checks; reserve the worker only for preparation -and installation. Prefer foreground work between bounded installation batches. -Classify a retryable pre-install fetch failure as deferred maintenance only -while the owner remains valid and no local installation occurred. Retain -fencing for lease loss, integrity failure, uncertain partial installation and -stale activation. Simply clearing `busy` without changing completion handling -can let a hydration completion clear a concurrent command's state; it is not -a sufficient fix. - -**Gate:** use a public Cell handle to read a resident row and perform a mutation -while a different inherited range is stalled. Prove queue progress, the exact -captured successor root, and no stale overwrite after checkpoint/truncate. -Repeat for owner loss, drain, canceled fetch, transient transport failure, -corruption and installation failure. Existing -[`residency` tests](../tests/runtime/lifecycle/residency.rs) cover shutdown, -origin failure, disk exhaustion, lease loss and takeover; the worker-only -probe does not exercise this actor boundary. Preserve those failure guarantees -while adding the distinct safe-to-defer outcome. - -**Implementation:** hydration now owns a separate effect without retaining -the foreground `busy` slot. Completion updates only residency, preserving any -concurrent command's busy state. Pending effects still prevent early drain or -transfer. Preparation/fetch timeouts, retryable fetch failures and pre-install -capacity refusal return a deferred outcome; the actor waits at least one second -and honors longer provider delays. An installation timeout remains ambiguous -and fences the owner, as do permanent failures. The worker owns these phase -deadlines; the actor no longer wraps all phases in an indistinguishable timeout. - -The public-handle regression first failed its one-second foreground deadline. -It now queries and mutates a resident row while a different hydration GET is -paused, then restores the exact published root and reads the mutation. A -second public test exposes one transient provider failure with storage retries -disabled, proves the owner remains serving, and waits for hydration to resume. -Seven lifecycle tests retain shutdown, permanent-origin failure, disk exhaustion, -lease-loss and takeover coverage. Forty-six coordination/simulator cases pass, -including hydration completion while a command owns the foreground slot. -Four worker tests also cover fetch cancellation, deadline deferral, retained-byte -pressure and progress on both workers. The public HTTP/mTLS RustFS collaboration -test passes with the split enabled: a forwarded application mutation is -acknowledged, publishes LTX and remains visible through the application query. -This is end-to-end correctness evidence, not a service performance qualification. - -### 20. Sparse activation serializes unrelated Cells through a global registry lock - -**Confirmed at `3611a7895f6`:** [`Registration::new`](../../crab-ltx/src/writable_vfs.rs) locks -the process-wide `views()` map before creating and sizing the sparse file, -syncing it and its parent, constructing the paged I/O bridge, reserving local -disk, and allocating the presence map. It releases the lock only after -inserting the activation. `x_open` and registration teardown use the same map. -Consequently a slow local sync can delay another sparse activation or its -teardown on a different SQL worker. Already open SQLite reads do not acquire -this registry lock; the risk concerns activation and lifecycle concurrency. - -The entry path is `CellWritableDatabase::open_writable` from -[`ActivateRestored`](../src/cell/worker/run.rs); eager checksum preparation is -earlier and asynchronous. Moving checksum I/O to `Host::run` therefore does -not fix this separate critical section. The compared main snapshot has the -same lock scope. Its fleet latency contribution is unmeasured. - -**Change to evaluate:** keep path claiming and registration publication atomic -under short map operations; perform filesystem barriers, bridge startup and -allocations outside the global lock. Preserve exclusive fresh-destination -ownership and remove only the failed activation's own claim. Ensure teardown -drops resource owners after releasing the map lock, including a last bridge -reference whose destruction joins its I/O worker. Avoid one thread or one -registry per Cell. - -**Gate:** pause activation A's `sync_all` or `sync_parent`; activation B on a -distinct path and teardown C must finish independently. Attempt the same path -twice and require one winner; inject each setup failure and prove no stranded -claim or removal of another activation. Existing -[`activation` hook tests](../../crab-ltx/tests/host/hooks/activation.rs) test -checksum preparation and cancellation, but do not prove this global-lock -isolation. A recovery storm with constrained disk IOPS is the service test. - -**Implementation:** the registry now reserves a canonical path under a short -lock, represented by a private claim guard and an initially empty weak -reference. The guard alone can publish or remove that entry. File creation, -size, file/parent barriers, bridge startup, disk admission and presence-map -allocation all run after releasing the lock. A completed activation publishes -its weak discovery reference; `xOpen` upgrades it while holding the map lock, -then retains the same strong file ownership as before. The registry cannot -destroy an activation or join its bridge while locked. This uses Rust's -[weak-reference ownership contract](https://doc.rust-lang.org/std/sync/struct.Weak.html). - -A failed setup drops its claim. Existing or partially created files remain -quarantined under the exclusive-create contract; cleanup does not remove local -files or another activation's claim. Canonical path resolution and WAL-sidecar -refusal are unchanged, as are SQLite file callbacks and one-writer authority. -The change adds 14 net production lines and no public API or configuration. - -The first public-API regression reproduced both unrelated open and close -missing their one-second bound while a file sync was paused. The strengthened -test pauses file sync, parent sync and bridge startup separately, and checks -distinct selected-root values for all three Cells. See the -[RustFS run command](../../crab-ltx/README.md#verification) for the same scenario -with real objects. Failure cases cover file creation, sizing, both barriers, -bridge startup, capture-directory creation and a pre-existing destination. -Repeated conflicting opens must refuse promptly without releasing the first -caller's claim. Per-activation disk/bridge waits still occupy its assigned SQL -worker; this fix does not qualify aggregate cold-start latency or disk capacity. - -The seven activation tests pass, along with ten sparse LTX tests, eight -minimal-feature integration tests and all 24 runtime residency tests. -Both the new isolation test and the public HTTP/mTLS -application/takeover test pass against local RustFS. Replica all-target Clippy -passes with warnings denied. These checks cover registration, failure cleanup, -hydration and restored application visibility; they do not establish a fleet -latency percentile. Before/after and RustFS logs use the `activation-registry-` -prefix in this checkout's external target directory. - -### 21. The initial asynchronous draft lost demand-read cache reuse - -**Confirmed in the draft, absent as a separate path on main:** -[`Io::page`](../../crab-ltx/src/paged_io.rs) reads the shared view-keyed page -cache and fetches ahead on a miss. `Io::hydration_pages` instead calls the -database's `read_run` directly, without consulting or filling that cache. -The VFS presence map records installed pages, not all prefetched pages. -Hydration can consequently download and decode pages already fetched by an -earlier SQL fault. The two-versus-three operation diagnostic in finding 9 is -consistent with this path difference; it does not isolate bytes, cache hits, -or a general throughput regression. - -Both paths still authenticate through the same replica reader. That reader -also computes a window's spans and consumes its first span (finding 8). -Fragmented roots make the draft fetch several spans serially. Sixty-four -pages bound payload, not network round trips or wall time; enough slow spans -can consume the five-second hydration deadline and trigger finding 19. - -**Change to evaluate:** share authenticated range/cache access between demand -reads and background fetch without reacquiring a synchronous SQL-worker wait. -Use exact view/root identity, bounded cache ownership and single-flight work -where useful. Evaluate bounded multi-span fetching for bulk hydration and -smaller/adaptive demand read-ahead separately. Account for encoded frames, -decoded payload and queued installation bytes together; the draft's retained -reservation covers page payload, not every transient allocation. - -**Gate:** reuse one fixed immutable root for before/after runs. Warm a known -range through SQL, then hydrate it under cold, warm and churned caches. -Measure useful prefetched pages, duplicate range bytes, fetch waves, decode -CPU, peak retained/RSS bytes, hydration completion and sibling/own-Cell p99. -Repeat with alternating-object page locators, cancellation and a full retained -budget. Preserve no cursor advancement on an abandoned batch and exact-root -verification; never treat cache content as recovery authority. - -**Implementation:** the asynchronous fetch now consults the same view-keyed -demand cache. It returns a cached prefix directly and stops a missing fetch -before a cached suffix. Fetched pages destined for immediate installation do -not create another cache. The regression warms one immutable activation through -SQLite opening, then installs eight missing pages: the initial draft made two -range calls; the changed path makes zero. Ten sparse LTX tests pass, including -checkpoint supersession, truncate/regrow, foreign activation and exact restore. -Concurrent demand/fetch single-flight, fragmented multi-span fetching and total -memory accounting remain separate qualification work. - -### 22. Reopening a persistent directory cache blocks async activation - -**Confirmed at `e50055c48bb`:** the async runtime acquisition paths call -[`replica_with_directory_cache`](../src/cell/actor/acquire.rs), which calls -[`Host::with_directory_cache`](../../crab-ltx/src/environment/host.rs) -synchronously. [`DirectoryCache::with_budget`](../../crab-ltx/src/environment/directory_cache.rs) -cleans temporaries, reads/parses the complete index, checks each file's length -and canonical path, and acquires its disk reservation before returning. -This constructor bypasses the admitted executor used by subsequent cache -reads and fills. Slow cache storage can occupy a Tokio worker even though -checksum-file creation and sparse registry setup have been isolated elsewhere. -Tokio's [fairness contract](https://docs.rs/tokio/1.53.1/tokio/runtime/index.html) -requires bounded task polling; wrapping this constructor in an async function -does not move its filesystem calls off that worker. - -There is a second, independent cost: for each accepted entry, construction -sums **all previously retained lengths** to check the byte cap. With `n` -valid entries below the cap this visits `n(n-1)/2` lengths; at the 16,384-entry -cap, 134,209,536 length visits occur before the final total. The constructor -also validates entries before applying the count cap. This is source-proven -bookkeeping complexity, not a measured fraction of HTTP response time. - -**Reproduction:** the ignored -[`directory_cache_restart_diagnostic`](../../crab-ltx/tests/host/hooks/activation.rs) -seeds a version-1 membership index and 1 KiB local files, then reopens it three -times with a 256 MiB disk budget and a derived 32 MiB cache cap. It verifies -membership, retained bytes and reservation release. Production source remains -`e50055c48bb`; only this diagnostic was added. On macOS arm64, mounted APFS, -Rust 1.97.0 and an optimized build: - -| Entries | Constructor samples (ms) | Median (ms) | -| --- | --- | ---: | -| 1,024 | 34.562 / 33.788 / 34.170 | 34.170 | -| 4,096 | 142.539 / 153.516 / 135.378 | 142.539 | -| 16,384 | 757.404 / 762.246 / 713.062 | 757.404 | - -These measurements include filesystem metadata and bookkeeping. They do not -isolate the fold, inject disk delay, exercise RustFS, measure authenticated -directory lookup, or establish a service percentile. Files were seeded locally -before timing; OS caches were not flushed. The debug run also passes and is -retained separately. Raw output is `audit-cache-restart-e500-release.log` and -`audit-cache-restart-e500.log` beneath this checkout's external Cargo target. - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ -TMPDIR="$HOME/Workspace/crabbuild-target/crab-8bc8/tmp" \ - cargo test -p crab-ltx --features replica --test host \ - directory_cache_restart_diagnostic --release --locked -- --ignored --nocapture -``` - -**Change evaluated:** maintain a checked running byte total during -reconstruction, and have runtime acquisition await one admitted blocking cache -construction job. Bound index input before allocation and avoid -validating an arbitrary number of entries beyond the accepted cache envelope. -Preserve canonical-path checks, private-temporary cleanup, reservation ownership -and origin verification. Cancellation must retain the dispatched job and its -reservations until it finishes. A cache remains an optional accelerator. - -**Gate:** repeat the entry-count curve with unchanged fixtures and slow metadata -I/O; prove an unrelated timer and resident action progress while acquisition -waits. Test canceled acquisition and simultaneous restarts against one shared -disk budget. Existing cache restart/eviction, corruption/symlink, concurrent-fill -and exact-root cache reuse tests protect behavior but do not cover constructor -latency or executor isolation. Full replay, cold acquisition and warm reacquisition -callers must use the same construction seam. The audited constructor and runtime -call also exist on `origin/main` snapshot `de0bb234abc`. - -**Implementation:** reconstruction now retains a running byte total, checks -the 16 MiB index-input cap before reading and checks membership/byte admission -before validating each file. `Host::with_directory_cache` now awaits the existing -admitted executor and returns a `Result`. All workspace callers await that one -path. This builder is absent from release tag `v1.2.4`; no synchronous alias or -new root export was added. The separate local capture APIs remain synchronous. - -Tracing the caller found repeated cache construction within one activation: -the final publisher path passed an already selected directory into a helper -that took its parent again. The public runtime regression reproduced five -cache openings across bootstrap, cold acquisition and clean reacquisition, -including incorrect parent directories. Each acquisition entry now creates -one cache host before recovery, and passes its clones through SQLite and -publication. The internal activation stages no longer replace that host. - -The off-async-worker and oversized-input regressions both fail against -`e50055c48bb` and pass with the change. A current-thread Tokio test pauses after -the first disk reservation, cancels its waiter and proves the blocking slot and -bytes remain held until dispatched work finishes. Concurrent cache construction -also shares a deliberately constrained disk budget without overcommitting it. - -With only the running-total change applied, the optimized diagnostic's -16,384-entry median fell from 757.404 ms to 528.169 ms. Filesystem validation -remains material; moving it off the async caller is a separate improvement. -This is one before/after diagnostic series, not a service percentile or a -supported recovery limit. Raw output is -`cache-construction-running-total-release.log` beneath the external target; -the failing seam logs are `cache-construction-before.log` and -`cache-owner-before.log`. - -The final optimized async-construction probe reports medians of 31.406, -130.442 and 524.585 ms for the same three entry counts; retain -`cache-construction-final-release.log` separately from the single-change -experiment. Six cache unit tests, eight directory cases, eleven activation -cases, twenty-five runtime residency cases and eight minimal-feature LTX cases -pass. The real RustFS HTTP/mTLS collaboration/takeover test and public CellNode -primitive takeover test pass, including the acknowledgement trace join in the -HTTP case. Replica/runtime all-target Clippy, formatting and documentation -validation pass. These checks do not qualify fleet recovery percentiles. - -### 23. Shared disk headroom prevents restart and cleanup can hide recovery evidence - -**Reproduced:** the three-node `e50055c48bb` run completed its offered load but -the killed owner's restart exited at the server's disk-budget check. The -[budget](../../crab-http-server/src/server.rs) subtracts at least 10 GiB from -available disk and requires 20 GiB remaining. The -[Compose renderer](../../crab-http-server/deploy/cell-issue-fleet/render.py) -caps the measured disk at 30 GiB, while all node volumes and RustFS consume one -VM filesystem. Named volumes isolate paths, not capacity or failure domains. -The later surviving-node mount sample was already below 30 GiB free. This does -not identify which workload consumed the missing space. - -The [recovery runner](../../crab-http-server/deploy/cell-issue-fleet/load.py) -restarts the old owner in `finally`. A restart exception overrides either its -earlier return value or the original failure, so the report has no `owner_loss` -receipt. Existing recovery tests prove that acknowledged results are checked -before restarting the old owner, but always let the restart succeed. They do -not cover combined recovery/restart failure. The startup budget and shared- -volume topology also exist in the compared main snapshot; the attributed -load receipt is branch-specific. - -**Change to evaluate:** use the existing capacity probe to preflight actual -mount headroom and record shared-disk topology. Keep takeover, all-acknowledgement -verification and restart results separately; any required failure still fails -the overall run. Retain the original failure alongside cleanup failure. -Provision test disk before retrying rather than weakening production reserves. - -**Gate:** successful takeover plus failed restart, failed takeover plus failed -restart, and full success must preserve distinct receipts. Test with a disk -budget crossing the startup threshold; then repeat the real RustFS run on a -provisioned disk. Independent disk faults require independent storage. The -observed restart failure is an environment/qualification limit, not evidence -that LTX lost an acknowledged mutation. - -**Implementation:** both the scheduled and functional qualifiers now write -takeover results into caller-owned receipts before restarting the old owner. -The shared cleanup boundary records the restart outcome and preserves an -earlier recovery/readback error if cleanup also fails. A restart failure after -successful recovery still fails the run. The parent qualifier saves its failed -report in `finally` and retains a failed load stage's artifact path. Scheduled -reports use schema 5 for the separate `owner_loss.restart` outcome. - -The regression reproduced cleanup replacing an acknowledged-data failure. -Twenty-six fleet tests now pass, including successful readback plus failed -restart, failed readback plus failed restart, full success, and persistence of -failed functional/load-stage reports. These tests use controlled HTTP and -Compose fixtures; they do not rerun the real disk-threshold fault or establish -successful recovery in the earlier missing receipt. Disk provisioning and a -fresh full fleet run remain required. - -### 24. Application response enrichment adds an avoidable query after commit - -**Confirmed:** [issue creation](../../crab-http-server/src/issues.rs) always -awaits `labels::catalog` after `CreateIssue` returns a committed result. -[The catalog](../../crab-http-server/src/labels.rs) routes and executes a typed -`ListLabels` query. For a returned issue with no selected label IDs, the -response selection is necessarily empty, so this query contributes no output. -A label-query error can also turn an already committed create into an HTTP -failure, making stable submission-ID retry essential. - -The sibling issue list/detail/edit paths make the same unconditional catalog -call; [pull views](../../crab-http-server/src/pulls.rs) already skip it for an -empty label selection. This behavior is present on the compared main snapshot. -It is application composition, outside the LTX codec, yet inside the public -action's measured latency. - -The new trace measures 7.482 ms median and 14.065 ms p95 outside the typed -invocation for forwarded writes. That difference includes other work: the -invocation timer starts in `PreparedCommand::execute`, after description and -encoding in [command preparation](../src/client.rs), and HTTP response readiness -follows enrichment. The trace does **not** assign all of this time to labels. - -**Change to evaluate:** fetch the catalog only when selection validation or -response rendering needs it, matching the pull-view rule. Create/detail can -inspect returned IDs; list can inspect the returned page before fetching labels. -Edit also uses the catalog to validate incoming IDs before the command: retain -that read for nonempty input, including invalid selections, and retain permission -checks even when clearing labels. An edit without label changes may still return -a labeled issue. Do not special-case the create operation because a duplicate -submission can return an existing issue that now has labels. -Instrument query, routing and enrichment boundaries before larger batching or -cached-description changes. Keep action authorization and receipt validation. - -**Gate:** public create/read/list/update needing neither selection validation -nor label rendering must issue zero label queries. Nonempty incoming selections -still reject unknown/duplicate IDs; selected labels retain their names/colors -and permissions, including forbidden attempts to clear them. -Retry an ambiguous create after subsequently labeling the issue and require -the existing issue plus current label rendering. Compare local and forwarded -HTTP actions over RustFS with the same workload and count peer operations. -Existing issue/label and peer takeover tests establish functional contracts; -query-count proof and the remaining latency gate are recorded below. - -**Implementation:** create, detail and list load the catalog only if returned -issues contain label IDs. Edit retains nonempty selection validation and -reuses that catalog for rendering; otherwise it fetches only if the committed -result contains labels. Permissions still apply to an explicit empty selection, -and the runtime still validates selected IDs atomically. The pull handlers -already use conditional enrichment and are unchanged. - -The public HTTP/mTLS regression failed before the change and passes with both -in-memory storage and real RustFS. It counts owner-side executed queries: -unlabeled create/edit performs one required repository-archive query, with no -label query; unlabeled list/detail performs only its requested query. A labeled -mutation or create replay performs the archive query plus one catalog query. -Replaying a committed create after assigning labels returns the existing issue -with its current labels. This proves replay behavior, not an injected dropped -HTTP response. The same test covers owner loss, restored collaboration state, -local actions on the successor, and native Git readback. Seven issue/label auth -tests pass, including unknown/duplicate selections and forbidden label clearing. - -An edit retaining existing labels can now perform enrichment after committing. -If that read fails, the write can have succeeded: reload the resource and use -its current version to resolve the outcome. Creation retries retain the stable -submission ID. No authority, durability, wire format, or label-validation rule -changed. Logs under the external target include `label-query-before-corrected.log`, -`label-query-after.log`, `label-query-auth.log` and `label-query-rustfs.log`. -These prove removed work and application behavior; a fixed-workload release -comparison is still needed to quantify latency or throughput improvement. - -The integration with `396e0ab1b40` keeps the newer `GetIssueDetail` composite -query for owner and replica detail reads. Issue content and label names come -from one SQLite snapshot and one typed query; a separate owner-side catalog -read would mix snapshots when the selected reader lags. The create/list/edit -conditional enrichment rules remain. Re-run their public query-count proof on -the combined source rather than interpreting this historical record as a new -replica latency result. - -### 25. A cleared checksum overlay retains historical allocation across small cuts - -**Confirmed at `e47d1fc2c48`:** `PageChecksums` derives `Clone` and stores -changed checksums in a `HashMap`. Both memory-backed and file-backed -[commit branches](../../crab-ltx/src/pages.rs) call `changes.clear()` after -merging the sealed cut. [WAL capture](../../crab-ltx/src/capture/wal.rs) clones -the checksum state before applying the next cut. Rust's -[`HashMap::clear` contract](https://doc.rust-lang.org/std/collections/struct.HashMap.html#method.clear) -retains allocation. The retained table is separate from the dense checksum -array whose allocation reuse was improved in finding 14. - -A temporary test inside the actual checksum module applied 32,768 512-byte -pages, sealed the candidate, then cloned and committed a one-page update. -On the checkout's Rust 1.97.0 toolchain it observed: - -| Observation | Overlay length | Overlay capacity | -| --- | ---: | ---: | -| After large cut merged | 0 | 57,344 | -| Empty candidate cloned for next cut | 0 | 57,344 | -| After one-page cut merged | 0 | 57,344 | - -This proves retained allocation and propagation through the actual candidate -path. It does not measure allocator bytes, RSS or service latency. A large cut -can therefore leave later tiny cuts paying for a historically large table, -even when no immutable snapshot retains the dense base. The same clear/clone -pattern exists in the compared main snapshot. File-backed capture shares it; -moving dense checksums to a file alone does not close the gap. - -**Change to evaluate:** consume or release the overlay at the successful -post-seal merge boundary so the next candidate's allocation follows its actual -changed pages. Compare allocator churn before choosing any retained-capacity -policy. Do not globally change `Clone` to discard entries: -[local recovery](../../crab-ltx/src/recovery.rs) accumulates live checksum -overlays, and [Db::resume](../../crab-ltx/src/db.rs) clones that materialized -state. Clearing those entries would change recovery semantics. File merge -failure must still fence the session; pre-seal failure must leave the original -checksum state intact. - -**Gate:** repeated large → tiny → truncate/regrow workloads on memory and -file bases at 512/4096-byte page sizes. Record allocated bytes, allocation -count, retained capacity, capture duration and per-node RSS with many Cells. -Keep `memory_candidates_preserve_snapshots_through_truncation_and_regrowth`, -`file_backed_overlay_matches_full_scan_across_update_truncate_and_regrowth`, -`cell_checksum_write_failure_fences_after_sealing_the_cut`, independent CRC -and process-exit continuation checks. The allocation probe does not replace -these integrity tests or the public RustFS action comparison. - -The audit also tested the private page-cache mechanism: 32 interleaved hits -did not retain one 4 KiB page after 32 other 64-page batches filled the 8 MiB -FIFO. This alone does **not** justify switching to LRU. The -[writable VFS](../../crab-ltx/src/writable_vfs.rs) installs demanded pages -locally and subsequent reads bypass the bridge cache. Investigate unused -prefetch eviction, retained copies of installed pages and concurrent duplicate -fetches; first prove those costs through real sparse reads and hydration. -The immutable metadata cache is a separate FIFO with repeated directory -lookups, so any policy change needs its own workload and verification scope. - -Both diagnostic tests passed in 0.12 seconds; that is test duration, not action -latency. Temporary test instrumentation was removed. The exact patch and log -are retained as `ltx-design-audit/current-head-mechanisms.patch` and -`ltx-design-audit/current-head-mechanisms.log` under this checkout's external -Cargo target directory. No runtime behavior, dependency, public contract, -qualification threshold or test inventory changed for this audit. -After removing the probes, all four existing checksum module tests passed. -The documentation check passed 28 schema/protocol assertions and 190 local -links; `git diff --check` passed. The pre-existing staged changes were preserved. - -**Implementation follow-up:** a successful sealed merge takes ownership of -the overlay. The memory branch consumes its entries into the dense base; the -file branch consumes them into the existing sorted write batch. Neither keeps -an empty allocated hash table. Ordinary `PageChecksums::clone` is unchanged, -so materialized recovery plans still preserve every unmerged entry. A failed -file merge still fences `Db` before another capture can use the partial state. - -The strengthened memory and file regressions both failed on the baseline. -They now require zero retained overlay capacity after merge, including the -next memory candidate and truncate/regrow. All four checksum module tests -pass, including 1,024/32,768-page allocation reuse and snapshot isolation; -the two minimal-feature cases pass too. Three host checksum tests cover the -large resumed capture, dense handoff and post-seal write failure. Independent -full-database CRC and process-exit continuation pass. Replica and minimal-feature -all-target Clippy pass with warnings denied. The real RustFS HTTP/mTLS test -also passes through owner takeover, collaboration readback and Git reads. -These results establish the -allocation-lifetime change and integrity; fleet action latency and aggregate -RSS remain unqualified. - -### 26. Small ownership targets cannot absorb a rounded-up deadband - -**Confirmed after the fleet run:** the -[planner](../src/fleet/placement.rs) computed receiver room as -`target - ceil(target * 2 / 100) - active_cells`. With twenty Cells and twenty -equal nodes, the target is one and even an empty receiver has zero room. With -targets two through forty-nine, the margin reserves more than two percent of -the target. This defect exists in `origin/main` at `de0bb234abc`; a longer idle -wait alone cannot qualify uniform ownership. Headroom-driven and urgent moves -have independent gates, so this is specifically a count-balancing defect. - -The public planner regression starts twenty Cells on one owner, repeatedly -applies settled transfer intents, and supplies fresh complete observations. -Before the fix, it stopped at **8/6/6** for three nodes, **8/3/3/3/3** for five, -**11/1/1/1/1/1/1/1/1/1** for ten, and **20/0/…/0** for twenty. Thus the failure -does not depend on the HTTP load keeping Cells active. The earlier live run -still cannot attribute the p99 increase to this arithmetic alone. - -Rounding the percentage down to whole Cells fixes that starvation. A second -regression then exposed a sibling gap: the planner bounded fleet-wide room, -but did not consume each receiver's room across the batch. With counts 5/1/0 -and equal weights, both donations went to a preferred receiver, producing -3/3/0 instead of 3/2/1. The planner now filters receivers without room and -checks their projected counts before accepting each donation. Urgent shedding -and material headroom-gain transfers retain their own policy; donor election, -freshness, residence, resource admission, actor settlement and control CAS are -unchanged. The existing nonzero-margin test uses a target of 100 and still -requires the two-percent margin to limit the batch to one donation. - -**Best-fix assessment:** correcting both integer quantization and batch -projection is preferable to forcing placement from the harness or weakening -inactivity and fencing. The production caller is the HTTP router's periodic -rebalance loop; its callee still releases through the actor and activates the -receiver through ordinary authority acquisition. All seven public planner cases, -fifteen private planner cases, and both server transfer/readback tests pass. -Runtime and HTTP all-target Clippy pass with warnings denied. Exact-source -Compose proof must still verify the real fleet effect. The status-only -preflight in finding 12 retains failures instead of calling unbalanced ingress -a capacity result. Loaded scale-out remains a separate design gap. - -### Re-audit decision and proof gaps - -**Is this the best fix for the reproduced waits?** Splitting remote fetch from -owner installation and separating its effect from foreground ownership remove -the worker and actor waits at their respective boundaries. Reusing the demand -cache removes the demonstrated duplicate reads. Short registry claims also -remove disk/setup waits from unrelated activation and teardown. Installation -and individual activation still occupy their SQL worker. -Demand SQLite VFS callbacks still return -synchronously under the [SQLite I/O contract](https://www.sqlite.org/c3ref/io_methods.html). - -Range-proportional compaction (4) now removes the all-index transfer and -whole-directory rebuild. Its sustained publication drain (5) remains to be -measured before considering bounded root coalescing. The highest recovery-size opportunity is bounded -authenticated checksum blocks (3, 14). These changes preserve one fenced writer, -one ordered root publisher and each command's stable receipt. Raising queues, -worker counts or provider concurrency alone does not remove the underlying work. - -The public application qualification must exercise Entity, Shard, Workflow and -read models through CellNode/application handles. Require fixed offered load, -actual execution distribution, local/forwarded and read/write action traces, -published-versus-acknowledged rates, and faults during an unpublished acknowledged -tail at 3/5/10/20 nodes. One shared Compose host cannot qualify independent-host -failure or aggregate dedicated CPU capacity. - -The staged reference-suite relocation still fails its old suite inventory -entry and remains outside this implementation. Root rules require approval for -that inventory change. Hydration's new public types live under `crab_ltx::db`, -beside their owning `Db` API; the frozen root prelude is unchanged. No inventory -was edited to suppress a failure. Full-plan completion remains unproven. - -### 27. Read replicas retain provider latency and refresh cache costs - -The read-replica extension landed on main at `396e0ab1b40`. It is an explicit -object-durability read policy, not another writer or a SQL-serving follower log. -The combined source needs new proof; earlier owner-only fleet percentiles and -sparse-writer microbenchmarks do not qualify this path. - -There is useful upstream performance evidence: three five-node RustFS pairs -at `c9d16ebeb358` measured 858–1,049 replica requests/s, median 6.70–7.33 ms -and p99 37.73–44.97 ms for a fixed 75% node-1 / 25% node-5 ingress mix. -Local-primary medians remained 1.86–1.91 ms. The [source-bound receipt](../../crab-http-server/deploy/cell-issue-fleet/qualification/2026-09-25-read-replicas.md#reader-discovery-optimization-under-fixed-ingress-traffic) -records the controlled membership-discovery improvement; the raw final report -hash was rechecked as `d5dc8e5ed6c1f77ca075457665a9ea54cfa86a40e0e2009e76657534dc0ec131`. -This is positive evidence for reducing metadata work. It does not qualify the -combined branch, a 20-node optimized workload, controlled write/refresh churn, or -independent hosts. - -**Evidence map.** HTTP issue detail selects the replica policy in -[`issues.rs`](../../crab-http-server/src/issues.rs). It calls the repository -router, then [`ReplicaReadRouter::query`](../src/client/routing.rs). -`selected` reads control and desired-reader policy concurrently on every -selection. `NodeDirectory::select_readers` shares a signed membership scan for -at most one second. The selected [`CellReadReplica`](../src/client/replica.rs) -executes SQL, then reads control and the owner's live enrollment before -releasing output. A remote attempt also performs peer authentication. The -ordinary owner query remains a sibling with different authority and lifecycle -ownership; its resident fast path is not proof of zero-provider replica reads. - -**Opportunity A: reduce repeated metadata work without weakening fencing.** -Measure warm local and remote replica reads with no new writes, then with policy -changes, owner death and authority outages. Record routing, peer authentication, -SQL, final control, final liveness and provider attempts separately. Concurrent -coalescing may remove duplicate routing reads. A post-SQL authority check -must not reuse an observation whose read started before that query completed. -An arbitrary TTL on the final check changes the response guarantee and needs -an explicit lease/invalidation contract. Keep unavailable authority fail-closed -until such a design is proven. -The existing lifecycle suite verifies concurrent selection, schema fencing, -owner failure and drain, but supplies no saturated 1-vCPU service curve. - -**Opportunity B: preserve useful immutable bytes across refresh.** -`CellReadReplica::refresh` opens a fresh exact-root view when publication -advances. [`Io::new`](../../crab-ltx/src/paged_io.rs) assigns each view a new -identifier; the shared demand cache keys pages by `(view, page)`. Unchanged -pages in an old snapshot therefore cannot directly hit that cache in the new -snapshot. Authenticated directory metadata has its own cache, so this is a -page-body reuse gap, not a claim that all metadata is downloaded again. -Measure a fixed hot read set while small updates advance the root. Record cold -bytes after each refresh, cache occupancy, retained old views and read p99. -Consider a bounded cache keyed by verified immutable object/frame identity, -with page checksum validation intact. A page number or mutable Cell ID alone -cannot safely share bytes across roots, truncation or incarnations. - -**Opportunity C: qualify replica CPU admission alongside writer isolation.** -The prior branch split writer admission by the owning Cell's worker shard; -main's replica SQL used undifferentiated admission. Applying writer affinity to -immutable views reproduced a progress failure: refresh waited behind an old -query despite an unused second SQL slot. Immutable connections now borrow any -available SQL slot, while writer jobs retain their own shard. Both paths share -the node charge and hold their permit through completion, including caller -cancellation. Shutdown closes admission before waiting for those charges. -The public lifecycle regression requires refresh to finish while an old-view -query is held, then verifies view lifetime, cancellation and drain. It now -releases its injected pause before asserting, so a regression fails without -hanging the process. This preserves main's concurrent refresh contract and the -branch's writer isolation. A cold replica still consumes shared SQL capacity; -measure mixed owner writes and replica reads on one vCPU before changing that -allocation policy. - -**Best next fix?** Finding 28 now reproduces avoidable demand read-ahead within -one view. Bound that fetch before changing cache identity across views, then -compare action-level replica traces and refresh misses on the combined revision. -Reusing authenticated immutable bytes or coalescing concurrent observations may -reduce further work. Increasing replicas, queue sizes or caching authority -without a new correctness contract does not establish lower latency. - -**Integration proof.** The combined checkout passes 27 focused LTX activation, -read-view cleanup and sparse-root cases; four worker-isolation cases; four -replica lifecycle cases; and the exact-root/refresh/policy/schema/drain case -against local RustFS. All 53 Python fleet tests, runtime/HTTP/LTX all-target -Clippy with warnings denied, server build, and 28 schema/protocol assertions -pass. These checks retain the separately staged test relocation; it remains -outside this change, with its inventory gate still open. Source-patch and log -hashes are retained beneath the checkout's external target directory. - -One initial in-memory HTTP run returned 504 for the healthy request after an -injected timeout. An instrumented rerun and the real RustFS HTTP/mTLS test -passed. The initial timing failure remains unexplained; no deadline or expected -status was relaxed. The RustFS run also verifies replica selection, owner loss, -restored application state and Git readback, and its received write joins to -the actual text logs. This shared-process debug fixture supplies functional -and trace-correlation evidence, not a throughput or tail-latency result. - -### 28. Demand read-ahead fetches a cached suffix after small updates - -**Confirmed at `5bbc7021c46`, using local RustFS.** A new diagnostic published -128 random 4 KiB payloads, opened an immutable SQL view, and read every payload -through `sum(length(hex(value)))`. It repeated the query, opened another view -of the same exact root, then opened a successor after updating only a separate -counter. Old views remained alive; page-cache capacity exceeded this working -set. Each query returned the expected aggregate with frame and page checksum -verification enabled. - -| View and operation | Range GETs | Origin bytes | -| --- | ---: | ---: | -| First root: open and scan | 3 | 540,969 | -| Same view: repeat scan | 0 | 0 | -| Same root, new view: open and scan | 3 | 540,969 | -| Successor after counter update: open and scan | 5 | 746,111 | -| Each new view: repeat scan | 0 | 0 | - -This establishes two separate costs. View identity prevents reuse across -views, as described in finding 27. Within the successor view, demand read-ahead -also fetched an already cached suffix. A second run instrumented the actual -fetch: page 15 fetched pages 15–78; the subsequent miss at page 6 fetched -pages 6–69, including **55 pages already cached**. The successor transferred -746,029 bytes versus 540,942 for the initial root, again about 38% more. -`Cache::insert` discarded the duplicate pages after transfer, verification and -decoding had already occurred. The measured excess is origin traffic; neither -38% lower service latency nor a fleet throughput gain has been demonstrated. - -**Evidence map.** The public entry is -[`VerifiedRoot::open_read_only`](../../crab-ltx/src/replica.rs), with connection -ownership in [`ReadOnlyRoot`](../../crab-ltx/src/replica/read_only.rs). -[`CellReadReplica::refresh`](../src/client/replica.rs) creates such replacement -views after an exact root advances. This diagnostic exercises that LTX opening -and SQL path directly; it excludes runtime routing and authority confirmation. -[`Io::page` and `fetch`](../../crab-ltx/src/paged_io.rs) check only the requested -page before asking `read_run` for 64 pages. `read_run` returns the first -contiguous immutable span in that window. It does not stop at an already cached -later page. Compared main `396e0ab1b40` has the same demand-fetch behavior. - -The sibling `Io::hydration_pages` already bounds a missing prefix before cached -pages. Writable VFS demand reads install pages locally, so later reads can -bypass this cache; immutable views repeatedly use it. The fix must preserve -both consumers, rather than assuming their hit patterns are identical. -Existing `read_only_roots_keep_exact_snapshots_without_materializing_pages`, -`directory_nodes_are_shared_across_exact_root_views`, and the sparse hydration -tests protect related correctness but do not assert demand-fetch byte reuse. - -**Best next fix and gate.** On a demand miss, compute the bounded uncached -prefix using the current view's cache before issuing the range read. Reuse the -existing 64-page/1-MiB bound and exact-root identity. In the deterministic case -above, the miss at page 6 should stop before page 15. Cache eviction between -selection and completion may cause a later miss; it must never cause a missing -requested page or substitute another view's data. Compare point, random and -scan workloads, fragmented roots, concurrent cache churn, hydration, deadlines, -and corruption. Keep the checksum and frame checks. This change can remove -known duplicate work without introducing a cross-root cache contract. - -Then evaluate bounded immutable-frame reuse across roots, keyed by an -authenticated locator and frame identity, while each current root independently -authorizes its page mapping. Preserve missing-metadata checks, truncation, -incarnation isolation and retained old snapshots. Cache policy, fetch -coalescing and publication batching are separate experiments. - -**Proof and limits.** Three diagnostic runs passed against RustFS -`1.0.0-beta.8-glibc`. These are debug-build work-count probes on the local host, -with no 1-vCPU/1-GiB limit applied to the Rust test. The uninstrumented successor -open-and-scan took 99.388 ms and its warm repeat 2.254 ms; single samples are -not latency percentiles. Temporary tests and fetch logging were removed after -capturing their exact patches. Under the checkout's external Cargo target, -`ltx-design-audit/read-view-audit-receipt.json` retains the source revision, -patch/log SHA256 values and measurements; the adjacent `read-view-refresh.*` -and `read-view-overlap.*` files reproduce both checks. The uninstrumented log -SHA256 is `4c3bf4555d63337bedead4edc05bacdfc9bc2d6632bc85ac5e784d10abcd7ac5`; -the overlap log SHA256 is -`690dbc1df0785cbf3cef1e63309e437955fddb4419985de2cae1c5c07cc441b0`. -The initial audit changed no runtime source, dependency version, public API or -qualification threshold. - -**Implementation and regression proof.** Demand fetching now uses the same -bounded `Cache::missing_prefix` selection as asynchronous hydration. Selection -holds the cache lock only for a bounded lookup; provider work runs after release. -Eviction during a fetch can cause a later miss but cannot change the requested -page's identity or skip its verification. The existing per-view request gate, -deadline, maximum fetch window and cache cap remain in force. - -The new [public SQL regression](../../crab-ltx/tests/cell/restore/read_ahead.rs) -failed before the fix on overlapping origin ranges. Afterward it passes for -512/4096-byte pages against both memory and actual RustFS, comparing full -payload hashes and requiring disjoint origin ranges within the isolated -working set. A repeated scan issues no range reads. RustFS reports 21 ranges / -544,913 bytes at 512-byte pages and five ranges / 540,912 bytes at 4-KiB pages. -The earlier diagnostic and this regression use different payload generators; -these counts are not a controlled latency comparison. Eleven sparse-writer -and hydration tests, four pager cache/deadline tests, minimal-feature checking, -LTX all-target Clippy with warnings denied and the downstream HTTP server build -also pass. Logs are retained -under `ltx-design-audit/read-ahead-*.log` in the external target directory. - -The fleet qualifier now retains ordered 5→20→50→5 offered-rate points at each -node count, with distinct reports and a lower-rate repeat after overload. -Exit 2 is accepted as a measurement only after integrity, traces, publication -drain and owner-loss checks finish; its `passed` flag remains false. The new -classification and sequence tests pass with the 55-test Python suite under -strict resource warnings. The reader-partition probe now closes HTTP error -responses before interpreting their bodies; existing success and rejected-body -cases also assert closure. Workflow syntax validation passes. Current-image execution of those curves, -longer churn and independent-host qualification remain open. - -The preceding [image run 36258714307](https://github.com/crabbuild/crab/actions/runs/36258714307) -passed build, Compose startup, restore, crash recovery and restart. Its receipt -and image source are GitHub merge commit `1527155752f9`, with parents -`396e0ab1b40` and branch head `5bbc7021c46`. Both the merge and that branch head -have tree `1a33007274f794d4353c0dcdb60e48c17896ad24`; the artifact's exact source -identity remains the merge commit. It does not cover the subsequent -demand-read or rate-curve changes. The separate -architecture gate still rejects the app-to-host development dependency; its -staged suite relocation and inventory approval remain outside this change. - -The follow-up [image run 36261394085](https://github.com/crabbuild/crab/actions/runs/36261394085) -passes build, Compose startup, restore, crash recovery and restart with the -demand-read fix. The receipt names merge source `a3a074cfa687`, whose tree -`9aa8caaadb16bb7c3e8dec244b2ba558891e9849` matches branch head `6169c9c0270`. -Receipt SHA256 is `38afbd6d821723db630cd84d3313fa1fdbeb88f782e96799c5e6f6d5400a7f27`; -its image digest is `sha256:b3dd7c8bb12d627b5d9b3041d1456ef9b1ba8e3fda153a2f7ea863b7fd318d97`. -The subsequent churn runner and measurement documents do not change production -runtime source. Fleet curves still require their separate receipt. Both -architecture CI jobs at this branch head report the existing app-to-host -development dependency, not a passing full qualification result. - -### 29. Fleet recovery verification omitted the issue body - -The baseline load and owner-loss checks compared titles and sometimes issue -numbers, leaving the application payload unchecked. A controllable HTTP service -reproduced five false-positive cases: corrupted creation body, boolean readback -number, changed or null readback body, and a changed earlier acknowledgement -after takeover. The recovered last issue alone could not detect the last case. - -The common comparison now requires an integer issue number, matching title and -complete body. Scheduled bodies include the run, Cell and arrival identity; -creation responses, immediate reads, all-acknowledgement verification, -published-root takeover and unpublished-tail takeover share this comparison. -The functional scale and entry-route coverage checks also verify their initial -issue bodies. The list API deliberately returns body `null`, so its duplicate -scan retains ID/title checks and is paired with the detail readback gate. - -Source: [HTTP create/detail and summary contracts](../../crab-http-server/src/issues.rs), -[shared verification](../../crab-http-server/deploy/cell-issue-fleet/qualify.py), -[scheduled actions](../../crab-http-server/deploy/cell-issue-fleet/load.py), and -[unpublished-tail recovery](../../crab-http-server/deploy/cell-issue-fleet/fault.py). -All 56 fleet harness tests pass with resource warnings treated as errors. -Load schema 7 and fault schema 2 distinguish the stronger proof and changed -payload from historical runs. Run 36265830657 exercises the stronger harness: -eleven points pass full payload recovery, while the final ten-node point fails -post-loss readback (finding 30). Twenty-node and unpublished-tail coverage on -that source remain unrun. The harness change adds no runtime or storage-format -change and does not establish a latency improvement. - -### 30. An inactive-log recovery claim serializes unrelated Cell takeovers - -**Observed:** the final ten-node point of run 36265830657 recovered `work-20` -from `node-10`, then failed to read the acknowledged issue in `work-01`, which -had the same prior owner. Six requests over approximately 3.4 seconds failed -with `Node("node recovery is already claimed")`. The recovery claim lifetime is -30 seconds. All writes in that point used object proof and publication drained -before the kill. - -**Source:** [request routing](../../crab-http-server/src/cells/router.rs) -loads a `takeover_proof`, then calls `claim_expired_for_takeover` if absent. -[The directory](../src/node/directory/recovery.rs) originally required the -current claimant to match even when the tombstone had no active log. -[The tombstone claim](../src/node/directory.rs) rejects other claimants until -expiry. Thus different Cells inherit an unnecessary node-wide exclusion. -An inactive log reaches this error; an active log returns `PendingPublication` -earlier in the request path. This separates the observed symptom from an -unfinished active-tail recovery. The same predicates exist on `origin/main` -at `396e0ab1b40` and in the measured image `8e61bcf3ad2`. - -The correct ownership boundary is a permanent expired-session fence plus a -separate CAS for each Cell. An absent/inactive log cannot have acknowledged -follower-only state: [node durability](../src/node/durability.rs) activates the -authoritative log before issuing follower proof. A tombstone prevents a stale -enrollment from activating later. An active log must still complete exclusive -recovery and pin its overlays before other nodes receive takeover proof. -[Cell takeover](../src/cell/actor/acquire.rs) checks the old session, verifies -the claimant, reloads current control, CASes ownership and restores the exact -root before serving. Sharing the node fence must preserve each of those steps. - -**Implemented:** `takeover_proof` accepts the permanent tombstone for any live -successor when the log is absent/inactive, or when active-log recovery has -sealed/retired it. A request-path claim failure reloads that proof to resolve a -competing fence or an ambiguous CAS response. Generic recovery claims retain -their exclusive generation and lease checks. No storage format, public type, -retry deadline, or acknowledgement requirement changes. - -Both the public directory regression and the two-Cell application regression -failed with the recorded claim-conflict error before this fix and pass after -it. The application regression publishes two different issue payloads and -receipts, reconstructs the controls left by one expired owner, and restores -them on different successors before claim expiry. It also passes against -local RustFS `1.0.0-beta.8-glibc` at port 19010 using a fresh isolated bucket -and prefix. This is a real SQLite/object-store regression; it does not kill -separate node processes. The directory test additionally covers stale -pre-claim observations, inactive enrolled logs, active-log exclusion, rejected -late activation and expired successors. All 34 existing node-directory tests -and seven router tests pass; the ignored RustFS case is run separately and -passes. - -Run the RustFS case with the existing `CRAB_HTTP_CELL_TEST_BUCKET`, -`CRAB_HTTP_CELL_TEST_ENDPOINT`, `CRAB_HTTP_CELL_TEST_PREFIX`, `AWS_ACCESS_KEY_ID` -and `AWS_SECRET_ACCESS_KEY` test environment. Use a fresh prefix per invocation: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/" \ - cargo test -p crab-http-server --locked --lib \ - cells::router::tests::rustfs_different_successors_restore_cells_from_one_fenced_node \ - -- --exact --ignored --nocapture -``` - -Corrected-image [run 36272807534](https://github.com/crabbuild/crab/actions/runs/36272807534) -passes startup, restore, crash recovery, restart and the three-node cluster -faults at exact source `ce02ac2e0f7e95b08f551b8a46a6e13bb5ac7e33`. The ARM64 -image digest is -`sha256:4193bb3ec8ff41960e0727c026d42fcc6541738acd52e4778a21e1cc1e362d3a`; -receipt SHA256 is -`b795e97c26f78be7cdd4dabea22ddee165bdb715575870377a43783708ca84bc`. -The downloaded receipt also passes the canonical `validate-cluster` command -on Rust 1.98 in source-only mode. This is process-loss evidence, including -unpublished follower state, but does not replace the failed multi-Cell rate -point or establish sustained capacity. - -Uniform [fleet run 36274405084](https://github.com/crabbuild/crab/actions/runs/36274405084) -failed during initial repository provisioning with that image: image and stage -source `ce02ac2e0f7`, job/fault harness `4741628d0db`. All three nodes were -healthy; a request raced catalog propagation and returned HTTP 403 (finding 32). -The report has no completed stages or timed load points. This run does not -retest the multi-Cell recovery fix or supply capacity evidence. The full -3/5/10/20-node arrival and recovery gates remain open. - -### 31. The archive check adds a serialized Cell invocation before mutations - -**Measured:** at ten nodes and 50 offered pairs/s, the archive-state check's -p50/p99 is 477.083/1,448.518 ms, versus capture's 0.528/7.932 ms. The timing is -measured around the check on the entry node; it includes routing, query and -waiting, not just the SQL statement. Actor queue p99 is also 933.306 ms. These -measurements prioritize round-trip and queue reduction over further small-cut -checksum tuning for this workload; they do not isolate host CPU or provider -service time. - -At the measured baseline, [HTTP middleware](../../crab-http-server/src/server.rs) invokes -`archived_mutation_response` before unsafe repository requests. It calls -[load_lifecycle](../../crab-http-server/src/repository_settings.rs), which routes -and executes `GetRepositoryLifecycle`; then the handler separately routes its -mutation. [CreateIssue](../../crab-http-server/src/cells/repository.rs) validates -and deduplicates its input without checking archive state in that transaction. -The preflight exists on `origin/main`; action instrumentation only exposes its -cost. Existing [HTTP/mTLS coverage](../../crab-http-server/src/server_peer_e2e_tests.rs) -expected this query, so it is not an accidental trace attribution. - -**Next experiment:** place writable-state admission at the repository command -transaction boundary, with an explicit policy for operations allowed while -archived. Prove ordered archive → mutation rejection, mutation → archive -success, unarchive → mutation success, concurrent submissions and replay of an -already-committed request. Only then remove redundant preflight reads for those -commands and compare the same workload with identical durability semantics. -A stale boolean cache would change the decision's authority and is not proof. - -This cannot be a CreateIssue-only patch: comments, issue edits, reviews, -settings and other registered mutations share the policy. Archive/unarchive -and membership have explicit HTTP exceptions. Git, LFS, contents and branches -also perform storage effects outside the collaboration command transaction; -their current archive checks need their own ordering contract. Count provider -reads and local/forwarded invocations per action as well as latency. Preserve -authorization, idempotency, durable rejection behavior and the current public -error mapping. - -**Implemented first slice:** all seven issue/comment/label mutation commands -read `repository_settings.archived` within their existing SQL transaction and -return a typed, durable rejection before application writes. Their healthy HTTP -paths omit the separate lifecycle query. HTTP client errors still perform the -policy lookup to retain the archive-error precedence shipped in v1.2.4; server -errors preserve ambiguous outcomes. Other command families keep their current -pre-handler checks. This is a bounded application change; runtime publication, -fencing, deadlines and acknowledgement policy are unchanged. - -The regression first reproduced an accepted typed mutation after archive and -the unwanted query on public HTTP create. The command proof now covers all -seven rejections, exact-root recovery, successful and rejected receipt replay -while archived, rejection replay after unarchive, and a fresh accepted request -after unarchive. Existing positive wire tags remain unchanged; `Archived` uses -tag zero. The repository source digest changes and needs normal release -admission; this does not establish mixed-version compatibility. - -The public fixture requires zero routed queries for an unlabeled create/edit -and one for label enrichment, while retaining its existing owner-loss and -durability checks. The initial memory-backed fixture passed in 46.33 seconds; -the same flow against pinned RustFS 1.0 GA on Colima passed in 36.61 seconds. -Eight focused HTTP tests also passed, covering all seven archived endpoints, -invalid-input precedence, authorization, duplicate creation, issue/comment -edits, labels and recovery. These are single correctness runs on a shared host; -their elapsed times are not an A/B performance comparison. - -**Concurrent ordering proof:** the same public application fixture now runs -[four archive/unarchive rounds](../../crab-http-server/src/archive_race_tests.rs) -through its application router and signed mTLS command clients. Each round -has a known preceding write, four concurrent label commands delivered twice, -and a known following write. Every accepted result precedes archive's durable -commit sequence; every rejection follows it. Duplicate receipts must match. -After owner fencing, shutdown, local-data loss and takeover, all 24 outcomes -replay with their original receipts. Accepted rows match exactly, rejected rows -are absent, and replay preserves the authority root at owner loss. - -The memory run passed in 39.72 seconds (five creates, nineteen rejections); -RustFS 1.0 GA on Colima passed in 40.71 seconds (six creates, eighteen -rejections). Both observed accepted and rejected concurrent commands beyond -the ordered controls. These remain in-process application hosts against a real -provider, separate from the resource-limited Compose fleet. No production code, -durability policy or request lifetime changed for this proof. - -Matched offered-load curves, prolonged multi-Cell archive/load/fault stress and -policy conversion of other mutation families remain open. Do not infer a p99 -or throughput improvement from query counts or these functional run durations. - -### 32. Catalog propagation can reject a newly created Cell before load starts - -**Observed:** run 36274405084 created repositories while three serving nodes -polled their catalogs independently. The initial `work-10` mutation first -returned four 404s. Node 01 installed catalog version 20 at -`21:54:58.104046Z`; its forwarded lifecycle query then reached node 03, which -rejected repository authorization at `21:54:58.181447Z`. Node 03 installed -version 20 at `21:54:58.183932Z`, 2.485 ms later. The final entry response was -HTTP 403. The report retained the error with `stages: []`; the workload and -unpublished-tail fault never ran. This is setup availability evidence, not a -latency or throughput measurement. - -**Source:** [server startup and refresh](../../crab-http-server/src/server.rs) -materialize a local catalog before serving, then replace it on a five-second -poll. [Peer authorization](../../crab-http-server/src/peer.rs) looks up the -repository in that local snapshot before resolving or executing its Cell. -A missing repository fails closed with the observed error. The entry node's -catalog visibility does not prove its peer's visibility. The same local -authorization and polling boundaries exist at `origin/main` `396e0ab1b40`. -The timing supports stale catalog as the cause; the generic denial alone -would not distinguish it from a principal/action mismatch. - -**Fixture change:** [run_stage](../../crab-http-server/deploy/cell-issue-fleet/qualify.py) -now completes release bootstrap and provisions the full fixed Cell population -before starting traffic nodes. The CLI initializes, publishes and drains each -Cell through [the maintenance initializer](../../crab-http-server/src/cells/initializer.rs), -then marks it ready. Nodes load that completed catalog on startup. The existing -Compose dependency chain waits for RustFS health and successful release setup -before the repository initialization command; these are the documented -[Compose startup conditions](https://docs.docker.com/compose/how-tos/startup-order/). -Subsequent scale stages reuse those repositories. All 403s remain fatal. -Startup-order and provisioning-failure regressions fail before this change and -pass after it; all 65 deterministic fleet harness tests pass. An exact-source -image and full fleet rerun are still required. - -**Product opportunity:** dynamic repository creation needs explicit readiness -across candidate execution nodes, or an authenticated catalog-version contract -with bounded refresh before policy evaluation. Preserve immediate denial for -actually unauthorized principals and revoked membership. Do not add blind -403 retries, cache an allow decision, or shorten the poll without measuring -catalog/provider load. Test create → forwarded first mutation with one lagging -peer, and membership revocation under the same conditions. The fixture change -does not implement or qualify that product contract. - -The same stage helper serves reader-replica and mode-rollout qualification, so -they inherit the startup ordering. A follow-up reconciles their old fixture -titles with `initial_issue`, and carries each acknowledged body update into -the next scale stage's expected state. Reverting to the initial body now fails -that stage's full readback. Replica discovery checks issue identity and receipt; -the timed owner/replica comparison, promotion, partition, load and rollout -paths compare the expected number, title and body. Body refresh remains its -own freshness measurement. - -The reader and rollout entry points now use the shared committed-tree image -builder and source-label verification, and pin every server service plus the -release bootstrap identity before startup. The retention continuation selects -the recorded immutable image when replacing services. This repairs sibling -wiring that still used mutable tags although `prove_node` required an immutable -image. Rollout retains its explicit runtime-source input for comparisons and -verifies it against the image label; reader scale-out requires the current -committed source. No source-mismatch fallback is added. - -The current fixture failed both replica discovery and rollout validation before -the change. All 69 deterministic harness tests now pass, including scale-out -after an acknowledged body update, rejection of old/corrupt payloads, and both -entry points rejecting a wrong-source image before starting nodes. These are -harness regressions; full live reader and rollout qualification remains open. - -### 33. Planning at the discovery start time suppresses fresh heartbeats - -**Reproduced:** the [repository router](../../crab-http-server/src/cells/router.rs) -read its clock before awaiting the signed node-directory scan. A heartbeat -published during that scan has an issue time later than the captured clock. -The directory permits bounded clock skew, but the -[placement planner](../src/fleet/placement.rs) deliberately rejects observations -from the future. Its complete-fleet count rule consequently returns no balance -when even one otherwise current member was published during discovery. The -same sequencing exists on `origin/main` at `396e0ab1b40`. - -A deterministic router regression advances its injected clock by one second -across discovery. The fixture uses real signed advertisements, settled SQLite -Cells and the ordinary release, authority and receiver-activation paths. -Before the fix, it releases zero Cells instead of one. This is a reproduced -clock-boundary defect; the test does not attribute the whole fleet delay to it. - -**Fix:** rebalance and initial ownership selection now share one private -placement snapshot. Discovery starts with the current clock, then observations -are evaluated against a new clock sample after the read. Expired advertisements -are excluded at that point; a partial view still cannot justify count balancing. -The planner's future-time rejection, freshness window, source residence, -settlement, transfer limits and authority gates remain intact. Initial placement -still reloads its selected canonical advertisement, and remote activation uses -a fresh request timestamp. No public API, deadline, concurrency limit or -serialized contract changes. - -The separate public `NodeDirectory::choose_advertised_placement` helper retains -its explicit caller-supplied logical-time contract. The live server paths use -the shared snapshot so they can sample time after I/O; reader selection and -recovery discovery have their own expiry/authority gates and do not invoke -the count-balancing planner. Their cached observations are not converted into -ownership authority by this change. - -**Fleet evidence:** [run 36291618746, attempt 2](https://github.com/crabbuild/crab/actions/runs/36291618746/attempts/2) -used the prior a62 AMD64 image. Eight of twelve load points at 3/5/10 nodes -passed; all 14,256 admitted pairs succeeded and passed integrity/recovery checks. -The four failed points omitted 144 scheduled arrivals at the generator. At -20 nodes, the first balanced sample arrived at 554 seconds, the next had -28 seconds of stability at 582 seconds, and the final had 62 seconds at -616 seconds. The strict 600-second gate rejected that final observation. -No 20-node load or unpublished-tail fault ran. The failure is retained; it -cannot establish that ownership never balanced or that this clock fix closes -the convergence budget. Report SHA-256: -`486e7c277d0ee154810d8d95c8acdb6b907d6064572e934d8338728c18bf421c`. - -**Verification:** all three focused router transfer tests pass, including the -new regression and the existing stale-generation/restored-value checks. The -same transfer regression passes against pinned RustFS 1.0 GA (2.69 seconds). -The full public HTTP/mTLS fixture also passes against GA (178.63 seconds): -application mutations, duplicate outcomes, owner takeover, restored collaboration -and native Git reads. Its archive race records 24 outcomes, 8 creates and -16 rejections, with identical duplicate/recovered receipts. This shared-host -run duration is not a service latency or fleet-capacity result. The exact -ignored regression is added to the existing CI RustFS recovery job with its -own object prefix; no dependency or inventory changes are required. - -**Best-fix assessment:** refreshing time at the consuming server boundary -corrects both live placement callers without relaxing the pure planner or -extending the qualifier deadline. Exact-source 20-node convergence and load -remain required before claiming a latency improvement. - -### 34. Concurrent catalog publishers can restore revoked access - -**Reproduced:** on `63ca7a724f7` (affected sources unchanged on `main` -`1b3702cbd9e`), import completion and periodic catalog -refresh both materialize a complete repository index after awaiting Git metadata. -The index loses the durable document revision, and each writer replaces it -unconditionally. Holding an older import document while refresh installs a -membership revocation restores the revoked member when that import finishes. -The refresh loop's separate version counter can then keep skipping the latest -durable document, leaving the stale policy installed until another revision. - -The regression signs in an administrator and a read member through public HTTP, -revokes the member through the membership API, and installs the new snapshot. -The member receives HTTP 404. Completing the older materialization changes the -same request to HTTP 200 on the original source. Both the in-memory fixture and -the fixture backed by pinned RustFS 1.0 GA reproduce that transition; the durable -revocation and session remain unchanged. A separate UUID-lookup regression -exercises the index used by peer authorization. - -**Fix:** `RepositoryIndex` carries its document revision. `RepositorySet` -compares and publishes the revision, name map, and UUID map under one write lock, -discarding equal or older completions. Startup installs that same versioned -shape. Import and periodic refresh call one `Server::install_catalog` path, -which verifies ready Cells and materializes Git metadata before publication. -The poller reads the shared installed revision rather than a private counter. -It still rejects a provider document older than the revision observed before -the read. A later concurrent installation can supersede an otherwise valid -in-flight result. - -The HTTP and UUID regressions pass after the fix. An additional test rejects a -new revision containing an uninitialized ready Cell without changing the -installed revision or exposing that repository. The RustFS HTTP test is part of -the existing provider CI gate and uses an isolated prefix for each invocation. -This change adds no request-path provider I/O, polling interval, wire field, -stored format, or dependency. Membership administration still authorizes against -the current durable catalog; data, Git, LFS, and peer routes share the protected -in-memory indexes. Five-second propagation to other nodes and the first-request -creation race in finding 32 remain separate work; finding 35 addresses the -remote execution node's first-request miss. - -### 35. A lagging execution node must discover a new repository before dispatch - -**Reproduced:** with the revision-safe installer from finding 34, an ingress -can still know a newly initialized repository while its execution node retains -the pre-creation catalog. The public HTTP/mTLS fixture now starts in that state, -with no catalog poller to repair it. Its first settings mutation returns HTTP -503 before the fix. The original ordering also failed at its first native Git -push, because that operation queries the repository Cell's lifecycle state. - -**Fix:** after peer authentication, a missing repository UUID triggers one -catalog load through `Server::install_catalog`, bounded by the existing signed -request deadline. The same installer verifies ready Cells and publishes the -revision, name index, and UUID index atomically. A shared async lock coalesces -concurrent first requests; waiters recheck the index after acquiring it. -Cancellation releases the lock. The receiver then evaluates the principal and -operation against the installed catalog, and normal dispatch retains its final -authorization check before SQL. - -Warm authorized calls and denials for known repositories add no catalog I/O. -The receiver returns an authorization denial if discovery does not establish -access; provider or readiness errors remain service failures. There is no -cached allow decision, new polling interval, retry policy, or wire field. - -**Proof:** the public HTTP/mTLS scenario exercises the first settings mutation, -subsequent Git pushes, duplicate delivery, published LTX, replica reads, owner -loss and recovery. A signed request with an unauthorized principal executes -neither catalog reads nor SQL once the repository is known. A later authorized -mutation also performs no catalog reads. Focused tests cover sixteen concurrent -misses sharing one installation, fresh membership revocation during discovery, -and a timed-out lock waiter followed by a successful refresh. The existing -provider CI runs the same HTTP/mTLS scenario against RustFS. -The local scenario passes with both in-memory storage and RustFS 1.0 GA -(`sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`) -on Colima. The HTTP nodes in this fixture run as separate listeners in one -test process; this is real-provider correctness proof, not a container fleet -measurement. - -Public ingress lookup still uses its local catalog; a request arriving at an -ingress that has not discovered the repository can remain unavailable until its -five-second poll. Known repositories retain the existing membership propagation -window. These boundaries need a separate explicit product contract. The local -fixture does not establish multi-host capacity, throughput, or a latency limit. -Cold discovery materializes the complete catalog, so its cost still grows with -catalog size and requires large-catalog qualification. - -## Safety and proof retained by the audit - -Both the [ARM64 image/Compose run](https://github.com/crabbuild/crab/actions/runs/36239430827) -and [AMD64 image/Compose run](https://github.com/crabbuild/crab/actions/runs/36239424906) -passed at `e50055c48bb`. The ARM64 image was checksum/source verified and -imported for the next fixed-workload fleet comparison. Those CI receipts cover -the preceding hydration and registry changes; they exclude the cache-construction -change above and do not establish sustained service capacity. - -The hydration follow-up passes 67 focused worker, coordination, sparse LTX and -public lifecycle tests, plus all eight minimal-feature LTX integration tests. -Runtime and replica-enabled LTX all-target Clippy pass with warnings denied; -workspace formatting and runtime documentation validation pass. The real RustFS -worker diagnostic and public HTTP/mTLS application test also pass. These checks -cover the implementation recorded in findings 9, 19 and 21; they do not close -the installation-latency, recovery-storm or sustained fleet gates. - -The action-tracing source `2bf1967c7f3` subsequently passed both the -[ARM64 image and Compose run](https://github.com/crabbuild/crab/actions/runs/36235888087) -and the [AMD64 image and Compose run](https://github.com/crabbuild/crab/actions/runs/36235876451). -Those receipts predate the hydration follow-up and must not be attributed to it. - -The action-tracing change passes the real RustFS HTTP/mTLS application test and -replays its formatter output through the same join CLI used by the fleet -runner. Twenty-three fleet tests cover scheduled arrivals, HTTP retry identity, -provenance and attribution refusal. Thirteen runtime durability cases and two -command/effect cancellation cases pass with the instrumentation. Runtime and -HTTP all-target Clippy pass with warnings denied; actionlint and runtime docs -validation pass. These checks do not establish tracing overhead or fleet -performance. The unrelated staged reference-suite relocation still fails the -layout inventory gate pending its previously requested approval; it is not -included in this tracing change. - -The placement-parity failure at `c6870fd6a88` is reproduced as a sampling race: -metrics returned one active Cell while its signed advertisement still returned -zero. A live RustFS probe on the unchanged `c12b41ef638` publisher observed -the same process converge to one in a newer advertisement about 1.2 seconds -later. The publisher/count source is unchanged between those revisions. -The original one-shot assertion rejects the retained mismatch consistently. - -The Compose collector now brackets each advertisement with fresh metrics, -checks static capacities and the expected live session, and requires the same -active count at two increasing generations on one unchanged container boot. -It fails on persistent mismatch, stagnant/regressing advertisements, malformed -metrics, restarts or the 30-second deadline. The final receipt retains the -matched node/metrics; the run log retains the convergence trace. Thirteen -collector/recovery tests pass, and the live collector passed on the RustFS -fleet with generations 1037/1038 in 2.65 seconds. The raw result and trace are -`placement-collector-live.json` and `placement-collector-live.log` beneath -the checkout's external target directory. This proves the collector against -that running image. The subsequent -[ARM64 image and Compose run 36233256995](https://github.com/crabbuild/crab/actions/runs/36233256995) -passed at `5feef9968e6`, including the collector and prior all-acknowledgement -changes. The [image workflow 36234166704](https://github.com/crabbuild/crab/actions/runs/36234166704) -also passed at cache-admission revision `efcef4b1ffa`. Neither run covers the -later action-tracing changes or sustained 3/5/10/20-node load curves. -The earlier shutdown refusal about an unsealed node log is a separate open -lifecycle observation. - -The workflow linter also reproduced eight `SC2016` failures in the existing -candidate-reuse source checks on the main snapshot. Their literal patterns now -use a quoted here-document and one mandatory check per line. Pinned actionlint -v1.7.11 passes both affected workflows, the source checks pass, and removing -each of the eight required fragments independently makes the check fail. -No warning baseline or qualification assertion changed. - -The follow-up [ARM64 image and Compose run 36227844137](https://github.com/crabbuild/crab/actions/runs/36227844137) -and [runtime/LTX property run 36227842782](https://github.com/crabbuild/crab/actions/runs/36227842782) -both pass at `c12b41ef638`. The image run retains qualification evidence and -the exact-source Linux image. This closes those two pending CI runs; it does -not supply sustained 3/5/10/20-node curves or a diagnosis of the earlier -placement assertion failure. The all-acknowledgement generator follow-up at -`9ec6da5176e` and the staged reference-suite relocation are outside that source -receipt. They still need their respective end-to-end and policy gates. - -The follow-up audit's HTTP/mTLS test passes with both in-memory -storage and real RustFS, including five concurrent calls after hint expiry, -owner loss, restored collaboration state, and Git clone/tag reads. Both tests -completed in 13.05 seconds together; that duration is not an action-latency -sample. Focused admission, enrollment-expiry, and signature regressions pass. -The runtime accepted-command cancellation test also passes. All-target Clippy -with warnings denied passes for LTX, runtime, and HTTP, including the minimal -LTX feature set. Replica-only continuation helpers now share their callers' -feature gates. The decoder fixture helper lives in the existing test module; -no test allow-list or warning baseline changed. Activation-delay injection, -performance measurement, and broad CI remain open as described in finding 15. - -Committed-source ARM64 -[run 36224838843](https://github.com/crabbuild/crab/actions/runs/36224838843) -built the `3cd0bd1bfe6` image but failed the Compose qualification at -`assert_placement_parity` in `qualify_compose_cluster.sh:448`. The assertion -compares a live advertisement's placement values with capacity and metrics -observations. The log does not isolate which conjunct failed; do not classify -this as an LTX latency regression, data loss, or a passing image qualification. -Retain the failure and diagnose the snapshots before another capacity claim. - -The checksum batching follow-up passed 37 focused cases: two file-overlay -cases, eleven capture/failure tests, seven modeled-crash cases, five resume -integrity cases, process-exit continuation recovery, and eleven independent -format/restore cases. Replica all-target Clippy with warnings denied and the -minimal-feature local roundtrip example pass; the latter restored its visible -issue after deleting the source database. The same three minimal-feature -unused capture/checksum warnings remain. The combined routing/checksum tree -also passed the real RustFS HTTP/mTLS application-mutation, owner-takeover, -restored-collaboration, and Git-read test. These prove functional behavior; -no checksum latency SLO or fleet capacity is claimed. - -The follow-up audit reran the LTX prune-accounting fault test and the runtime -published-root/local-prune-failure test; both passed. The scheduled runner's -six HTTP/scheduler tests also pass, including stopping when an acknowledged -issue returns 404. Local RustFS uniform and overload smoke results are -[recorded with their image limitation](../../crab-http-server/deploy/cell-issue-fleet/README.md#scheduled-harness-smoke-2026-09-26). -Those results prove the harness and published-root recovery, not a latency SLO. - -Streaming cleanup passed 44 host-hook tests, six bundle cases, two node-frame -cases, ten independent format/restore cases, four minimal-feature codec cases, -and the runtime published-root/local-prune-failure test. The LTX replica build -passes all-target Clippy with warnings denied. At that revision the minimal-feature tests emitted unused capture/checksum -warnings; no warning baseline or policy inventory changed. The cost runner built in release mode and completed -the real RustFS comparison above. Remaining decoder memory and fleet latency -gates are explicitly open. - -The decoder follow-up passed 62 focused cases: seven replica codec, eleven -independent format/restore, 26 sparse/exact-root, two external-vector replay, -six minimal-feature codec, six bundle, two node-frame, and two public cleanup -cases. Replica all-target Clippy with warnings denied and the release cost -runner build passed; that revision retained the same minimal-feature warnings. Deep decoder -fuzzing and the broader runtime paths are delegated to the existing CI gates. - -Current-source ARM64 -[CI run 36216278190](https://github.com/crabbuild/crab/actions/runs/36216278190) -built the image and receipt validator but failed the cluster gate waiting for -the elected successor's recovery-work counter after owner loss. The script had -already verified recovered issue visibility. The missing counter evidence must -not be treated as data loss or successful image qualification. Source tracing -found that the gate sampled only the elected successor, which need not be the -recovery claimant, and compared it with physical `server-c`'s earlier counters. -The qualifier now derives work from each surviving process's own before/after -snapshots, rejects resets or missing data, and retains those snapshots in the -receipt. Six deterministic cases cover the distinct claimant/owner roles, -aggregation, process changes, invalid counters, and inactive-log zero work. -The native ARM64 [follow-up run 36219430132](https://github.com/crabbuild/crab/actions/runs/36219430132) -passed the corrected Compose gate at `a16c8efcc2b`, including owner loss and -follower recovery. That result qualifies that source's functional fault path; -it predates decoder commit `0360485311b` and the routing/checksum changes above. -It supplies neither sustained 3/5/10/20-node latency curves nor current-head -image qualification. -The [HTTP container run at `0360485311b`](https://github.com/crabbuild/crab/actions/runs/36220235482) -also passed, along with that revision's decoder fuzz workflow. Routing commit -`4e0d71fe43d` and the checksum batching above still need their own image proof. - -The no-job property-workflow failure has a reproduced configuration cause: -job-level `env` referenced `runner.temp`, but GitHub only exposes that context -at the later [step environment boundary](https://docs.github.com/en/actions/reference/workflows-and-actions/contexts#context-availability). -The repository's pinned actionlint v1.7.11 rejects the previous file at that -expression. Moving the existing target-directory setting to both Cargo steps -passes the same check and preserves the 1,000-case workload. The repaired -[property run 36222852526](https://github.com/crabbuild/crab/actions/runs/36222852526) -passed both runtime and LTX suites at `8586757a6eb`. The follow-up -[property run 36224241056](https://github.com/crabbuild/crab/actions/runs/36224241056) -passes at worker-admission revision `b8798fdcfdc`. -[Native ARM64 container run 36222681957](https://github.com/crabbuild/crab/actions/runs/36222681957) -passes at `7b20ebe484f`; it predates worker admission and image-provenance -enforcement. The separate app-to-host dev-dependency policy failure remains open. - -Seven existing tests passed locally with real SQLite and in-memory object -storage: four `environment::tests::directory_cache` cases, missing cached-root -metadata refusal, scheduled compaction with byte-identical restore, and -contiguous sparse hydration coalescing. These validate retained invariants; -they do not measure the proposed optimizations or RustFS latency. - -Run the same focused checks with a checkout-specific external target: - -```sh -export CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" -export TMPDIR="$CARGO_TARGET_DIR/tmp" -test -d "$TMPDIR" && test -w "$TMPDIR" -cargo test -p crab-ltx --features replica --locked --lib environment::tests::directory_cache -cargo test -p crab-ltx --features replica --locked --test cell warm_root_cache_does_not_mask_missing_metadata -cargo test -p crab-ltx --features replica --locked --test cell scheduled_cell_compaction_promotes_fanout_and_preserves_root -cargo test -p crab-ltx --features replica --locked --test cell sparse_hydration_coalesces_contiguous_cell_frames -node crates/crab-cell-runtime/docs/validate.mjs -``` - -Keep `synchronous=FULL` while measuring these changes. SQLite documents that -`NORMAL` in WAL mode can lose committed transactions after power loss; -[the upstream contract](https://www.sqlite.org/pragma.html#pragma_synchronous) -requires any relaxation to be justified by the external-proof and fresh-restore -model, including crash tests. Removing redundant cache-index syncs has a much -narrower proof obligation than changing SQLite commit durability. diff --git a/crates/crab-cell-runtime/docs/primitives.md b/crates/crab-cell-runtime/docs/primitives.md deleted file mode 100644 index 522c504f2..000000000 --- a/crates/crab-cell-runtime/docs/primitives.md +++ /dev/null @@ -1,397 +0,0 @@ -# Implement SQL, KV, Blob, Queue, Cron, and Workflow primitives - -All primitives execute through typed Rust bindings and the same Cell actor. They share request deduplication, SQLite transactions, LTX publication, exact-root recovery, admission, and receipts. - -| Document intent | Value | -| --- | --- | -| Content type | Reference | -| Audience | Native module authors and runtime contributors | -| Goal | Choose and use a primitive without creating a second durability path | - -[Back to the Cell runtime index](README.md) - -## Use one common transaction boundary - -Commands receive `CommandContext`; queries receive `QueryContext`. Neither exposes a raw connection. - -```mermaid -flowchart LR - Command[Typed command] - Context[Bounded CommandContext] - Primitive{Primitive procedure} - Ledger[sys_requests + sys_meta] - App[Application tables] - LTX[LTX publication] - - Command --> Context --> Primitive - Primitive --> Ledger - Primitive --> App - Ledger --> LTX - App --> LTX -``` - -Runtime tables use the `sys_` prefix; native primitive tables use reserved -`kv_`, `blob_`, `queue_`, `cron_`, and `workflow_` prefixes. The SQLite -authorizer denies application SQL access to every reserved table, transaction -control, connection configuration, and schema changes. - -## Deduplicate every mutation - -Application requests use a 16-byte request ID and a canonical operation digest. The ledger stores the encoded outcome before commit. - -| Existing row | New request | Result | -| --- | --- | --- | -| No row | Valid identity and digest | Execute once | -| Same ID and digest | Any retry | Return stored outcome | -| Same ID, different digest | Conflicting reuse | Durable rejection | -| Expired identity | Any payload | Reject before handler | - -Destination effects use `sys_inbox` and a 32-byte effect ID. Internal scheduler operations use short-lived identities that public listeners cannot submit. - -## Use SQL for repository-local relational state - -`SqlCell` runs bounded parameterized batches against an explicit Cell with the SQL role. - -```rust,ignore -let batch = SqlBatch { - statements: vec![SqlStatement { - sql: "INSERT INTO issues(title, state) VALUES(?, ?)".into(), - parameters: vec![ - SqlValue::Text(title), - SqlValue::Text("open".into()), - ], - }], -}; - -let committed = sql.batch(identity, batch).await?; -``` - -The SQL boundary enforces: - -- 128 statements per batch -- 1 MiB encoded input -- 1,000 result rows -- 1 MiB encoded output -- Read-only statements on the query path -- Mutating statements on the command path -- No transaction or schema-control statements - -Repository handlers should prefer typed command and query types over exposing arbitrary SQL at the HTTP boundary. - -## Use KV for scoped atomic metadata - -`KvNamespace` hashes the scope to a fixed shard. Keys and list prefixes never cross that shard. - -```rust,ignore -let request = KvAtomicRequest { - scope: b"repo:123".to_vec(), - checks: vec![KvCheck { - key: b"settings/version".to_vec(), - condition: KvCondition::Version(expected), - }], - mutations: vec![KvMutation::Put { - key: b"settings/default_branch".to_vec(), - value: b"main".to_vec(), - expires_at_ms: None, - }], -}; - -let outcome = kv.atomic(identity, request).await?; -``` - -The KV procedure applies all checks before any mutation. A failed check returns a durable `PreconditionFailed` outcome. - -| KV contract | Limit or behavior | -| --- | --- | -| Atomic items | 128 checks and mutations combined | -| Value | 4 MiB | -| Atomic operation | 4 MiB plus 64 KiB for keys and framing | -| List page | 1 MiB ordinarily; one larger value occupies its own page | -| Version | Incarnation plus sequence, 28 bytes | -| Expiry | Logical timestamp evaluated inside the Cell | -| List order | Binary key order within one scope and prefix | -| Cleanup | Bounded scheduler Tick | - -KV values remain in the Cell's SQLite database and LTX history. A module that -uses the 4 MiB maximum must declare an atomic input limit and get/list output -limits of at least 4 MiB plus 64 KiB for framing. The aggregate atomic and -list-page budgets prevent a batch of maximum-sized values from bypassing -admission. For frequently replaced large bodies, use Blob to avoid repeated -SQLite and LTX writes. - -Deleting and recreating a key produces a new version. An old version cannot match the new incarnation and sequence. - -## Use Queue for at-least-once work - -Queue sends hash the producer ID to a shard. Consumers claim one explicit shard at a time. - -```rust,ignore -let sent = queue - .send(identity, QueueSendRequest { - producer_id: request_id, - payload, - available_at_ms: now_ms, - }) - .await?; - -let claimed = queue - .claim( - claim_identity, - shard, - QueueClaimRequest { limit: 16, lease_ms: 30_000 }, - ) - .await?; -``` - -Queue state transitions are: - -```mermaid -stateDiagram-v2 - [*] --> Ready: send - Ready --> Leased: claim - Leased --> Done: ack with token - Leased --> Ready: retry or lease expiry - Leased --> Leased: extend with token - Ready --> DeadLetter: attempt limit or retention limit - DeadLetter --> [*]: effect acknowledged - Done --> [*]: retention cleanup -``` - -The claim command publishes its lease before returning payloads. Consumers validate the exact token at the claim receipt before starting external work. - -A ready message past its retention limit is dead-lettered rather than dropped: -the expire class moves it to `DeadLetter` with its payload, and the configured -dead-letter target receives a typed effect when one is registered. The retention -class runs before that transition inside one Tick, so a dead-lettered message is -removed by the cleanup on a later Tick once its effect has settled — terminal -rows therefore stay observable for at least one Tick. - -| Queue contract | Limit or behavior | -| --- | --- | -| Payload | 256 KiB | -| Claim batch | Bounded by registered command output and item limit | -| Lease | 5s to 300s | -| Attempts | 20 | -| Retention | 30 days from enqueue | -| Ordering | No FIFO guarantee | -| Delivery | At least once | - -A dead-letter transition inserts a typed durable effect in the same transaction. The source row retains its payload until that effect reaches a terminal state. - -Queue controls are shard-scoped and use the same request ledger as sends and leases: - -- Pause stops new claims and makes published-claim revalidation fail, while live leases may still ack, retry, or extend. -- Resume reopens claims and advances a monotonic control generation. -- Purge deletes only non-leased messages in batches of at most 128. -- Redrive moves dead messages back to ready only after any dead-letter effect is terminal. -- Info returns bounded aggregate counts instead of scanning message payloads. - -## Use Blob for transactional object data - -`BlobNamespace` hashes the object key to a stable shard. Multipart upload -metadata, part digests, the published manifest, request outcomes, and LTX state -commit in one SQLite transaction domain; part bytes are immutable, -content-addressed objects in the configured Crab object store. A completed -manifest never points at an unrecorded part reference, and range reads verify -each object-store part before returning bytes after restore or failover. - -Blob supports: - -- Multipart begin, idempotent part upload, atomic complete, and abort -- Create-only and ETag compare-and-swap publication or deletion -- Per-part BLAKE3 integrity verification on write and range read -- Bounded range reads and lexicographic per-shard listing -- Atomic replacement followed by deletion of the unreferenced prior upload -- Scheduler cleanup of expired, unpublished uploads - -| Blob contract | Limit or behavior | -| --- | --- | -| Key | 1 to 1,024 bytes | -| Part | 256 KiB | -| Parts | 4,096 | -| Object | 1 GiB | -| Range read | 512 KiB | -| User metadata | 8 KiB | -| Upload lifetime | 1 minute to 7 days from mutation issuance; acceptance rejects an already expired upload | -| List | 128 objects from one explicit shard | - -Blob bodies do not live in the Cell database. `BlobNamespace` uploads each -bounded part to the configured object store before committing its digest and -size in SQLite. The manifest is the durable publication boundary; unreferenced -content-addressed parts are safe to retry. The configured object-store -lifecycle policy must reclaim abandoned parts. The -`BlobArtifactStore::sweep_unreferenced` building block limits each pass to 128 -deletions but scans the unordered listing until that limit is reached. A -product-level collector must quiesce writes throughout the scope, pass -references from every Cell sharing it, and use a grace cutoff. The helper is -not wired to a product collector yet. - -## Use Cron for failover-safe recurring triggers - -Cron schedules are durable rows advanced only by the serialized Cell Tick. Each due occurrence inserts a typed cross-Cell effect and advances `next_due_ms` in the same transaction. - -The destination receives `CronInvocation`, which includes schedule ID, generation, occurrence, scheduled timestamp, and the module payload. Registry construction verifies every compiled target namespace, command ID, codec version, and input limit against the release descriptor. - -| Cron contract | Limit or behavior | -| --- | --- | -| Minimum interval | 1 second | -| Maximum interval | 1 year | -| First due time | Up to 5 years ahead | -| Payload | 256 KiB | -| Catch-up | One durable occurrence at a time, bounded by Tick budget | -| Delivery | Durable effect with destination inbox deduplication | -| Controls | Upsert, pause, resume at an explicit time, delete | - -A schedule is a fixed interval plus an explicit first due time. Cron -expressions and time zones are not part of this contract: `Upsert` takes -`interval_ms` inside the interval bounds above and every fire advances the -schedule by exactly one interval. An application that needs calendar semantics -computes the next due time itself and resumes the schedule at that instant with -the documented controls, so the expression dialect and zone database stay above -the primitive. - -Blob upload lifetime and Cron's first-due window are evaluated from the -mutation's issued timestamp. The serialized Cell still rejects a Blob upload -whose expiry has passed before acceptance. A Cron schedule whose due time -passes while the mutation is waiting is accepted and becomes eligible on the -next Tick, preserving the caller's absolute schedule without making request -latency a correctness failure. - -An owner crash after commit cannot lose an occurrence: the effect and next occurrence are in the same LTX root. A retry cannot execute the destination command twice because its inbox resolves the stable effect identity. - -## Use Workflow for durable state machines - -A workflow definition is compiled Rust with a stable digest. New runs pin the current digest; existing runs continue with the retained definition they started with. - -```rust,ignore -impl WorkflowDefinition for MergeWorkflowV1 { - fn digest(&self) -> Digest { MERGE_V1_DIGEST } - - fn transition( - &self, - state: &[u8], - event: &[u8], - context: WorkflowContext, - ) -> Result { - let event = MergeEvent::decode(event)?; - decide_merge(state, event, context) - } -} -``` - -Workflow transitions run inside SQLite and may produce: - -- New durable workflow state -- Timers -- Native activities -- Typed cross-Cell effects -- Terminal completion, failure, or cancellation - -The transition callback cannot perform network or object-store I/O. Native activities run after their claim root is published. - -```mermaid -sequenceDiagram - participant T as Workflow transition - participant DB as SQLite - participant P as LTX publisher - participant A as Activity supervisor - participant E as External system - - T->>DB: Persist state + activity intent - DB->>P: Publish exact root - P-->>A: Published claim receipt - A->>DB: Validate lease at receipt - A->>E: Run registered Rust future - A->>DB: Publish completion or retry -``` - -Workflow IDs select the shard. Signal IDs make delivery idempotent. Activity completion requires the exact lease token and attempt. - -Workflow controls preserve the deterministic history boundary: - -- Pause is accepted only when no activity lease is live. Ready activities and timers remain durable but cannot be claimed or fired. -- Resume returns the same run to running state without synthesizing an event. -- Restart is accepted only for a terminal run. It deletes the terminal local history and starts a new run under a new request-derived run ID and the current definition. -- Cancel remains an idempotent workflow event and cancels outstanding local work. - -The quiescent-pause rule avoids converting an already-running external side effect into a lost completion and unintended replay. - -| Workflow contract | Limit or behavior | -| --- | --- | -| State or event payload | 1 MiB | -| Activity payload | 256 KiB | -| Activity attempts | 20 | -| Activity lifetime | 7 days | -| Definitions | Current plus every digest referenced by stored runs | -| Execution | Deterministic transition; retryable native activity | - -## Deliver cross-Cell effects through an inbox - -A command emits typed Cell effects through `CommandContext::emit_effect`, which -uses the command-owned allocator. Effects carry typed Cell commands only and -inherit the source tenant and application. - -```mermaid -flowchart LR - Source[Source transaction] - Ledger[sys_effects] - Supervisor[Effect supervisor] - Peer[Authenticated peer] - Inbox[Destination sys_inbox] - Target[Target command] - - Source --> Ledger --> Supervisor --> Peer --> Inbox --> Target -``` - -Registry validation requires every effect target namespace to be declared. Cross-tenant targets fail before writes. - -The delivery path preserves these properties: - -- Effect bytes and operation digest remain stable across retries -- Destination incarnation is resolved at delivery time -- Destination inbox deduplicates execution -- Destination success means its exact root was published -- Source acknowledgement happens in a later source transaction -- `Resolve` recovers an ambiguous destination result - -The design doesn't claim an atomic transaction across source and destination. It provides durable at-least-once delivery with idempotent destination execution. - -| Effect contract | Limit or behavior | -| --- | --- | -| Encoded input | At most 1 MiB; claim and acknowledgement budgets reserve their fixed overhead inside the same bound | -| Claim batch | 1 to 32 effects per claim | -| Lease | 5s to 300s, and an extension stays inside the same bounds | -| Attempts | 20, after which the effect fails instead of retrying | -| Effect lifetime | At most 7 days from emission; a longer requested expiry is rejected | -| Inbox retention | 7 days past the effect's own expiry, then the destination inbox drops the record | -| Destination | Same tenant and application; a cross-tenant target fails before any write | -| Delivery | At least once, with idempotent destination execution | - -## Let the scheduler advance time-based state - -Each mutating procedure recomputes the earliest due timestamp inside its transaction. The typed Tick advances bounded work from all installed classes. - -| Maintenance class | Example | -| --- | --- | -| Request ledger | Delete expired outcomes | -| KV | Remove expired entries | -| Queue | Reclaim leases, expire messages, clean terminal rows | -| Workflow | Fire timers, retry activities, clean terminal runs | -| Blob | Delete expired unpublished uploads | -| Cron | Publish due occurrences and advance schedules | -| Effects | Claim, retry, extend, acknowledge, clean source or inbox rows | - -When a Tick reports no local transition, the compiled registry tells the scheduler whether an activity or effect runner can claim work for that namespace. - -## Install only compiled primitive modules - -Primitive mechanics are reusable, but registration is not automatic. A new module must include: - -1. A concrete native Rust module caller -2. A stable namespace and shard count -3. A SQL migration with a checked digest -4. Typed operation IDs and codec fixtures -5. Exact-root restore coverage -6. Capacity and failure tests for its workload - -Do not add a public generic SQL, KV, Blob, Queue, Cron, or Workflow endpoint. Product-specific HTTP handlers remain the external API. diff --git a/crates/crab-cell-runtime/docs/runtime.md b/crates/crab-cell-runtime/docs/runtime.md deleted file mode 100644 index b44d8b99e..000000000 --- a/crates/crab-cell-runtime/docs/runtime.md +++ /dev/null @@ -1,497 +0,0 @@ -# Trace Cell execution and failure recovery - -The Cell runtime serializes accepted commands, binds each SQLite commit to an immutable LTX root, and publishes that root through one authoritative control CAS. This page defines the actor, transaction, timeout, takeover, and shutdown behavior. - -The product path races exact object-store publication with a write-all follower -proof. Either proof may release a command result. The actor remains occupied -until object publication finishes, so later work cannot observe an unpublished -head. If the owner dies first, takeover seals the failed node log and pins and -consumes its recovery overlays before serving. See -[Follower durability and warm failover](failover-and-followers.md). - -| Document intent | Value | -| --- | --- | -| Content type | Reference | -| Audience | Runtime contributors | -| Goal | Implement or debug one Cell without violating publication and fencing invariants | - -[Back to the Cell runtime index](README.md) - -## Separate actor ownership from SQL execution - -One actor owns the Cell state machine. A fixed worker pool owns SQLite connections. - -```mermaid -flowchart LR - Caller[CellClient] - Mailbox[Bounded Cell mailbox] - Actor[Cell actor
control + pending cuts] - Worker[Stable SQL worker
Db owner] - Publisher[CellPublisher] - Origin[(Object store)] - - Caller --> Mailbox --> Actor - Actor --> Worker - Worker --> Actor - Actor --> Publisher --> Origin -``` - -`CellHandle` is a cloneable mailbox sender. It never exposes a SQLite connection. Stable Cell-ID routing keeps one `Db` on one operating-system thread until close. -Bootstrap passes the runtime-configured replica host to both that worker-owned -`Db` and the publisher, so local filesystem and resource admission apply to -the same captured cuts. -`CellNode` binds the compiled application's per-namespace database and capture -ceilings to its runtime before serving. Every bootstrap, idle acquisition, and -takeover checks the supplied `CellReplica` limits before changing ownership or -opening a root. Restored paths also check the recovery-store limits. Standalone -`CellRuntime` users supply their own LTX policy. - -The node bounds: - -- SQL worker count from 1 through 16 -- 256 queued commands per worker shard -- 10,000 active Cells per node before resource-derived reductions -- Per-Cell request and byte admission -- Node-wide memory, disk, and activity admission - -Worker-job admission uses one slot per SQL worker. A job waiting for a busy -worker holds neither another worker's slot nor a node worker-job reservation. -Once dispatched, the job owns its slot and reservation until execution ends, -including when its caller is canceled. Lifecycle and publication-confirmation -messages retain their bounded worker queue and do not need a job permit. - -Cancellation of a caller doesn't cancel accepted work. The actor still records and publishes the result, so a retry can resolve it. - -`CellClient::with_local_resolver` binds product owner selection before the -underlying transport. Its `LocalCellResolver` returns a local handle, `None` to -delegate, or an error that stops dispatch. Describe, command, query, and mutation -resolution share this path. Products may restore an idle cataloged Cell using -runtime admission and authority CAS; the framework's default runtime resolver -only looks up existing ownership. Apply admission backpressure after the local -resolver so the same request budgets cover local and remote invocations. - -An embedding service can opt into `CellClient::with_admission_backpressure` -when its request budget permits waiting for owner capacity. Client clones share -finite call-count and encoded-input byte budgets; exhausting either still fails -immediately. Mailbox operations acquire per-Cell FIFO semaphores using the -runtime's request and byte limits. -The byte charge is encoded input plus the operation's maximum result size, so -routing and execution can overlap within the owner bounds. Weighted FIFO -admission prevents small calls from starving older large waiters. Different -Cells have independent gates, retained only by admitted calls. Known capacity -refusals from other callers receive paced retries until the admission -wait expires. Commands retain their request identity, digest, and expected -incarnation, and revalidate expiry before each attempt. Fencing and ambiguous -outcomes are returned unchanged, including ambiguity caused by capacity during -publication. The wait limit never cancels an accepted attempt. Describe shares -the client budgets and capacity retry, but skips the owner mailbox gate because -metadata description does not enter that mailbox. Queries and resolution use -the mailbox policy; replica queries retain their -separate admission path. Unconfigured clients retain immediate refusal. -Each transport stage has its own wait budget; an encompassing typed or HTTP -operation can take longer. Command expiry is still rechecked before dispatch. - -The input budget charges the retained envelope, one attempted copy, and envelope -overhead. Runtime result and mailbox reservations remain authoritative. This -does not bound HTTP request bodies or caller-owned typed inputs; the embedding -service must account for those separately. - -## Execute commands in six phases - -The actor completes these phases in order: - -```mermaid -sequenceDiagram - participant A as Actor - participant W as SQL worker - participant L as crab-ltx - participant O as Object store - - A->>W: Execute registered command - W->>W: Lookup request ID and digest - W->>W: Savepoint + handler + sys_requests - W->>L: Commit and capture cuts - W-->>A: PendingCommit - A->>O: Prepare immutable root - A->>O: CAS control to new root - O-->>A: Exact successor or conflict - A->>W: Confirm and prune exact cuts -``` - -The worker transaction applies this procedure: - -1. Validate the request ID, operation digest, expiry, and input bound -2. Return a stored outcome when both identity and digest match -3. Reject identity reuse when the stored digest differs -4. Open an application savepoint -5. Execute the registered synchronous handler -6. Store success or durable rejection in `sys_requests` -7. Advance `sys_meta.sequence` and derive `next_due_ms` -8. Validate any durable database page reservations after all receipt and metadata writes -9. Commit SQLite and capture every unpublished cut - -Handler errors roll back the application savepoint. Runtime ledger updates still commit when the error is a durable business rejection. Every registered call reports its owning module, kind, outcome, and duration to the installed `CellTelemetry` sink from the thread that executed the handler, so the server can chart one primitive module without knowing its operations. - -SQLite may automatically roll back the whole command on capacity or interruption -errors. The managed LTX writer recognizes completed rollback using autocommit -and its WAL commit observer, preserves the original error, and keeps the Cell -servable. No request receipt or commit sequence advances for that failed -command. An observed commit or failed rollback still requires fencing; an -unreachable result must never be reported as a proven rollback. - -For typed commands, a direct SQLite `FULL` remains the original SQLite error -locally and maps to `RESOURCE_EXHAUSTED` / `NOT_STARTED` over peers. Command -execution wraps fenced commit/publication errors as unknown before this mapping; -a nested `FULL` therefore cannot become a refusal. This mapping is specific to -typed command execution. Migration and other peer operations retain their own -outcome contracts. - -### Bounded application BLOB writes - -`CommandContext::write_sql_blob` fills an already allocated BLOB at a byte offset, -with at most 1 MiB of operation data per call. An application can allocate an -image with SQL `zeroblob` and fill it in bounded slices without repeatedly -allocating replacement images. Allocation, writes, and application indexes stay -inside the command savepoint and publish through the normal LTX boundary. -Propagate write errors so partial images roll back. - -The method opens only the main database, rejects runtime/primitive and SQLite -internal table names, and closes the handle before returning. SQLite incremental -I/O does not invoke the SQL authorizer, triggers, or CHECK constraints; applications -must maintain their invariants explicitly in the same command. It cannot grow -the BLOB. SQLite rejects unsupported table types and writable indexed columns. -The SQL capability integration fixture covers bounds, protected names, rollback -on rejection/error, and byte-for-byte recovery after publication. - -### Durable database capacity for deferred work - -A Cell can install `primitives::capacity::SCHEMA` and use -`CommandContext::reserve_database_capacity(key, bytes)` during prepare. The -primitive rounds bytes up to pages and records a durable claim under a stable -key. Repeating the key requires the same rounded count. The claim remains until -an explicit release; timeouts never reclaim it. No padding BLOB is written. - -Before committing commands, effect deliveries, bootstrap, or migrations, the -executor checks `page_count - freelist_count + reserved_pages <= max_page_count`. -This check includes runtime receipts and metadata. Refusal rolls back all writes, -leaves no receipt, and keeps a proven-rollback owner usable. A running total makes -the check independent of the number of claims. Cells without the primitive have -no reservation check beyond their existing SQLite capacity limit. - -A resolver releases its own claim before applying deferred work within the same -command. Success publishes both together; rejection or failure restores the -claim. The final check still protects other claims. SQL and incremental BLOB -access cannot modify the protected `capacity_` tables. Trusted migrations must -preserve both tables and their accounting; they are not an untrusted SQL API. - -This reserves SQLite page capacity only. Applications must bound their own -future page demand, including index changes and runtime receipts. WAL, capture, -local disk, and memory admission remain independent. `database_used_bytes` -reports occupied pages and excludes reusable freelist pages. - -The runtime lifecycle capacity tests cover receipt and effect refusal, failed -release, changed-session/address root restore, bootstrap, and migration. SQL -capability tests cover direct access and incremental BLOB protection. - -## Publish before replying - -`PendingCommit` owns the request identity, predecessor control, encoded reply, commit sequence, and captured cuts. The actor doesn't accept the next mutation until this commit reaches a terminal publication result. - -| Publication result | Actor action | Caller outcome | -| --- | --- | --- | -| Capture fails after SQLite commit | Fence local worker; restore from authority | Outcome unknown until resolved | -| CAS accepted | Confirm root, prune exact cuts | Committed result with receipt | -| Response lost, origin equals proposal | Adopt exact successor | Committed result with receipt | -| Root accepted, local cut pruning fails, object-only proof | Fence local worker; recover from the pinned root | Outcome unknown until resolved | -| Transient preparation failure | Retry with bounded backoff and renew ownership | Caller continues waiting | -| CAS winner differs | Fence, discard local handle, reload authority | Outcome unknown | -| Deadline expires after SQL started | Interrupt SQLite, fence admission, wait for callback exit | Outcome unknown | - -When a follower proof wins but immutable-object publication returns a storage -error, the actor keeps the ordered publication obligation and retries it with -bounded backoff for a short grace period. The fleet-proven result and logical -head remain readable while the exact object root catches up; if the root still -cannot be published when the grace period expires, the Cell fences and leaves -the node-log tail for takeover recovery. A control conflict, lease loss, or -non-storage publication error fences immediately. -The publication-start trace records its queue wait, queued count and bytes, -and signed commit-sequence lag from the last published root. A negative lag -would expose a root rewind instead of being clamped away in observability. - -The actor never reruns a handler after SQLite may have started it. `Resolve` reads the durable ledger at an authoritative root. - -Runtime capture leaves each complete local LTX file readable but defers its -file and directory flush. The actor then submits those exact bytes to the node -log and object publisher. A follower fsync or authoritative object-root CAS—not -the owner-local file—proves durability before a result can be observed. After -the root publishes, the worker reverifies and deletes the matching local cut -without first flushing either that copy or its deletion. Every activation owns -a fresh local session, so a crash can leave only quarantined residue; it cannot -turn that residue into acknowledged state. Standalone `crab_ltx::Db::capture()` -remains synchronously durable. - -Immutable-root preparation pins every selected capture by its open file handle -through verification and upload retries. Later path replacement cannot change -the source. LTX inspection verifies its declared metadata and digest, and the -multipart uploader hashes the complete file again before publishing the -immutable object. The path performs no defensive local copy or scratch flush; -the proposal cannot reach authority until all immutable dependencies upload -successfully. Multi-cut batches open and inspect up to four pinned captures at -once while the immutable predecessor graph is verified independently. The -ordered descriptors are joined only before exact chain validation. - -Fresh `Db` captures privately retain the page index already authenticated by -their encoder. Root preparation reuses it instead of decoding the same local -LTX file again, while multipart upload still verifies every source byte against -the captured digest. Caller-constructed local segments do not carry this -private provenance and retain the full inspection path. -Index retention is capped at 1 MiB per pending `CaptureBatch`; larger capture -cohorts fall back to decoding. Descriptor construction, directory updates, and -index upload share the retained bytes rather than copying them at each stage. -The executor retains only one unpublished batch. - -Root preparation also overlaps independent content-addressed uploads. The LTX -body and index, changed and initial directory nodes, and root metadata use -bounded concurrency under the runtime's shared I/O permits. Initial directory -construction retains at most eight encoded nodes awaiting upload. The proposal -remains private until every dependency upload completes, so authority cannot -observe a partial root. - -## Use receipts for read consistency - -A receipt identifies the Cell incarnation and commit sequence. A query with a minimum receipt runs only after the local owner reaches that position. - -```rust,ignore -let committed = issues - .create(identity, create_input) - .await?; - -let observed = issues - .get(Some(committed.receipt), committed.output.number) - .await?; -``` - -Queries run on the owning SQL worker under a read-only application boundary. They don't produce LTX, modify control, or bypass namespace and schema checks. - -Replica SQL shares the owner's bounded SQL worker admission pool. Its slot -stays charged until the blocking query exits, including cancellation. Cell peer -codecs use the node primitive-job ledger with deadline-bounded waiting; S3 -session enrollment runs outside the CPU reservation so provider latency does -not reject otherwise idle concurrent reads. - -Explicit replica routing also has one five-second deadline covering discovery -and every selected-reader attempt. Each attempt receives an equal share of the -remaining time divided by the remaining candidates, so a blackholed connection -or stalled local resolver leaves time to try a healthy reader. Local and remote -attempts share this rule; the remote request carries the reduced time budget. -Receipt, authority, and authorization checks still apply, and replica routing -never falls back to the owner. A cancelled SQL waiter retains its snapshot and -admission until the running SQL job exits. - -## Apply one absolute operation deadline - -Native commands and queries receive a five-second wall deadline. The same deadline covers: - -- Waiting for the SQL worker -- Synchronous Rust handler execution -- SQLite progress interruption -- Sparse page faults -- Object-store range reads triggered by the sparse VFS - -Worker admission and the native queue share an atomic start/cancel boundary. -If the deadline wins before execution starts, the worker skips the operation; -commands and queries return a deadline error, and resolution returns Unknown. -The untouched Cell remains usable. Cancellation of a caller's response future -alone does not cancel an accepted mutation. Reservations stay held until the -worker acknowledges deadline cancellation or the running callback exits. -Background hydration likewise retries a queued expiry without fencing or marking -hydration complete. Migration is different: it has already closed the old -capability, so any failure still fences and recovers ownership. - -Arbitrary Rust cannot be preempted safely. When a callback exceeds the deadline, admission closes immediately, but the runtime retains worker and byte permits until the callback exits. - -After exit, recovery closes the SQLite handle and reloads control. It releases only authority that still names the same Cell, incarnation, code, schema, owner, and epoch. - -HTTP peer codecs use a separate primitive-job budget. Immediate and queued -admission share that budget; queued admission checks one absolute deadline -before waiting and after waking. Canceling a waiter reserves nothing, and -runtime shutdown wakes queued callers with `RuntimeClosed`. Callers must -reserve retained request bytes before waiting. Enrollment storage I/O releases -the codec slot; decoded requests remain untrusted until signature verification -rechecks the signed enrollment's lifetime. - -The HTTP receiver applies its received transport budget to enrollment, -resolution, activation, dispatch waiting, and reply encoding. That budget -starts after the request body arrives and remains distinct from the native -work deadline above. Canceling the HTTP wait does not cancel an accepted -mutation's durable completion; the caller resolves an ambiguous result using -the same stable request identity. - -## Recover from panic without losing the worker - -The worker catches native callback panic at its fixed thread boundary. SQLite unwinds the transaction, the affected Cell becomes fenced, and the worker continues serving other Cells. - -| Panic point | Result | -| --- | --- | -| Before control publication during bootstrap | Release admission; no owner becomes visible | -| Inside an accepted command | Return outcome unknown; quarantine tentative local state | -| Inside a native activity | Record activity failure or retry; keep worker pool alive | - -Production code must not use panic as application control flow. - -## Acquire and take over a Cell safely - -Idle acquisition and stale-owner takeover both reserve local capacity before claiming authority. - -```mermaid -flowchart TD - Load[Load catalog proof and control] - Kind{Control state} - Idle[CAS Idle to Recovering] - Observe[Observe unchanged owner for 15s] - Takeover[CAS next epoch and local owner] - Open[Open exact authoritative root] - Verify[Verify sys_meta identity, schema, sequence] - Serve[Enter Serving] - - Load --> Kind - Kind -->|Idle| Idle --> Open - Kind -->|Recovering or Serving| Observe --> Takeover --> Open - Open --> Verify --> Serve -``` - -Any control change restarts the 15-second observation period. The winner hydrates only after its ownership CAS succeeds. A losing contender doesn't download the database. - -Sparse activation starts with no materialized pages. Full recovery reserves destination bytes before download and installs through an exclusive same-directory scratch file. - -## Reuse a local database only under a continuity record - -A clean release writes one resume record beside the database it closed. The -record names the Cell, incarnation, schema, installed code, and the exact -published root the file holds; the capture continuation and dense page -checksums live in `crab-ltx` sidecars next to the same file. - -A same-node wake matches that record against the observed control and then -moves the file onto the fresh activation path instead of restoring the root. -The record never authorizes a tick, a read, or a write: the worker still -verifies `sys_meta` identity, schema, sequence, and the SQLite position against -the authoritative root, so a stale, foreign, torn, or half-written image is -discarded and restored instead of served. The ownership epoch is deliberately -absent from the match, because acquiring an idle Cell raises the epoch while -leaving the root untouched. - -Two fences keep the fast path honest. A database whose WAL is not checkpointed -is refused, because the file may sit behind the continuation it would seed from; -and a sparse activation must be fully materialized, because an unfaulted page -is a hole rather than data. Every failure above falls back to the exact restore -and costs one cold activation, never correctness. - -## Renew and self-fence ownership - -One node-level scanner renews owned Cells every three seconds. A mutation publication also advances owner progress. - -The runtime closes admission when it cannot prove ownership for ten seconds. It doesn't wait for the 15-second takeover window to expire. - -Renewal changes owner liveness fields only. It preserves root, code, schema, and durable `next_due_ms`. - -## Schedule durable work from SQLite state - -Bootstrap and every command transaction derive the earliest durable deadline from SQLite. The published control binds `next_due_ms` to the same root. - -The node scheduler: - -1. Ticks resident Cells whose published due time has passed, from memory -2. Consumes due hints: one key per released deadline, listed from the current - minute bucket and the five behind it, each confirmed against its control -3. Reads revision-pinned catalog pages and assigns 256 catalog shards through - rendezvous hashing -4. Runs the shard scan as a backstop every thirtieth cycle, visiting at most - 128 due Cells per scan -5. Sends the typed maintenance Tick locally or to the authenticated owner -6. Acquires an idle or stale Cell only when no valid owner can execute the Tick -7. Runs registered activity and effect supervisors outside SQLite - -Steps 1 and 2 run every cycle, so a Cell this node owns and a Cell whose owner -released it with a deadline are both ticked without a population scan. Step 4 -is what covers a missing hint — a failed write, a hint older than its window, -or a Cell released before hints existed — and bounds that case at one backstop -period instead of a full shard pass. - -A hint names one released Cell's deadline in the minute bucket that deadline -falls in, and a listing walks the current bucket and the five behind it. A -release whose deadline is already further behind than that window publishes -nothing: no listing would see the key, so the backstop covers it instead of -leaving behind an object nothing consumes. A key a listing meets but cannot -parse is deleted, so a foreign object under the prefix cannot be re-listed -forever. - -A Tick advances at most 128 ledger, expiry, lease, timer, or retention items. Protected shares prevent one maintenance class from starving another, and a Tick that reserves a share for a class it does not run fails instead of silently shrinking its usable work. - -When node-log recovery is configured, each scheduler cycle uses -`NodeDirectory::live_for_recovery` to read live membership and expired-log -candidates in one fresh directory scan. Candidate selection reuses that bounded -observation for its existing one-second lifetime; claimant admission and the -fencing CAS still reload authoritative records. Failed or cancelled fresh scans -discard the previous observation. Peer authentication, release activation and -ordinary `live` calls retain their direct reads. - -## Shed under node pressure - -The actor samples its own reservation ledger four times a second: memory is resident plus retained bytes, disk is the replica budget, and jobs are the worker, primitive, and hydration aggregate the placement block advertises. The sample therefore reports what this node admits, not a host guess. - -The hysteretic classifier enters shedding at 80 percent on any dimension and returns to normal below 60 percent, and either transition needs evidence sustained for one second. While shedding, the actor starts one bounded eviction per sample through the same movement budget and victim selection a transfer uses, so a hot node releases settled Cells instead of admitting work it cannot hold. A brief spike never triggers a move. - -## Drain in ownership order - -`CellRuntime::active_cell_targets` returns tenant-scoped identities from the -actor's verified activation proofs. It includes active owners even if a caller -cancelled after activation, excludes draining owners, and refuses a fenced node. -The inventory is bounded by local residency and is advisory: movement still needs -`idle_transfer_candidates` and the exact-generation `release_idle_cell` preflight. -Catalog proofs retain their original tenant/application in memory; catalog -serialization and root formats do not change. - -A clean per-Cell drain closes SQLite before releasing control to `Idle`. Releasing control first would allow a successor to open while the previous writer still owns local mutable state. - -Node shutdown follows this order: - -```mermaid -flowchart TD - Close[Close HTTP and Cell admission] - Requests[Drain accepted HTTP and Git work] - Schedulers[Stop schedulers and activity supervisors] - Publish[Publish every accepted Cell command] - Sql[Close SQLite handles] - Release[Release owned controls] - Pool[Close and join SQL and blocking pools] - Log[Close the object-covered node log] - Session[Stop heartbeats and withdraw the session] - - Close --> Requests --> Schedulers --> Publish --> Sql --> Release --> Pool --> Log --> Session -``` - -Readiness closes as soon as terminal drain starts. Lease maintenance continues -through Cell publication and the node-log close barrier; early withdrawal -fences that authority and makes the drain fail. Work and lease tasks share one -bounded supervisor and one absolute shutdown deadline. Remaining runtime, -pool, or handle clones stay permanently closed after shutdown. -The SQL pool also waits for admitted immutable-reader queries and snapshot -opens running on blocking tasks. Closing reader admission and joining the -dedicated SQL threads alone does not drain those tasks. Their existing job -charges remain held until execution exits, including after caller cancellation; -offline retention cannot proceed through a successful node drain before then. - -## Preserve these invariants - -Runtime changes must preserve all of these conditions: - -- One actor serializes mutations for one Cell -- Every accepted mutation reaches a published result or an explicit unknown outcome -- The runtime replies with success only after exact root publication -- Lost CAS responses are accepted only when origin equals the exact proposal -- Fenced local SQLite state never becomes a future publication source -- Draining closes SQLite before releasing control -- Internal, application, and effect outcomes use their distinct retention contracts -- Scheduler state comes from the same SQLite transaction as application state - -Use [delivery.md](delivery.md) for the tests that prove these paths. diff --git a/crates/crab-cell-runtime/docs/rust-api.md b/crates/crab-cell-runtime/docs/rust-api.md deleted file mode 100644 index 4cd253a79..000000000 --- a/crates/crab-cell-runtime/docs/rust-api.md +++ /dev/null @@ -1,360 +0,0 @@ -# Add a native Rust service to Crab - -Crab services are statically linked Rust modules. A module declares stable schemas, operation IDs, codecs, namespaces, workflow definitions, and activities; `crab-http-server` freezes that inventory before readiness. - -| Document intent | Value | -| --- | --- | -| Content type | How-to and API reference | -| Audience | Crab feature contributors | -| Goal | Add one typed product feature and deploy it through the existing server | - -[Back to the Cell runtime index](README.md) - -## Understand the compile-time boundary - -The server is the only composition root. Repository owners cannot upload executable code or choose a module set at runtime. - -```mermaid -flowchart LR - Source[Rust module + migration] - Registry[RegistryBuilder] - Binary[crab-http-server image] - Descriptor[Canonical release descriptor] - Fleet[Compatible Crab fleet] - - Source --> Registry --> Binary - Registry --> Descriptor - Binary --> Fleet - Descriptor --> Fleet -``` - -The runtime intentionally excludes: - -- Dynamic libraries -- WebAssembly or JavaScript execution -- Subprocess handlers -- Network registration -- Public primitive SDKs -- Per-repository executable bundles - -A separate workspace crate may organize domain code, but `crab-http-server` still owns registration, authentication, routing, and lifecycle. - -## Declare a module descriptor - -`ModuleDescriptor` is static inventory. `RegistryBuilder::finish` compares it with the functions actually bound by the binary. - -```rust,ignore -static REPOSITORY: ModuleDescriptor = ModuleDescriptor { - name: "repository", - source_digest: REPOSITORY_SOURCE_DIGEST, - retained_codes: RETAINED_CODES, - schema_min: 1, - schema_max: 2, - migrations: MIGRATIONS, - commands: COMMANDS, - queries: QUERIES, - workflow_definitions: WORKFLOW_DIGESTS, - activity_types: ACTIVITY_TYPES, - namespaces: NAMESPACES, -}; -``` - -The registry rejects: - -- Duplicate module, namespace, command, or query IDs -- Missing or extra function bindings -- Noncontiguous schema migration coverage -- Incorrect migration digests -- Narrowed codec ranges or byte limits -- Commands whose module does not own the target namespace -- Undeclared effect targets -- Queue dead-letter cycles -- Missing workflow definitions or activity bindings - -The bounds are part of the contract: - -| Field | Bound | -| --- | --- | -| Build source revision | 1-128 characters, with a nonzero lock digest | -| Modules per build | 1-128 | -| Namespaces per build | At most 128, each with a 1-128 character name and 1-4096 shards that are a power of two | -| Module schema range | `schema_min` at least 1, `schema_max` at least `schema_min`, nonzero source digest | -| Migration SQL | Nonempty, at most 1 MiB, contiguous from `schema_min`, digest-bound | -| Operation input and output limits | 1 byte to the wire bound (4 MiB + 64 KiB) | - -Registration order does not change canonical release bytes. - -## Register typed commands and queries - -A command mutates one Cell. A query observes one Cell at an optional minimum receipt. - -```rust,ignore -pub struct RenameRepository; - -impl Command for RenameRepository { - const MODULE: &'static str = "repository"; - const ID: u32 = 21; - const CODEC_VERSION: u32 = 1; - - type Input = RenameInput; - type Output = RepositorySettings; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: RenameInput, - ) -> Result> { - rename_repository(context, input) - } -} -``` - -The registry uses monomorphized decode, execute, and encode trampolines. It does not expose a raw byte-handler escape hatch. - -```rust,ignore -registry.bind_command::()?; -registry.bind_query::()?; -``` - -Command and query IDs are independent. Changing a wire shape requires a new codec version, not a silent reinterpretation. - -## Encode bounded wire values - -Every command input and output implements `WireValue`. The codec supports fixed-width scalars, bounded bytes and text, counts, and strict option tags. - -```rust,ignore -impl WireValue for RenameInput { - fn encode(&self, out: &mut BoundedEncoder) -> Result<(), CodecError> { - out.write_text(&self.name) - } - - fn decode(input: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - name: input.read_text()?.to_owned(), - }) - } -} -``` - -Decoding rejects trailing bytes, invalid tags, noncanonical floating-point values, and declared-limit overflow. Add exact byte fixtures for every new input and output version. - -## Use transaction-scoped capabilities - -`CommandContext` exposes deterministic metadata and bounded procedures. - -| Capability | Purpose | -| --- | --- | -| `cell_id()` | Read the verified target Cell ID | -| `target()` | Derive deterministic effect targets | -| `sequence()` | Allocate stable transition-local identities | -| `now_ms()` | Use the runtime-sampled logical timestamp | -| `sql()` | Execute an authorized bounded SQL batch | -| `emit_effect(&EffectCommandIntent)` | Append one typed cross-Cell intention to the command ledger | - -`QueryContext` exposes the Cell ID, commit sequence, logical timestamp, and bounded read-only SQL. - -Command and query timestamps are at least the logical time persisted in the -Cell snapshot. This includes forwarded commands, effect delivery, and explicit -replica queries: a backward clock sample cannot hide already-due work or expose -values that expired before that committed time. Queries do not advance persisted -time. Request expiry and owner/session fencing continue to use their own clock -and lease checks. - -Contexts do not expose database paths, raw object storage, control records, HTTP clients, or transaction commit methods. - -## Call a command through CellClient - -Product code resolves a `CellTarget`, authenticates the product request, and creates a stable mutation identity before dispatch. - -```rust,ignore -let result = client - .command::( - &target, - MutationIdentity { - request_id, - issued_at_ms: issued_at, - expires_at_ms: expires_at, - }, - RenameInput { name }, - ) - .await; - -match result { - Ok(committed) => respond(committed.output, committed.receipt), - Err(InvocationError::Rejected(committed)) => reject(*committed), - Err(InvocationError::Pending(evidence)) => resolve_later(*evidence), - Err(error) => fail(error), -} -``` - -`CellClient` validates namespace, role, code, schema, and incarnation. It chooses the local actor or authenticated peer path without changing command semantics. - -For a command whose caller may be cancelled while awaiting dispatch, use -`ApplicationHandle::prepare_command::` and retain a clone before calling -`PreparedCommand::execute`. The prepared value fixes the exact input digest, -identity, and owner incarnation before any mutation is sent. After cancellation, -resolve its `evidence()` through a separate application handle. Retry the retained -prepared command only when resolution is `Absent`; treat `Committed` as final and -`Unknown`, `Expired`, or resolution failure as unresolved. Preparation itself has -no mutation side effect. - -## Stream mutable Cell state safely - -Use `CellStateStream` when a response producer must query mutable Cell state -after the response head. Emission is serial, receipt-monotonic, bounded by one -deadline, and fail-closed on cancellation or owner fencing. - -```rust,ignore -let mut stream = client - .open_state_stream::(&target, deadline) - .await?; - -let first = stream.emit(GetEventsInput { after: None }).await?; -send_chunk(first.output).await?; - -let next = stream - .emit(GetEventsInput { - after: Some(first.receipt.commit_sequence), - }) - .await?; -send_chunk(next.output).await?; - -stream.finish(); -``` - -Each `emit` passes the preceding `Receipt` as the next minimum watermark. An -HTTP routes use `crab_http_server::state_observing_body` to send a chunk only -after `emit` returns; it must not read the Cell handle or logical head directly. -The helper serializes input consumption and cancels the stream when the body is -dropped. A custom adapter can call `stream.cancellation().cancel()` from a -disconnect handler to wake a pending emission. - -## Keep HTTP policy in crab-http-server - -A route adapter performs product concerns before invoking the runtime. - -```mermaid -flowchart LR - Route[Axum route] - Auth[Authenticate and authorize] - Resolve[Resolve repository UUID] - Input[Validate product input] - Client[Typed CellClient call] - Response[Map durable outcome] - - Route --> Auth --> Resolve --> Input --> Client --> Response -``` - -The runtime must not know browser sessions, repository names, organization roles, Git ref policy, or HTTP status codes. - -## Register a primitive capability - -Primitive modules bind fixed operation IDs to their typed handles. - -```rust,ignore -impl KvModule for RepositoryCache { - const MODULE: &'static str = "repository"; - const ATOMIC_COMMAND_ID: u32 = 30; - const GET_QUERY_ID: u32 = 31; - const LIST_QUERY_ID: u32 = 32; -} - -register_kv::(&mut registry)?; -``` - -The namespace descriptor must declare the matching role and shard count. Blob, Queue, Cron, and Workflow modules also implement maintenance registration so the scheduler can advance upload expiry, leases, occurrences, timers, and retention. Cron target bindings are checked against the destination namespace owner, command ID, codec version, and exact input limit before readiness. - -Read [primitives.md](primitives.md) before binding a primitive. - -## Run external work as a native activity - -Command handlers stay synchronous and deterministic. An activity may call Git, object storage, or another service after its claim root is published. - -```rust,ignore -impl ActivityHandler for PublishRelease { - const TYPE: &'static str = "publish-release"; - - fn execute( - context: ActivityContext, - payload: Vec, - ) -> Pin + Send + 'static>> { - Box::pin(async move { - publish_release(context, payload).await - }) - } -} -``` - -Async activities run under a CPU-derived bound. Blocking activities reserve a slot in a joined fixed operating-system thread pool before claim. - -The supervisor validates the exact published lease, heartbeats through durable commands, and records completion or retry. If a completion response is lost, it checks the request ledger: a committed result is returned without rerunning the handler, while an absent request is retried with the same identity and result. If the ledger remains unknown or cannot be read, `run_once` returns pending evidence; the caller must treat that activity result as unresolved. Panic becomes activity failure and does not kill the pool. - -## Forward only private registered messages - -The peer protocol is private to compatible Crab nodes. [`contracts/peer.proto`](contracts/peer.proto) defines messages but no generated public service. - -```mermaid -sequenceDiagram - participant R as Receiving node - participant D as Node directory - participant O as Owning node - participant A as Owning Cell actor - - R->>D: Load signed live owner advertisement - R->>O: mTLS + signed bounded request - O->>O: Verify fleet, release, time, action - O->>A: Dispatch registered command/query - A-->>O: Typed reply or mutation evidence - O-->>R: Strict encoded response -``` - -The protocol permits at most two forwarding hops. Mutation retries happen only when transport proves the first attempt did not start. Ambiguous attempts use `Resolve`. - -Commands, queries, and mutation resolution carry a signed expected Cell ID, -incarnation, code digest, and schema. The receiver compares all four with its -resolved owner handle before execution or ledger lookup; actor admission still -fences ownership changes after resolution. A stale observation returns a -not-started refusal. It cannot authorize a different Cell or schema. - -A host that already read this description from Cell authority can bind one -routed request with `CellClient::with_observed_description`. This avoids the -extra peer `Describe` round trip. The binding is specific to that Cell and -does not refresh itself: after refusal, resolve a fresh route and retain any -pending mutation's original identity and digest. Other clients obtain the -description through their transport. Effect delivery retains its destination -incarnation and durable inbox contract; migration already carries both source -and successor code/schema. - -These required fields revise the unshipped private wire contract. Compatible -application rollout tests use nodes implementing the same peer contract; -older private binaries that omit or reject these fields cannot participate. - -Native SQL, KV, Blob, Queue, Cron, Workflow, Activity, and Effect operations -cross this private boundary as registered `CellCommand`/`CellQuery` codecs. -The wire contract deliberately has no primitive-specific peer operation path; -those commands remain behind the application registry and its module checks. - -SQL text never crosses the migration peer boundary. The owner derives a trusted migration from its frozen registry. - -## Add a native feature - -Implement one vertical slice in this order: - -1. Add the application migration and its checked digest -2. Declare stable namespace, operation, codec, and byte-limit descriptors -3. Implement `WireValue` for inputs and outputs -4. Implement typed commands, queries, or activities -5. Bind every declaration in the server's compiled registry -6. Add the authenticated product route adapter -7. Test replay, rejection, source-loss restore, and exact publication -8. Inspect the built binary's canonical release descriptor -9. Roll the complete server image through release activation - -Keep the change one canonical path. Do not retain a legacy storage fallback unless a shipped contract requires it. - -## Preserve version compatibility during rollout - -A rolling-compatible binary must retain every code and schema pair that authoritative Cells may still use. It must also retain referenced command codecs, migrations, workflow definitions, activities, namespaces, and byte limits. - -The release gate rejects a candidate that narrows those contracts. An incompatible change requires maintenance activation and a purpose-built transform when stored work cannot drain naturally. - -Rollback operates at whole-image granularity. The older image may return only while its compiled registry still supports every authoritative Cell. diff --git a/crates/crab-cell-runtime/docs/standalone-replication-audit.md b/crates/crab-cell-runtime/docs/standalone-replication-audit.md deleted file mode 100644 index 9e204c869..000000000 --- a/crates/crab-cell-runtime/docs/standalone-replication-audit.md +++ /dev/null @@ -1,96 +0,0 @@ -# Standalone replication compatibility audit - -Status: **Decision recorded — HARD REMOVE executed by plan 017**. This record is -the evidence boundary for the breaking cleanup; the canonical Cell path is the -only shipped replication surface after this change. - -Audit date: 2026-09-18. Planned source: `4a77b6f1252a` (`origin/main`). -Approver: repository maintainer (explicit authorization in this task). -Decision date: 2026-09-18. Target release: current unreleased breaking change -(or the next breaking release if this branch is cut into a release). - -## Export and ownership inventory - -The `crab-ltx` `replica` feature exposes these standalone surfaces: - -| Export | Owner | Stored shape / authority boundary | -| --- | --- | --- | -| Retired epoch-head/paged/scheduler exports | Removed from `src/replica.rs`, `src/paged_vfs.rs`, and `src/schedule.rs` | `head.json`/`manifest.json` epoch-head records and their standalone examples are not read or migrated | -| `bundle::{Bundle, BundleEntry, BundleRow}` | `src/bundle.rs` | CRB1 envelope retained as a Cell recovery-overlay input; rows are scoped by `BundleEntry::for_cell` | -| `with_paged_io_deadline` | `src/paged_io.rs` | Process-local deadline scope for sparse faults | -| `Hydration` | `src/writable_vfs.rs` | Local writable sparse state and owner-driven hydration | -| `CellReplica`, `PreparedRoot`, `RootRef`, `RecoveryOverlay`, `RootObjectRef` | `src/replica.rs`, `src/replica/` | Canonical Cell immutable root graph and exact object extents; mutable authority remains `CellAuthority` | - -The final rows are the retained canonical path. The retired epoch-head records -must not be interpreted as Cell roots. - -## Workspace callers and teaching surfaces - -An exhaustive current-tree search and feature-gated module review found no -production caller outside `crab-ltx` itself before execution. The current tree -now contains only canonical callers: `CellReplica`, Cell-scoped `Bundle`, and -`with_paged_io_deadline` for the shared sparse worker. The standalone teaching -surfaces removed by plan 017 were: - -- `crates/crab-ltx/README.md` examples; -- the standalone replication, paging, sparse-writer, compaction, and RustFS - scale examples; -- standalone remote/publication/capability tests; -- the former `crates/crab-ltx` parity/scalability notes and `UPSTREAM.md`. - -Cargo metadata confirms the crate is `publish = false`, but that fact is not -treated as evidence of non-shipment. The release tag `v1.2.4` contains the -standalone modules and examples (the source commit is -`4d097cce362048b843d557394827847e03102eab`). - -## External-consumer search - -On 2026-09-18, the available repository and local Git history were searched for -the exported names, crate name, storage prefixes, and example names. No public -consumer was found in the checked workspace or Git history. This is **unknown -external usage**, not proof of absence: the crate is unpublished, tags contain -the source/examples, and no private package registry or downstream repository -was in scope. A maintainer must review release/package telemetry before any -breaking removal. - -## Unique proof map - -| Standalone proof | Canonical Cell proof | Status | -| --- | --- | --- | -| Epoch-head CAS and historical `open_exact` | `CellAuthority` control CAS plus `CellReplica` exact `RootRef` | Covered by Cell authority/root publication and takeover tests; the standalone head is intentionally not migrated | -| Bundle selection by repository/epoch | Cell bundle rows and authenticated directory extents | Covered for Cell-scoped rows by `cell_roots` bundle preparation/recovery tests | -| Paged frame hash/CRC and writable sparse VFS | `CellPagedDatabase`, `Db::hydrate_step`, shared VFS | Covered by Cell root sparse-read, coalescing, hydration, and checksum-failure tests | -| Caller-driven level schedule | Cell scheduled compaction and actor hydration tick | Covered by Cell scheduled compaction and runtime owner scheduling | -| Standalone source-loss/reopen tests | Cell source-loss takeover/publication tests | Covered by Cell root reopen, restore, and runtime failover suites | -| Celld/rustyriver compatibility fixtures | Crab CRB1/LTX exact-root tests | Missing external wire-compatibility qualification; not an authorization contract | - -The hard-removal implementation ports the unique safety ownership to the Cell -tests before deleting the standalone test owners. It retains no runtime reader, -alias, or prefix reinterpretation. - -## Options and decision boundary - -The recorded decision is **HARD REMOVE**. The authorized scope is the -standalone `Replica`/`ReplicaHead` epoch-head operations, standalone paged -database/VFS exports, `CompactionSchedule`, their feature-gated source, -examples, tests, and teaching docs. Shared `Bundle`, authenticated index/frame -helpers, `Hydration`, deadline control, and all `CellReplica` APIs remain. - -Migration boundary: no runtime compatibility reader, fallback, alias, or -prefix reinterpretation is allowed. Objects written under the tagged -standalone `ltx//...` layout remain outside the Cell root graph. Any -external consumer must perform an explicit offline export/import into a -Cell-scoped root before upgrading; this change does not delete remote data. - -## Evidence commands - -```text -rg -n "ReplicaHead|PagedDatabase|PagedConnection|CompactionSchedule|prune_published|open_paged" crates/crab-ltx crates/crab-cell-runtime crates/crab-http-server -cargo metadata --format-version 1 --locked -git tag --contains 4d097cce362048b843d557394827847e03102eab -``` - -Related architecture records: [crab-ltx README](../../crab-ltx/README.md), -[UPSTREAM.md](../../crab-ltx/UPSTREAM.md), -the [canonical LTX scaling design](canonical-ltx-scaling.md), and -[execution plan 017](../../../advisor-plans/017-execute-standalone-replication-decision.md). diff --git a/crates/crab-cell-runtime/docs/storage.md b/crates/crab-cell-runtime/docs/storage.md deleted file mode 100644 index 70e6b7849..000000000 --- a/crates/crab-cell-runtime/docs/storage.md +++ /dev/null @@ -1,396 +0,0 @@ -# Inspect Cell storage and exact recovery - -Cell storage separates one mutable authority record from immutable SQLite history. Readers verify every immutable object by digest, while writers update authority with an observed object-store ETag. - -| Document intent | Value | -| --- | --- | -| Content type | Reference | -| Audience | Storage, LTX, and runtime contributors | -| Goal | Implement identity, control, root, page, and recovery paths without weakening verification | - -[Back to the Cell runtime index](README.md) - -This page describes the implemented `cells/v1` object-store contract. The -in-place session record and control-pinned recovery overlay that make follower -fsync a valid response-release proof are specified in -[Follower durability and warm failover](failover-and-followers.md). - -`v1` names the one current development format; it is not a release counter. -Until Crab ships a persistent Cell format, storage changes update this layout, -its document schemas, all readers and writers, tests, and documentation in one -change. Development data may be recreated. Do not introduce a new `cells/vN` -prefix, dual readers, compatibility branches, or migration code merely because -the structure changes. - -## Derive stable identities - -Tenant, application, namespace, session, incarnation, and request IDs are 16 bytes. Cell IDs and BLAKE3 digests are 32 bytes. - -```text -LP(x) = u32_be(length(x)) || x - -cell_id = BLAKE3( - "crab.cell.v1\0" || tenant_id || application_id || - namespace_id || LP(partition) -) -``` - -Partition bytes have a 1,024-byte limit. Routing uses stable namespace rules: - -| Primitive | Partition input | -| --- | --- | -| Repository SQL | Repository UUID | -| KV | Hash of scope | -| Blob | Hash of object key | -| Queue send | Hash of producer ID | -| Queue claim | Explicit shard number | -| Cron | Hash of schedule ID; explicit shard for list | -| Workflow | Hash of workflow ID | - -Shard counts are powers of two from 1 through 4,096. An existing namespace cannot change its shard count. - -## Keep object paths typed - -`crab-ltx::CellStorageLayout` constructs every path. Callers never concatenate untrusted path fragments. - -```text -cells/v1/identity.json -cells/v1/apps//release.json -cells/v1/apps//releases/.json -cells/v1/apps//catalog/tenants//<00..ff>/head.json -cells/v1/apps//catalog/objects/.json -cells/v1/apps//cells//control.json -cells/v1/apps//cells//inc//objects/. -cells/v1/apps//pins/.json -cells/v1/apps//pins/objects/.json -cells/v1/nodes/.json -``` - -Path IDs use fixed-width lowercase hexadecimal. Immutable kinds are `ltx`, `index`, `dir`, `root`, and `bundle`. - -`identity.json` strict-creates the tenant and application identity for one configured storage root. Concurrent initializers may adopt only the exact same winner. - -## Treat control as the only mutable Cell authority - -`control.json` identifies the owner and exact durable root. Its body is strict canonical JSON with an 8 KiB limit. - -| Field | Contract | -| --- | --- | -| `version` | Integer `1` | -| `cell` | 64 lowercase hexadecimal characters; matches the path | -| `incarnation` | 32 lowercase hexadecimal characters | -| `epoch` | Canonical decimal `u64`, at least `1` | -| `revision` | Canonical decimal `u64`, at least `1` | -| `progress` | Canonical decimal `u64` | -| `state` | `recovering`, `serving`, `idle`, or `tombstoned` | -| `owner` | `null` or session plus endpoint, bounded to 512 bytes | -| `root` | `null` or exact `RootRef` | -| `code` | Compiled module digest | -| `schema` | Positive `u32` | -| `next_due_ms` | `null` or nonnegative decimal `i64` | - -`RootRef` contains `digest`, `txid`, `checksum`, and `commit_sequence`. Native APIs add Cell and incarnation IDs so a reference cannot cross scopes. - -ETags are mutation tokens, not content hashes. Authority reads bypass caches. Every replacement validates the runtime transition table before calling conditional update. - -## Store roots as bounded immutable graphs - -One root identifies the complete SQLite state at one transaction ID. - -```mermaid -flowchart TD - Control[control.json
mutable CAS] - Root[root object
immutable] - SegPages[segment descriptor pages] - Bodies[LTX or bundle bodies] - Indexes[LTX indexes] - Directory[authenticated page directory] - - Control -->|digest| Root - Root --> SegPages - SegPages --> Bodies - SegPages --> Indexes - Root --> Directory -``` - -The root has a 32 KiB limit and names at most 64 segment-page digests. Each segment page has a 64 KiB limit and at most 96 descriptors. The full graph allows at most 4,096 segment descriptors. - -The graph obeys these checks: - -- The first segment is a full snapshot -- Later transaction ranges are contiguous through the root transaction ID -- Pre and post checksums connect every segment -- Every body and index digest matches downloaded bytes -- Offset and length arithmetic uses checked operations -- Bundle descriptors identify one exact extent with no fallback location -- The root's sequence and schema match SQLite `sys_meta` - -The owner marks routine compaction due after eight appends and starts one -promotion only after 250 ms without Cell work, with no queued command or -publication, and with a live lease. The actor retains the exclusive publisher -token and tracks the maintenance effect through drain. A command arriving -during maintenance waits under the normal queue and byte limits. If the next -append would reach 32 descriptors, or a lower configured limit, compaction -runs on the publication path before that append. The 4,096-descriptor graph -limit and existing byte-limit/full-compaction fallback remain unchanged. - -## Locate pages with an authenticated radix tree - -The page directory avoids a resident locator map for large databases. Leaves cover 256 SQLite page numbers; branches have fanout 256. - -```mermaid -flowchart TD - R[Root directory digest] - B0[Branch 0] - B1[Branch 1] - L0[Leaf pages 1 to 256] - L1[Leaf pages 257 to 512] - L2[Leaf pages 65,537 to 65,792] - - R --> B0 - R --> B1 - B0 --> L0 - B0 --> L1 - B1 --> L2 -``` - -Each node starts with a 32-byte `CRBDIR01` header. A leaf record stores: - -| Value | Size | -| --- | ---: | -| Page number | 4 bytes | -| Physical object digest | 32 bytes | -| Absolute frame offset | 8 bytes | -| Frame length | 4 bytes | -| Frame BLAKE3 | 32 bytes | -| Decoded page checksum | 8 bytes | - -Branch records store page range, child digest, live-page count, and XOR checksum. Parent aggregates must equal their children. - -Initial construction k-way merges ordered indexes and uploads leaves as they become complete. Incremental publication rewrites only paths touched by changed or truncated pages. - -## Read sparse pages without blocking the SQL pool - -Sparse SQLite opens the exact root and materializes pages on demand. A dedicated page-I/O worker performs object-store reads, so SQL workers may wait without consuming the same executor needed to satisfy their fault. - -The read path: - -1. Resolve the leaf through digest-pinned directory nodes -2. Coalesce adjacent frames up to 1 MiB -3. Read the exact object range -4. Verify the frame BLAKE3 and page checksum -5. Materialize the page through the injected filesystem -6. Charge the page once to the shared disk budget - -Missing allocated pages are corruption. The runtime never converts them to zero-filled application data. - -## Share one local disk budget - -`DiskBudget` and `DiskReservation` account every local byte on the configured volume. -When a `CellRuntime` owns the host, its ledger installs a reconciliation -admission on that budget. Existing bytes are imported at startup and later -reserve, resize, release, and late-host paths update the same ledger, so the -runtime's advertised disk usage cannot drift from LTX's local admission. - -```mermaid -flowchart LR - Budget[Node DiskBudget] - Main[SQLite main files] - Wal[WAL and retained LTX] - Sparse[Sparse pages] - Restore[Restore scratch] - Http[Git, LFS, Release staging] - - Budget --> Main - Budget --> Wal - Budget --> Sparse - Budget --> Restore - Budget --> Http -``` - -Managed writes reserve twice the maximum capture size before SQLite starts. After capture or checkpoint, the runtime reconciles that reservation to measured main, WAL, and retained-LTX bytes. - -Pre-transaction capacity rejection is retryable and does not fence the Cell. A failure after SQLite starts retains conservative admission until the handle closes. - -## Restore an exact root atomically - -Full restore reserves the destination database bytes before remote reads. Sparse takeover begins with zero materialized pages and charges each fault. - -Full restore follows this procedure: - -1. Reject an existing destination or SQLite sidecar before download -2. Create an exclusive private scratch file in the destination directory -3. Stream verified adjacent frame runs capped at 1 MiB -4. Reduce page checksums independently to the root checksum -5. Verify final length and synchronize the scratch file -6. Link the scratch file into the absent destination -7. Synchronize the parent directory -8. Remove the private scratch name - -Cancellation and verification failure remove only the scratch file owned by that attempt. The restore path never replaces an existing destination. - -## Compact without changing logical state - -Compaction is a representation-only publication. It preserves transaction ID, checksum, commit sequence, schema, due summary, and database endpoint. - -The implementation externally merges authenticated index streams by page number. It reads bounded frame ranges and uploads scratch-backed output without retaining a whole database or LTX body in memory. - -Scheduled compaction promotes eight or more contiguous inputs from one level. -A quiet-period attempt publishes at most one prepared root through the same -authority CAS used by commands. A retryable preparation failure retains the -debt for a bounded retry; ambiguous CAS is reconciled against the exact root. -Admission pressure may run the compaction cascade or force a full level-nine -replacement before the next append. - -## Catalog Cells before creating control - -Each tenant catalog has 256 shards selected by the first Cell-ID byte. Head paths include the tenant because entry identity validation is tenant scoped; immutable pages remain content addressed within the application. Each shard head names at most 256 immutable pages, and each page contains at most 256 sorted entries. - -Tenant-scoped catalog heads do not make release or backup management multi-tenant. -Offline retention marks one tenant and sweeps the application prefix, so it -rejects a root containing another tenant's catalog before deleting any objects. -This unreleased head layout replaces application-only heads; development roots -using the old layout require reprovisioning. There is no fallback reader. - - -A version-two head is the shard locator as well as the page list: every page -appears with its digest and the first Cell ID it can contain. Routing -binary-searches those keys and reads one page. Without the locator a lookup -downloads every page in the shard, so cold routing cost would grow with the -Cell population. Version-one heads, which carried digests only, are not read. - -The reader still verifies what it reads: the page digest, that the page opens -at its located first Cell ID, that every entry stays inside the shard and in -order, and that the page ends below the next locator key. A page that -disagrees with its locator is a hard error rather than a reported absence. -The locator itself is trusted, because only provisioning writes a head and it -does so through the head CAS; an object store that loses or rewrites a head is -a storage fault, not a routing input. A shard head is at most 64 KiB, which -holds 256 locator pairs. - -An entry stores: - -- Cell ID -- Namespace ID -- Partition bytes -- Role -- Initial code digest -- Initial schema version - -The per-shard ceiling is 65,536 entries. Provisioning reads the complete -shard, uploads the immutable catalog pages, and CASes its head before creating -`control.json`. A crash may leave an unused catalog entry, but never an -unproven mutable Cell. - -`CellAuthority::create_initial` requires a verified `CatalogProof`. Readers recompute every Cell ID and enforce ordering across page boundaries. - -## Pin one exact application backup boundary - -A backup pin is an immutable application-wide recovery root. Creation observes -all 256 catalog heads before traversing their pages, then binds the exact set of -cataloged Cells to one canonical control per Cell. - -```mermaid -flowchart LR - Pin[Pin pointer] - Release[Release snapshot] - Shards[Catalog shard manifests] - Controls[Canonical controls] - Roots[Verified LTX graphs] - - Pin --> Release - Pin --> Shards - Shards --> Controls - Controls --> Roots -``` - -The pin stores the application identity, creation time, all catalog revisions, -the release-snapshot digest, control count, and nonempty shard manifests. The -release snapshot contains the canonical release record and the exact descriptor -digests selected by it. Control pages and manifests are content addressed under -`pins/objects/`; the pin pointer is strict-created last. - -Creation and verification fail closed when: - -- a catalog page, release descriptor, control page, root object, LTX body, - index, bundle, or directory node is absent or has the wrong digest; -- catalog membership and captured controls differ; -- a control crosses its catalog shard or controls are not globally ordered; -- the pin ID already identifies a different canonical body. - -Repeating creation with an existing pin ID reopens and verifies the existing -boundary. Restore first verifies the full source pin, then copies immutable -objects to another prefix in the same bucket with create-if-absent semantics. -It independently verifies the destination graph before publishing unowned -`Idle` controls, exact catalog heads, the ready release record, and finally the -pin pointer. - -```mermaid -flowchart LR - Verify[Verify source pin] - Copy[Conditionally copy immutable graph] - Recheck[Verify destination graph] - Authority[Create Idle controls and catalog heads] - Commit[Create release and pin pointers] - - Verify --> Copy --> Recheck --> Authority --> Commit -``` - -This ordering makes an interrupted offline restore resumable and keeps stale -source node sessions out of the new authority root. A destination that has -divergent identity, controls, catalog heads, release selection, or immutable -bytes fails closed. Cross-provider archive export remains a separate service -operation. - -## Collect unreachable immutable objects behind maintenance - -Collection is an explicit maintenance activation, never a background request -handler. The release first enters `Maintenance`, normal nodes drain, and one -signed zero-capacity executor becomes the only NodeDirectory member. Backup -creation also holds a zero-capacity advertisement for its complete operation, -so maintenance either waits for an in-flight pin or fences a later creator at -its second `Ready` check. - -```mermaid -flowchart LR - Fence[Release = Maintenance] - Drain[Drain nodes and backup creators] - Mark[Verify and mark live roots] - List[Stream application objects] - Sweep[Delete old unreachable V1 objects] - Ready[Release = Ready] - - Fence --> Drain --> Mark --> List --> Sweep --> Ready -``` - -The mark phase fails before deletion unless it can authenticate: - -- current and desired release descriptors; -- every current catalog page and non-tombstoned control root; -- every retained pin's release, catalog, control pages, and LTX graph; and -- the absence of an owner on every current control. - -Reachable paths live in a temporary SQLite `WITHOUT ROWID` table on bounded -local scratch storage. Remote inventory is consumed as a stream, and candidate -lookups use batches of 256 paths. The collector deletes only recognized V1 -content-addressed release, catalog, pin, and Cell-incarnation object paths. -Mutable authority and unknown future layouts are never candidates. - -Deletion also requires the provider object's modification time to be older than -the configured grace. One pass deletes at most 100,000 objects; the server -defaults to 10,000 when collection is requested. Reaching the selected bound -leaves the release in `Maintenance`. Repeating the same activation resumes from -a new verified mark scan, so writes never reopen between partial passes. - -## Preserve storage verification invariants - -Storage changes must preserve these conditions: - -- Control is the only mutable owner and root authority -- Immutable bytes are verified before decoding or execution -- A root is scoped to one Cell and incarnation -- Full and sparse restore produce the same verified SQLite state -- Destination admission runs before downloading restore data -- Compaction changes representation, never logical position -- Local files are caches and cannot override object-store authority -- Provider construction, credentials, and HTTP policy remain outside `crab-ltx` diff --git a/crates/crab-cell-runtime/docs/validate.mjs b/crates/crab-cell-runtime/docs/validate.mjs deleted file mode 100644 index 9eec1d280..000000000 --- a/crates/crab-cell-runtime/docs/validate.mjs +++ /dev/null @@ -1,149 +0,0 @@ -import assert from 'node:assert/strict'; -import { spawnSync } from 'node:child_process'; -import { mkdtempSync, readFileSync, readdirSync, existsSync, rmSync } from 'node:fs'; -import { tmpdir } from 'node:os'; -import path from 'node:path'; -import { fileURLToPath } from 'node:url'; - -// Validate design inputs only; this does not exercise an embedded runtime implementation. -const root = path.dirname(fileURLToPath(import.meta.url)); -const contracts = path.join(root, 'contracts'); -const runtime = readFileSync(path.join(contracts, 'runtime.sql'), 'utf8'); -let assertions = 0; - -function sqlite(schema, operation, expected = '') { - const result = spawnSync('sqlite3', ['-batch', '-bail', ':memory:'], { - input: runtime + '\n' + schema + '\n' + operation, - encoding: 'utf8', - }); - if (result.error) throw result.error; - assert.equal(result.status, 0, result.stderr); - assert.equal(result.stdout.trim(), expected); - assertions++; -} - -function rejects(schema, operation, reason) { - const result = spawnSync('sqlite3', ['-batch', '-bail', ':memory:'], { - input: runtime + '\n' + schema + '\n' + operation, - encoding: 'utf8', - }); - if (result.error) throw result.error; - assert.notEqual(result.status, 0); - assert.match(result.stderr, reason); - assertions++; -} - -const kv = readFileSync(path.join(contracts, 'kv.sql'), 'utf8'); -const queue = readFileSync(path.join(contracts, 'queue.sql'), 'utf8'); -const workflow = readFileSync(path.join(contracts, 'workflow.sql'), 'utf8'); -const blob = readFileSync(path.join(contracts, 'blob.sql'), 'utf8'); -const cron = readFileSync(path.join(contracts, 'cron.sql'), 'utf8'); -for (const schema of ['', kv, queue, workflow, blob, cron]) sqlite(schema, 'PRAGMA integrity_check;', 'ok'); - -rejects('', `INSERT INTO sys_meta VALUES(1, zeroblob(31), zeroblob(16), 0, 0, 1);`, /CHECK constraint/); -rejects(kv, `INSERT INTO kv_entries VALUES(X'', X'01', zeroblob(27), X'', NULL);`, /CHECK constraint/); -rejects(kv, `INSERT INTO kv_entries VALUES(X'', X'01', zeroblob(28), zeroblob(4194305), NULL);`, /CHECK constraint/); - -const leased = `INSERT INTO queue_messages VALUES(zeroblob(16), X'01', 1, 1, 0, 1000, zeroblob(16), 100, NULL, NULL);`; -rejects(queue, `INSERT INTO queue_messages VALUES(zeroblob(16), X'01', 1, 1, 0, 1000, NULL, 100, NULL, NULL);`, /CHECK constraint/); -rejects(queue, leased + `UPDATE queue_messages SET state=2;`, /CHECK constraint/); -rejects(queue, `INSERT INTO queue_messages VALUES(zeroblob(16), X'01', 2, 1, 0, 1000, NULL, NULL, NULL, zeroblob(32));`, /CHECK constraint/); -rejects(queue, `INSERT INTO queue_messages VALUES(zeroblob(16), X'01', 3, 1, 0, 1000, NULL, NULL, NULL, zeroblob(31));`, /CHECK constraint/); -sqlite(queue, leased + ` -UPDATE queue_messages SET state=2, token=NULL, lease_until_ms=NULL -WHERE message_id=zeroblob(16) AND state=1 AND token=X'01' AND lease_until_ms>50; -SELECT changes(); -UPDATE queue_messages SET state=2, token=NULL, lease_until_ms=NULL -WHERE message_id=zeroblob(16) AND state=1 AND token=zeroblob(16) AND lease_until_ms>50; -SELECT changes();`, '0\n1'); - -const run = `INSERT INTO workflow_runs VALUES(X'01', zeroblob(16), zeroblob(32), 0, X'', 0, NULL, NULL);`; -rejects(workflow, run + ` -INSERT INTO workflow_events VALUES(zeroblob(16), 1, zeroblob(32), zeroblob(32), X''); -INSERT INTO workflow_events VALUES(zeroblob(16), 2, zeroblob(32), zeroblob(32), X'');`, /UNIQUE constraint/); -rejects(workflow, `INSERT INTO workflow_events VALUES(zeroblob(16), 1, zeroblob(32), zeroblob(32), X'');`, /FOREIGN KEY constraint/); -const activity = `INSERT INTO workflow_activities VALUES(zeroblob(16), zeroblob(16), 'build', X'', 0, 0, 0, 1000, NULL, NULL, NULL, NULL, NULL);`; -rejects(workflow, run + activity + `UPDATE workflow_activities SET completion_token=zeroblob(16);`, /CHECK constraint/); -rejects(workflow, run + activity + `UPDATE workflow_activities SET state=1, lease_until_ms=100;`, /CHECK constraint/); -rejects('', `INSERT INTO sys_effects VALUES(zeroblob(32), zeroblob(32), X'', 1, 1, 0, 1000, NULL, 100, 1, NULL);`, /CHECK constraint/); -sqlite(workflow, run + `UPDATE workflow_runs SET status=4; SELECT status FROM workflow_runs;`, '4'); - -const upload = `INSERT INTO blob_uploads VALUES(zeroblob(16), X'01', zeroblob(32), 0, NULL, NULL, X'', 0, 60000, 0, NULL, 0, 0);`; -rejects(blob, upload + `INSERT INTO blob_parts VALUES(zeroblob(16), 1, zeroblob(32), 262145, NULL);`, /CHECK constraint/); -rejects(blob, `INSERT INTO blob_parts VALUES(zeroblob(16), 1, zeroblob(32), 0, NULL);`, /FOREIGN KEY constraint/); -rejects(cron, `INSERT INTO cron_schedules VALUES(zeroblob(16), 0, X'', X'', 999, 0, 0, 1, 1, 0);`, /CHECK constraint/); - -// Only message contracts are intended: product HTTP APIs remain in Crab. -const peer = readFileSync(path.join(contracts, 'peer.proto'), 'utf8'); -assert(!/^\s*service\s/m.test(peer), 'private contract must not generate a public service'); -assert(!/\b(?:WorkflowDecision|WorkflowAction)\b/.test(peer), 'native decisions do not cross peer RPC'); -assertions += 2; - -const scratch = mkdtempSync(path.join(tmpdir(), 'crab-cell-contracts-')); -try { - const compile = spawnSync('protoc', [ - `--proto_path=${contracts}`, - `--descriptor_set_out=${path.join(scratch, 'peer.pb')}`, - 'peer.proto', - ], { encoding: 'utf8' }); - if (compile.error) throw compile.error; - assert.equal(compile.status, 0, compile.stderr); - assertions++; - const target = 'target { tenant_id: "tenant0000000001" application_id: "app0000000000001" namespace_id: "sql0000000000001" partition: "p" }'; - const identity = 'identity { request_id: "1234567890123456" incarnation: "abcdefghijklmnop" issued_at_ms: 1 expires_at_ms: 2 }'; - const expected = 'expected { cell_id: "cccccccccccccccccccccccccccccccc" incarnation: "abcdefghijklmnop" code: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" schema: 1 }'; - // Serialization fixtures only; they do not prove authentication or signature validation. - const fixtures = [ - { - type: 'PeerRequest', - input: 'version: 1 hop_count: 1 remaining_ms: 1000 authorization { origin_session: "session000000001" actions: "comment.write" } mutate {' - + target + identity + expected + 'cell_command { command_id: 17 codec_version: 2 input: "comment" } }', - expected: /expected \{\s+cell_id: "cccccccccccccccccccccccccccccccc"\s+incarnation: "abcdefghijklmnop"\s+code: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"\s+schema: 1\s+\}\s+cell_command \{\s+command_id: 17\s+codec_version: 2\s+input: "comment"/, - }, - { - type: 'MigrationRequest', - input: target - + 'incarnation: "abcdefghijklmnop" from_code: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" from_schema: 1 ' - + 'to_code: "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" to_schema: 2', - expected: /from_schema: 1\s+to_code: "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"\s+to_schema: 2/, - }, - ]; - for (const fixture of fixtures) { - const type = 'crab.cell.peer.v1.' + fixture.type; - const encoded = spawnSync('protoc', [ - '--proto_path=' + contracts, '--encode=' + type, 'peer.proto', - ], { input: fixture.input }); - if (encoded.error) throw encoded.error; - assert.equal(encoded.status, 0, encoded.stderr.toString()); - const decoded = spawnSync('protoc', [ - '--proto_path=' + contracts, '--decode=' + type, 'peer.proto', - ], { input: encoded.stdout, encoding: 'utf8' }); - if (decoded.error) throw decoded.error; - assert.equal(decoded.status, 0, decoded.stderr); - assert.match(decoded.stdout, fixture.expected); - assertions++; - } -} finally { - // Only the unique directory created by this validation run is removed. - rmSync(scratch, { recursive: true }); -} - -let links = 0; -for (const name of readdirSync(root).filter(name => name.endsWith('.md'))) { - const text = readFileSync(path.join(root, name), 'utf8'); - assert.equal((text.match(/^```/gm) || []).length % 2, 0, `${name}: unbalanced fences`); - assert(!/[\t ]+$/m.test(text), `${name}: trailing whitespace`); - for (const match of text.matchAll(/\[[^\]]*\]\(([^)]+)\)/g)) { - if (/^https?:/.test(match[1])) continue; - const [file, anchor] = match[1].split('#'); - const target = path.resolve(root, file || name); - assert(existsSync(target), `${name}: missing ${match[1]}`); - if (anchor && target.endsWith('.md')) { - const headings = [...readFileSync(target, 'utf8').matchAll(/^#+\s+(.+)$/gm)] - .map(m => m[1].toLowerCase().replace(/[^\p{L}\p{N}_\-\s]/gu, '').replace(/ /g, '-')); - assert(headings.includes(anchor), `${name}: missing anchor ${match[1]}`); - } - links++; - } -} -console.log(`Passed ${assertions} schema/protocol assertions and ${links} local links; Markdown fences/whitespace valid.`); diff --git a/crates/crab-cell-runtime/docs/vfs-ltx-scale-plan.md b/crates/crab-cell-runtime/docs/vfs-ltx-scale-plan.md deleted file mode 100644 index 54e26a847..000000000 --- a/crates/crab-cell-runtime/docs/vfs-ltx-scale-plan.md +++ /dev/null @@ -1,558 +0,0 @@ -# Execute the SQLite VFS and LTX Cell scaling plan - -Crab already runs SQLite locally over a verified sparse VFS and captures LTX -from its WAL. This plan hardens the request path around that implementation, -qualifies cold recovery and durability, and proves an application made of many -Cells. It does not introduce a shared writable SQLite file or another -replication protocol. - -| Document intent | Value | -| --- | --- | -| Content type | Technical design and executable delivery plan | -| Audience | Runtime, HTTP, application, and qualification contributors | -| Status | In progress; routing and LTX improvements plus response attribution implemented; scale qualification remains open | -| Decision | Fix owner routing first; change pager or durability policy only after phase-specific evidence | - -[Back to the Cell runtime index](README.md) - -## Use the existing implementation - -| Responsibility | Existing owner | Contract to preserve | -| --- | --- | --- | -| Product authentication, authorization, and ingress | [HTTP router](../../crab-http-server/src/cells/router.rs) and [peer receiver](../../crab-http-server/src/peer.rs) | Authorize the target and action before dispatch; bound peer hops | -| Cell identity, owner, epoch, lifecycle, root | [Cell authority](../src/control/authority.rs) | Only conditional control writes grant or change ownership | -| Actor admission and one SQL writer | [Cell actor](../src/cell/actor.rs) | A fenced or draining actor refuses queued and new work | -| Local SQLite, WAL capture, LTX | [Db](../../crab-ltx/src/db.rs) | Local commit alone never releases a response | -| Sparse exact-root page access | [Writable VFS](../../crab-ltx/src/writable_vfs.rs) and [paged I/O](../../crab-ltx/src/paged_io.rs) | Verify inherited pages, reserve disk, and create a fresh local file | -| Durable acknowledgement | [Follower design](failover-and-followers.md) and [publication](../src/publication.rs) | A response follows object-root proof or the selected followers' fsync proof | -| Application-level cross-Cell work | [Application framework](application-framework.md) | One Cell transaction; durable effects and idempotent inboxes across Cells | - -The request and recovery paths should remain: - -```text -client -> gateway -> entry node -> owner hint or authoritative slow path - -> authenticated peer hop if needed -> actor admission -> local SQLite - -> LTX -> follower fsync proof OR immutable root + authority CAS - -> response - -owner loss -> seal/recover acknowledged follower tail -> exact root - -> fresh sparse SQLite file -> verified page faults -> resident Cell -``` - -A route hint, placement plan, or cached page is never Cell authority. A -follower log stores recent LTX; it does not serve SQL. A durable response may -precede object publication only when the selected follower proof covers it. -The [canonical scaling contract](canonical-ltx-scaling.md) now includes explicit -read-only replicas in object durability mode. [Plan 036](../../../advisor-plans/036-cell-read-replicas-and-fenced-promotion.md) -keeps their policy, receipt and response authority checks separate from owner -reads. Qualify replica routing, refresh and promotion against the same resource -budget; earlier owner-only measurements do not establish replica performance. - -## Baseline the current tree - -The [20-node gateway record](../../crab-http-server/deploy/cell-issue-fleet/qualification/2026-09-25-gateway-load.md) -is the comparison baseline, not a supported limit. In its fixed-load phase, -385 of 400 requests entered nonowners, 23 HTTP 503 attempts were retried, -read p95 was 389.089 ms, and write p95 was 783.251 ms. The observed 56.22 -logical requests/s is not saturation throughput. The three-node public-host -[RustFS action record](../../crab-cell-app/performance/2026-09-25-public-host-rustfs.md) -isolates local and forwarded actions but does not exercise the 20 Compose -nodes. Both runs used one machine. - -Run from the repository root. Choose a fresh project and state directory for -each comparison run. This disposable stack uses local RustFS credentials -`crab/crab`; never expose it outside the local Docker network. The -`qualify.py` command builds the image, performs the 3 -> 5 -> 10 -> 20 -functional scale check with a fixed 20 Cells, and with `--load-stages` measures -scheduled arrivals while each stage has exactly that many active nodes. Keep all raw JSON and logs outside -the checkout. - -```sh -state="$HOME/.codex/cell-vfs-ltx-scale/$(date +%Y%m%d-%H%M%S)" -project="crab-cell-issue-vfs-ltx-$(date +%s)" -python3 crates/crab-http-server/deploy/cell-issue-fleet/qualify.py \ - --state "$state" --project "$project" --load-stages -``` - -The command writes `report.json` and `load-3-stage.json`, -`load-5-stage.json`, `load-10-stage.json`, and `load-20-stage.json`. -`load.py` rejects a stage name that does not match the project's active -node containers. The [local stage-load record](../../crab-http-server/deploy/cell-issue-fleet/qualification/2026-09-25-stage-load.md) -captures one completed run. Omit `--load-stages` for the original functional -check. -That historical run used completion-paced lanes and one Cell per node. The -current runner keeps Cell count and offered rate independent of node count; -use `--cells`, `--load-rate`, `--load-duration`, and `--load-max-in-flight` to -hold the comparison workload fixed. Raw pair samples and resource/metrics -snapshots accompany each summary. Scheduled and historical rates must not be -compared as the same workload. -For a fast syntax-only check before building images: - -```sh -state="$HOME/.codex/cell-vfs-ltx-scale/render-check" -python3 crates/crab-http-server/deploy/cell-issue-fleet/render.py \ - --state "$state" --project crab-cell-issue-vfs-ltx-render -docker compose --file "$state/compose.yaml" \ - --profile five --profile ten --profile twenty config --quiet -``` - -For local Rust tests, first verify `$HOME/Workspace` is mounted and writable. -Use a Cargo target directory dedicated to this worktree, for example -`$HOME/Workspace/crabbuild-target/crab-8bc8`; set -`CARGO_TARGET_DIR` on every Cargo invocation. The existing RustFS action -test requires a running Compose stack and its bucket: - -```sh -CRAB_CELL_TEST_BUCKET=crab-cell-issue-fleet \ -CRAB_CELL_TEST_ENDPOINT=http://127.0.0.1:19010 \ -CRAB_CELL_TEST_PREFIX=reference-performance \ -AWS_ACCESS_KEY_ID=crab AWS_SECRET_ACCESS_KEY=crab \ -CRAB_CELL_PERF_ITERATIONS=100 \ -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-cell-app --test reference_application \ - public_host::reference_public_host_rustfs_action_performance \ - --release --locked -- --ignored --nocapture -``` - -This action test runs three application hosts in the test process against -RustFS; it is a different topology from the Compose fleet. Its result must -be labeled separately in any comparison. - -## Work packet 1: make the qualification comparable - -**Change:** extend the existing Compose qualifier to run the current -`load.py` workload at 3, 5, 10, and 20 nodes immediately after each -stage becomes healthy. Keep the existing command as the default functional -check. Save one raw load report per offered rate in each stage with source revision, image digest, -Compose profile, node limits, RustFS identity, workload size, and retry counts. -The runner must not silently retry a mutation with a new request ID. - -**Files:** [qualify.py](../../crab-http-server/deploy/cell-issue-fleet/qualify.py), -[load.py](../../crab-http-server/deploy/cell-issue-fleet/load.py), and the -[Compose example guide](../../crab-http-server/deploy/cell-issue-fleet/README.md). - -**Proof:** each stage uses exactly its active nodes and Cells; the gateway -entry histogram remains within the existing 70-130% even-split check; every -write is read back; each Cell advances its RustFS root; the owner-loss case -recovers the last acknowledged issue. Record p50/p95/p99 and max for reads, -writes, and recovery, plus retry and error counts. Do not set a production -latency or Cell-count limit from one shared-host Compose run. - -**Exit:** four stage groups retain their ordered rate reports and can be -compared to a second run with the same profile. A report with retries is valid -evidence but is not an error-free service result. - -The qualifier now builds a clean committed Git archive, verifies the image's -revision label even with `--skip-build`, and pins every server service to the -inspected image ID. Running containers must match that ID. Load reports record -server revision/platform separately from generator source; imported images -still require their CI source/checksum receipts. Wrong-source refusal and real -Docker archive/tag-retention proof pass. Four current-source stage curves and -a repeated 20-node run are still required. - -The runner now accepts ordered `--load-rate 5 20 50 5` points, retaining each -point's samples, traces, resource observations, publication drain and owner-loss -checks. Verified overload returns exit 2 with `passed: false`, allowing later -controls to run; failed integrity or recovery stops the sequence. CI retains -overload as capacity evidence and separately exercises unpublished-tail loss. -Its completion status is not proof that every offered rate was served. The -repeated control can reveal drift as data accumulates; it does not reset the -database to its initial state. These curves are implemented but still require -current-image execution before satisfying the exit gate. - -## Work packet 2: remove redundant peer resolution - -The entry [router](../../crab-http-server/src/cells/router.rs) checks for an -actor-owned resident handle, then on a miss reads catalog, exact control, -and live-owner state. On the receiving node, -[peer handling](../../crab-http-server/src/peer.rs) resolves the target to -decide whether to activate or forward, then -[dispatch](../src/peer/dispatch.rs) resolves it again before execution. -This duplicates object-store metadata work on the common forwarded path. - -**Change:** make the receiver's one resolution available to dispatch through -a private, request-scoped result. The result may carry only a local handle -that still passes actor admission, or a typed not-local outcome. Preserve -peer signature verification, product reauthorization, tenant/application -target checks, remaining deadline, activation policy, and the existing hop -limit. If migration races with dispatch, the actor must refuse or fence the -request; the receiver must not execute using stale authority. - -**Proof:** an instrumented store observes one receiver resolution rather -than two on a forwarded read; owner-local resident reads still make zero -object-store calls; malformed/unauthorized/stale peer requests fail closed; -takeover, drain, and compatible rollout tests retain their existing outcomes. -Use the current [public takeover test](../../crab-http-server/tests/public_cell_takeover.rs) -and [runtime residency tests](../tests/runtime/lifecycle/residency.rs) -as entry points, then add one targeted receiver test for metadata calls and -fencing. Do not add a second dispatch path for compatibility. - -**Exit:** identical results under owner loss and at least one fewer catalog -head, catalog page, and control read per healthy forwarded request. - -The receiver now passes its verified local handle into the canonical dispatch -path. The mTLS public-host test counts one catalog head, one page, and one -control read for each peer operation, and continues through owner loss over -both in-memory storage and local RustFS. The first receiver change still left -two peer operations per public comment read: `Describe` followed by `Query`. -The direct dispatcher test proves it does not resolve again when supplied a -handle. - -The sender has a separate cost: `PeerHttpRoundTrip::send_inner` calls -`owner()` for each operation, and `owner()` reads exact control and the signed -node advertisement. A router-only hint would leave those reads in place. -Count sender control/directory reads and peer round trips per public action -before changing this path; the receiver counts above do not include them. - -The routed client now receives the exact description already read from Cell -control and skips `Describe`. Its signed command/query/resolve payload carries -the expected Cell ID, incarnation, code, and schema. The owner compares those -fields before execution; actor fencing and product authorization still apply. -Focused tests reject each stale field, retain signature binding, and preserve -mutation digest, duplicate delivery, and unknown-result evidence. Clients -without an observed route still obtain a description through their transport. -The public comment-read test now proves one peer operation and one receiver -catalog/control resolution through HTTP and mTLS over both in-memory storage -and real RustFS, including owner loss and restored application data. Fourteen -typed client/effect tests and the three public-host ambiguity, recovery, and -compatible-rollout cases pass. Current-image fleet latency qualification and -packet 3's entry-router reads remain open. - -## Work packet 3: give ingress a bounded owner hint - -After packet 2, measure whether entry-node metadata remains the dominant -forwarded cost. If so, keep one short-lived, process-local owner observation -per Cell target in `crab-http-server`. Populate it only from an authenticated -successful slow-path resolution and a live signed node advertisement. Use it -only to choose the peer destination. The receiver still authorizes and admits -the request; the Cell authority CAS remains decisive. A hint must never -create, transfer, or renew ownership. - -The observation must be consumed by the outgoing `PeerHttpRoundTrip` path as -well as the entry router. Today `send_inner` reloads control and the node -advertisement for every `Describe`, `Query`, and command; avoiding only the -router's first lookup would leave most forwarded metadata reads unchanged. - -The first sender slice now retains a process-local observation for at most -five seconds, never beyond the signed node lease minus one second, and caps it -at 4,096 Cells. A newer control revision cannot be replaced by a delayed older -read. A refused or ambiguous peer attempt invalidates only the session it -used; the existing single authoritative retry retains the original deadline -and signed request bytes. The mTLS typed-query test observes one sender control -read for `Describe` plus `Query` on both in-memory storage and RustFS, then -checks invalidation after owner loss. At that committed slice, the entry router -still reads catalog and control on each forwarded action. - -The routing/admission follow-up shares the sender's observation with the -entry router. Warm public reads now pass the zero-entry-catalog/control-read -assertion while retaining receiver resolution. A five-caller expiry test -exposed CPU admission held across enrollment I/O; splitting that I/O and -waiting for codec capacity passes the functional regression in memory and -against real RustFS. [Audit finding 15](ltx-performance-audit.md#15-peer-verification-couples-provider-latency-to-scarce-cpu-admission) -records passing deadline, cancellation, shutdown, accounting, and enrollment- -expiry tests. One received budget now covers resolution, activation, dispatch -waiting, and reply encoding. Delayed owner resolution returns 504 without -executing the query in memory and RustFS tests. Activation-delay injection and -the p95 exit gate below remain open; this follow-up is not capacity qualification. - -Set a fixed maximum lifetime no longer than the observed node-session lease; -do not add a public config option. Invalidate on peer refusal, stale session, -target mismatch, release mismatch, and node-liveness loss. Retry the -authoritative slow path at most once within the original deadline. Keep -normal routing to one peer hop and the existing bounded stale-owner redirect -behavior. Cache keys and telemetry labels must not contain arbitrary Cell -IDs or tenant strings. - -**Proof:** a warm, healthy forwarded read performs zero entry-node -catalog/control object reads, while a stale hint routes through the slow -path or fails closed without executing on a fenced owner. Test deletion, -drain, owner loss, rollout, simultaneous hint expiry, and a peer that -returns an ambiguous transport result. Report cache hit/miss/stale counters -and end-to-end latency; never infer correctness from a cache hit. - -**Exit:** under the same 20-node workload, read and write p95 improve -against packet 1 without raising retries, 503s, or object-store requests -per logical action. If the measured benefit is absent, remove the hint -and keep packet 2's simpler routing path. - -## Work packet 4: qualify the existing pager and durability paths - -Do not replace the [writable VFS](../../crab-ltx/src/writable_vfs.rs). -Exercise cold open, page fault, background hydration, resident promotion, -idle eviction, and takeover over RustFS. A verified resident read has zero -object-store calls; sparse reads count their exact root/page requests. -Record p50/p95/p99, bytes, page I/O queue depth, disk reservations, and -time to first read. Reject corrupted page digests, a changed exact root, -insufficient disk, and a canceled hydration without reusing an unverified -mutable file. B-tree-guided speculative prefetch stays disabled until recorded scan traces -beat point reads under bounded provider latency, as required by the -[canonical scaling contract](canonical-ltx-scaling.md). -The current bridge already coalesces up to 64 pages per fault. Separately -qualify that window's bytes consumed versus prefetched for point queries, -random access, scans, and hydration; fewer calls alone do not prove lower cost -or latency. The [RustFS activation probe](../../crab-ltx/examples/README.md#rustfs-scale-workload) -now measures phase durations and read bytes at fixed database sizes and I/O -admission settings. Its three samples per setting are diagnostic; sustained -multi-Cell tails and first-mutation latency remain required. - -The [read-view RustFS audit](ltx-performance-audit.md#28-demand-read-ahead-fetches-a-cached-suffix-after-small-updates) -reproduces two additional costs: fresh immutable views refetch unchanged page -bodies, and a fragmented root's demand read-ahead refetches a cached suffix. -Demand misses now share hydration's uncached-prefix selection within the -existing window. Require no duplicated cached suffix in the regression, then compare -point/random/scan traffic and cache churn. Next evaluate authenticated frame -reuse across exact roots, retaining truncate/regrow and incarnation isolation. -Measure GETs, useful/fetched bytes, decode work and public-action p99 separately; -the observed 38% extra origin bytes do not establish a latency improvement. - -Include two Cells on the **same SQL worker** in the interference matrix. A -cold page wait, hydration step, or published-cut cleanup currently occupies -that worker; the independent provider driver does not let the resident Cell -execute meanwhile. Measure worker admission, shard queue, provider wait, and -proof-to-confirmation delay separately. Sweep cut sizes because published-cut -cleanup verifies every page before releasing retained accounting. Streaming -now removes its complete input buffer; the decoder also streams its footer and -omits unused replica lookup entries. The remaining observed-page index still -needs memory qualification. The [audit findings 9–11 and 13](ltx-performance-audit.md) -define the ownership constraints and focused failure tests for changing these -paths. The [RustFS cleanup comparison](../../crab-ltx/perf/README.md#streaming-published-cut-cleanup-2026-09-26) -measures this first change separately from public response latency. - -Worker admission now reserves capacity for each fixed SQL worker separately. -Queued jobs on a busy worker cannot occupy the slots of idle workers. The -focused interference regression preserves cancellation before dispatch, -retention after dispatch, and normal publication. This removes global permit -hoarding; same-worker sparse faults and maintenance still need the asynchronous -fetch and short installation experiment from audit finding 9. - -Hold database size and changed pages fixed while comparing fresh, sparse, -resident-after-hydration, and clean-resumed Cells. [Audit finding 14](ltx-performance-audit.md#14-checksum-bookkeeping-depends-on-activation-history-and-uses-tiny-file-io) -identifies a full checksum-array copy per cut in fresh sessions, individual -checksum file I/O in restored sessions, and per-page reads during clean handoff. -Measure checksum calls/bytes, allocations, eviction, and reactivation before -choosing a common bounded block strategy. Benchmark first mutation and repeated -updates in each state; a sparse-only capture result does not cover the fresh -path. Preserve failure fencing and exact continuation verification. - -For writes, attribute SQLite command time, LTX capture, follower append -and fsync, root preparation, object CAS, queue wait, and final proof source. -The runtime already supports follower and object proofs. Tune batching or -publication only after traces show which wait dominates. Preserve the -[acknowledgement and recovery implications](failover-and-followers.md#preserve-these-guarantees): -every released result is covered, and a successor seals/replays any -follower-only tail before serving. Inject owner process and local-disk loss, -one follower loss, RustFS delay/failure, and ambiguous CAS. - -**Proof commands for the existing focused suites:** - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-ltx --features replica --test host \ - hooks::activation --locked -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-ltx --features replica --test cell \ - roots::directory --locked -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-ltx --features replica --test cell \ - sparse_hydration_coalesces_contiguous_cell_frames --locked -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-cell-runtime --features test-support --test runtime \ - resident_route_reports_zero_origin_reads_and_latency_percentiles --locked -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-cell-runtime --features test-support --test runtime \ - restored_sparse_route_promotes_before_zero_origin_reads --locked -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-cell-runtime --features test-support --test runtime \ - follower_proofs_advance_logical_head_and_bound_the_object_backlog --locked -``` - -These focused tests are a local correctness gate. The RustFS action command -above and the protected qualification workflows remain separate gates. Do -not claim a recovery percentile from one owner-loss sample. - -### LTX audit gates before tuning - -The [source audit](ltx-performance-audit.md) adds concrete experiments in -execution order: small-body single PUT, zero-write disk-cache hits, bounded -parallel checksum-directory loading, and incremental compaction metadata. -Measure response proof and sustained drain alongside those changes. Keep -root coalescing and durability-mode changes behind their stronger recovery -gates. Qualify the runtime's 8/32-segment compaction boundaries and the -benchmark's periodic payload before using its cost rows as capacity evidence. - -The small-body and disk-cache changes are implemented with focused integrity, -retry, restore, and cache-lifecycle tests. Bodies up to 256 KiB use a verified -single PUT through one LTX transfer function; larger bodies retain streaming. -Disk-cache hits no longer rewrite unchanged membership. These changes still -need public-action and sustained-drain measurements before assigning a fleet -latency benefit. - -Writable activation now dispatches checksum-file operations through the host's -blocking-job admission. A file owner retains dirty admission through canceled -work and cleanup; focused tests cover slow I/O, failures, one-slot execution, -fresh destinations, and retry. Sibling leaf reads now overlap in an ordered -stream capped at eight reads under the shared I/O admission; internal branches -stay depth first. Delayed-provider tests prove overlap and the ceiling, while -corruption and multi-parent tests preserve ordered validation. The metadata -walk remains eager, and the larger activation and interference qualification -in packet 4 remains open. - -Checksum persistence now coalesces adjacent changed entries into writes capped -at 64 KiB; clean handoff buffers dense checksum reads to the same bound. Local -real-SQLite tests reproduce the old per-entry I/O and verify reduced host calls, -exact sidecar bytes, corruption refusal, and post-seal failure fencing. Fixed-size -fresh updates now reuse an exclusively owned dense memory index, restored old -checksum reads use a 4 KiB window, and successful merges release the changed-page -overlay allocation. Shared snapshots or growth can still allocate; activation -still walks the complete authenticated checksum directory. The fresh/restored -comparison, allocation and handoff measurements, and sibling-Cell latency gate in -[audit finding 14](ltx-performance-audit.md) remain open. - -The local [replica cost record](../../crab-ltx/perf/README.md#cell-publication-cost-per-command) -measures about 0.3 ms for a small sparse deferred capture, but 87–139 ms -at p50/p95 for a small successor-root preparation over loopback RustFS. -That preparation writes five immutable objects for a small command. These -measurements have different harnesses and are not an end-to-end latency -decomposition. The runtime serializes publication for each Cell, even when -a follower proof releases a response earlier. The current plan cannot yet -identify which portion of a hot Cell's latency or capacity belongs to LTX. - -Close these gaps in order, using the same reference action and issue-service -workloads before and after each change: - -| Priority | Missing evidence or pressure point | Experiment and acceptance gate | -| --- | --- | --- | -| 1 | The current durability counters report completed fleet and object proofs. An object publication can finish after a response was released on fleet proof, so those counters do not identify the proof that released the action. | Record one bounded-cardinality response-winner observation at `prove_command`, with queue, SQLite, capture, follower, publication, and confirmation times tied to the same action trace. Compare first action after node-log activation with steady actions. Prove one winner per successful action and preserve the existing ambiguous-result behavior. | -| 2 | Root preparation reports one duration and upload count. It does not distinguish predecessor root GET/HEAD, directory reads, immutable PUT, provider wait, and authority CAS. | Extend the RustFS cost harness or an instrumented store to count GET, HEAD, PUT, bytes, attempts, and time by finite phase. Run single and many-Cell concurrency, plus hot-Cell increasing offered load. Report publisher queue age, unpublished bytes, admission rejections, root lag, provider saturation, and sustainable published roots/s. A latency change passes only if throughput or p95/p99 improves without increasing failed proofs or provider pressure. | -| 3 | A cached predecessor still needs origin presence checks. `load_graph` HEAD-checks cached root metadata; a test requires the next prepare to fail when those objects disappear. | Measure the HEAD wave separately. Keep the [missing-metadata invariant](../../crab-ltx/tests/cell/roots/lifecycle.rs) while testing any reuse of verified predecessor state. Do not remove HEADs solely because metadata is cached or because PUT keys are content addressed. Compare chain lengths around 1, 96, and 97 descriptors before considering reuse of unchanged segment pages. | -| 4 | Tiny steady writes hide checkpoint, full-image fallback, and sparse activation tails. A capture can read the whole WAL or emit a full database image, while the shared paged I/O driver has a bounded request queue and jobs. | Run hot and skewed Cells across checkpoint and compaction thresholds, a pinned reader, large changed-page sets, and simultaneous cold opens. Attribute checkpoint runs/busy/restarts, full WAL reads, full-image bytes, page-fault queue and deadline errors, cache misses, provider I/O, and p99. Test with the 1 GiB node limit; retain exact-root and disk-admission fault tests. | -| 5 | SQLite uses `synchronous=FULL` on managed connections even though the Cell response waits for an external proof. The local sync might be visible in write latency, but no power-loss equivalence has been established. | Benchmark SQLite commit and response latency with the current mode as control. Consider a different mode only in an isolated experiment that proves no response can use an unverified local WAL or continuation after power loss, including failure before capture, after follower proof, and during object publication. Keep the current mode until crash and recovery qualification justifies a contract change. | - -Response-winner instrumentation is now wired through the final command/effect -reply boundary and exported as `crab_cell_command_responses_total`, -`crab_cell_command_response_seconds`, and -`crab_cell_command_confirmation_seconds`, with only `recorded|fleet|object` -source labels. Lost-response reconciliation, follower-first replies followed -by object publication, and canceled callers have focused assertions. Full -action phase attribution and sustained arrival-rate curves are still open. - -The replica's `cold` origin counter includes predecessor-graph reads made by -publication, so it cannot by itself measure cold activation. Add a finite -operation/phase distinction instead of Cell-ID labels. Keep the existing -64-page coalesced fault window and bounded caches while measuring cold-open -storms; raising global I/O concurrency or cache size without the 1 GiB -resource profile can make tail latency worse. The local `capture()` comparison -also measures a stronger file-and-directory barrier than the runtime's -`capture_deferred()` path; use the latter for a runtime optimization decision. - -Gate any root coalescing on ordered acknowledgements and replay: every -accepted command keeps its exact stable receipt; a successor reconstructs -every acknowledged follower-only tail; immutable roots remain a monotonic -prefix; and the pending publication limit stays bounded. A faster object -preparation that still cannot drain the offered hot-Cell rate is not a -supported throughput increase. - -## Work packet 5: prove application-level scale - -Use the existing [reference application](application-framework-example.md) -and [Cell-backed issue example](../../crab-http-server/deploy/cell-issue-fleet/README.md). -The issue example proves gateway distribution, owner routing, and one Cell -per repository. Extend the reference application workload to exercise -entity, shard, workflow, and read-model Cells through public -`CellNode`/application handles and stable typed operations. One command -changes one Cell. Cross-Cell changes flow through durable effects, -deduplicated inboxes, and idempotent activities; no global SQLite -transaction or follower SQL read is implied. - -The load matrix needs three shapes: many evenly distributed Cells, a -deliberately hot Cell, and skewed read/write Cells. At each of 3, 5, 10, -and 20 nodes, report per-node CPU, memory, disk, active Cells, queue depth, -object requests, and throughput curves over increasing client concurrency. -Separate owner-local, forwarded, sparse, resident, fleet-proof, and -object-proof actions. A fixed 20-lane rate is not the maximum throughput. -Every successful write must have a visible readback or durable receipt and -survive owner loss. Reject duplicate effects and regressions in published -root sequence. - -Keep the existing post-drain owner-loss smoke and add a fault during scheduled -arrivals. The current runner drains publication before killing the owner, so -it cannot qualify outstanding follower-only responses under load. Trigger the -new fault from a recorded fleet acknowledgement and an uncovered tail, keep -the required followers available, and verify every acknowledged request after -takeover. Measure unaffected Cells during the failure as well. The exact fault -budget and acceptance conditions are in -[audit finding 16](ltx-performance-audit.md#16-post-load-recovery-does-not-qualify-acknowledged-tails-during-load). - -Verify executing-owner distribution independently of gateway distribution. -Record owner/epoch changes, wait for the declared placement settling criterion, -and retain skewed stages as skewed evidence. The observed five-node smoke -owned 6/5/5/1/3 of its 20 Cells despite healthy routing. Record the Docker VM's -CPU/memory too; per-container limits alone do not provide independent resources -on an oversubscribed host. Current-source uniform capacity qualification must -meet [audit finding 12](ltx-performance-audit.md) before comparing stage rates. - -Use the existing public-host cases as the first application correctness gate: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-cell-app --test reference_application \ - three_node_host_resolves_ambiguous_result_and_deduplicates_delivery --locked -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-cell-app --test reference_application \ - three_node_host_recovers_published_state_after_owner_loss --locked -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-cell-app --test reference_application \ - three_node_host_additive_code_rollout_preserves_acknowledged_state --locked -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo test -p crab-http-server --test public_cell_takeover --locked -``` - -The reference application's [Compose smoke](../../crab-cell-app/PERFORMANCE.md#three-constrained-compose-nodes) -now runs three public hosts as separate 1-CPU/1-GiB containers against GA -RustFS, with signed peer requests, a round-robin ingress, generated-client -duplicate/readback proof, and renewable signed sessions withdrawn after drain. -The source-selected child test and Compose wrapper reject zero-test runs. -This seven-Cell, six-lane integration check is the starting point for this -packet; the larger application distribution, fault, and capacity matrix remains -open. - -The entity-ledger correctness gate additionally provisions twelve SQL entity -Cells across three public hosts. It verifies generated targeting, per-Cell -request deduplication, visible receipt-bound readback, and stored 4/4/4 ownership -through a signed load balancer and GA RustFS. Its hosts share one process; -the many-Cell process, workload-shape, resource, and capacity matrix above -remains open. See [the runnable gate](../../crab-cell-app/PERFORMANCE.md#entity-targeting-correctness). - -Run these before the scaled load, then repeat the relevant cases after any -change to routing, admission, VFS, or durability. Build and broad provider -proof belong in CI or a dedicated test environment. - -## Release decision - -The work is ready for a supported profile only when: - -1. All focused safety tests and docs validation pass; the changed HTTP, - runtime, LTX, and application surfaces pass their relevant CI gates. -2. Four stage reports and repeated same-profile 20-node runs show the - expected routing reduction, stable latency distributions, no unexplained - 503s, and no RustFS file-descriptor or I/O-queue exhaustion. -3. A multi-host run repeats owner loss, follower loss, object-store fault, - rolling release, and skewed load with isolated node disks and network - failure domains. One-host Compose does not meet this gate. -4. Signed qualification receipts bind source revision, image, node profile, - provider, workload, fault schedule, and raw artifacts as required by - [delivery and qualification](delivery.md). Supported limits and SLOs - come from those receipts, not from this plan. - -Stop a work packet and report the first failing invariant if a successful -result disappears after owner loss, a stale owner executes a command, a -resident read reaches object storage, a follower-only acknowledgement lacks -a recoverable tail, or a new route skips product authorization. Record the -failing request ID, Cell target, owner epoch, proof source, and raw report -outside the checkout; do not weaken a test or baseline to proceed. - -Run the documentation contract check after editing this file: - -```sh -node crates/crab-cell-runtime/docs/validate.mjs -``` diff --git a/crates/crab-cell-runtime/model/CellCoordination.cfg b/crates/crab-cell-runtime/model/CellCoordination.cfg deleted file mode 100644 index 1c6c9069e..000000000 --- a/crates/crab-cell-runtime/model/CellCoordination.cfg +++ /dev/null @@ -1,20 +0,0 @@ -SPECIFICATION Spec -CONSTANTS - Nodes = {"a", "b"} - Commands = {"c1", "c2"} - FaultMode = "none" -INVARIANTS - NoDualServing - AckHasDurability - RetainedHasOwner - OwnerIsLive - EpochIsNatural - PublishedIsNatural -PROPERTIES - EpochMonotonic - PublishedMonotonic - NoAdmissionAfterFence - ReleaseAfterDurability - RetainedRecoveryObligation - DrainOrRecoveryReady -CHECK_DEADLOCK FALSE diff --git a/crates/crab-cell-runtime/model/CellCoordination.tla b/crates/crab-cell-runtime/model/CellCoordination.tla deleted file mode 100644 index 063d8bea4..000000000 --- a/crates/crab-cell-runtime/model/CellCoordination.tla +++ /dev/null @@ -1,228 +0,0 @@ ------------------------------- MODULE CellCoordination ------------------------------ -EXTENDS Naturals, FiniteSets, TLC - -CONSTANTS Nodes, Commands, FaultMode -NoOwner == "none" - -VARIABLES owner, epoch, state, accepted, acknowledged, retained, published, - admitted, quiescing, recovery_required, released -vars == <> - -Serving == "serving" -Fenced == "fenced" -Idle == "idle" - -Init == - /\ owner = NoOwner - /\ epoch = 0 - /\ state = [n \in Nodes |-> Idle] - /\ accepted = [c \in Commands |-> FALSE] - /\ acknowledged = [c \in Commands |-> FALSE] - /\ retained = FALSE - /\ published = 0 - /\ admitted = [c \in Commands |-> FALSE] - /\ quiescing = FALSE - /\ recovery_required = FALSE - /\ released = FALSE - -Admit(n, c) == - /\ owner = n - /\ state[n] = Serving - /\ ~quiescing - /\ ~accepted[c] - /\ accepted' = [accepted EXCEPT ![c] = TRUE] - /\ admitted' = [admitted EXCEPT ![c] = TRUE] - /\ UNCHANGED <> - -Prepare(n) == - /\ owner = n - /\ state[n] = Serving - /\ retained = FALSE - /\ published < 2 - /\ \E c \in Commands: accepted[c] /\ ~acknowledged[c] - /\ retained' = TRUE - /\ UNCHANGED <> - -Publish(n) == - /\ owner = n - /\ state[n] = Serving - /\ retained = TRUE - /\ published < 2 - /\ retained' = FALSE - /\ published' = published + 1 - /\ recovery_required' = FALSE - /\ UNCHANGED <> - -Ack(c) == - /\ accepted[c] - /\ published > 0 - /\ ~acknowledged[c] - /\ acknowledged' = [acknowledged EXCEPT ![c] = TRUE] - /\ UNCHANGED <> - -Renew(n) == - /\ owner = n - /\ state[n] = Serving - /\ UNCHANGED vars - -Fence(n) == - /\ owner = n - /\ state' = [state EXCEPT ![n] = Fenced] - /\ quiescing' = TRUE - /\ UNCHANGED <> - -Drain(n) == - /\ owner = n - /\ state[n] = Serving - /\ ~quiescing - /\ quiescing' = TRUE - /\ UNCHANGED <> - -Crash(n) == - /\ owner = n - /\ state' = [state EXCEPT ![n] = Fenced] - /\ owner' = NoOwner - /\ recovery_required' = (recovery_required \/ retained) - /\ quiescing' = TRUE - /\ UNCHANGED <> - -Takeover(n) == - /\ owner = NoOwner \/ \E m \in Nodes: m = owner /\ state[m] = Fenced - /\ n \in Nodes - /\ epoch < 2 - /\ owner' = n - /\ epoch' = epoch + 1 - /\ state' = [m \in Nodes |-> IF m = n THEN Serving ELSE Idle] - /\ quiescing' = FALSE - /\ released' = FALSE - /\ UNCHANGED <> - -Release(n) == - /\ owner = n - /\ state[n] = Serving - /\ ~retained - /\ ~recovery_required - /\ \A c \in Commands: ~accepted[c] \/ acknowledged[c] - /\ owner' = NoOwner - /\ state' = [state EXCEPT ![n] = Idle] - /\ released' = TRUE - /\ UNCHANGED <> - -EarlyAck(c) == - /\ FaultMode = "early_ack" - /\ accepted[c] - /\ ~acknowledged[c] - /\ acknowledged' = [acknowledged EXCEPT ![c] = TRUE] - /\ UNCHANGED <> - -DualOwner(n) == - /\ FaultMode = "dual_owner" - /\ owner # NoOwner - /\ n \in Nodes - /\ n # owner - /\ owner' = n - /\ state' = [state EXCEPT ![n] = Serving] - /\ UNCHANGED <> - -AdoptDifferentWinner(n) == - /\ FaultMode = "different_winner" - /\ owner # NoOwner - /\ n \in Nodes - /\ n # owner - /\ owner' = n - /\ UNCHANGED <> - -EarlyRelease(n) == - /\ FaultMode = "early_release" - /\ owner = n - /\ retained - /\ owner' = NoOwner - /\ state' = [state EXCEPT ![n] = Idle] - /\ UNCHANGED <> - -NormalNext == - \/ \E n \in Nodes, c \in Commands: Admit(n, c) - \/ \E n \in Nodes: Prepare(n) - \/ \E n \in Nodes: Publish(n) - \/ \E c \in Commands: Ack(c) - \/ \E n \in Nodes: Renew(n) - \/ \E n \in Nodes: Drain(n) - \/ \E n \in Nodes: Fence(n) - \/ \E n \in Nodes: Crash(n) - \/ \E n \in Nodes: Takeover(n) - \/ \E n \in Nodes: Release(n) - -FaultNext == - \/ \E c \in Commands: EarlyAck(c) - \/ \E n \in Nodes: DualOwner(n) - \/ \E n \in Nodes: AdoptDifferentWinner(n) - \/ \E n \in Nodes: EarlyRelease(n) - -Next == NormalNext \/ FaultNext - -NoDualServing == Cardinality({n \in Nodes: state[n] = Serving}) <= 1 -AckHasDurability == \A c \in Commands: acknowledged[c] => published > 0 -RetainedHasOwner == retained => owner # NoOwner \/ recovery_required -OwnerIsLive == owner = NoOwner \/ \E n \in Nodes: n = owner /\ state[n] # Idle -EpochIsNatural == epoch \in Nat -EpochMonotonic == [] [epoch' >= epoch]_vars -PublishedIsNatural == published \in Nat -PublishedMonotonic == [] [published' >= published]_vars -NoAdmissionAfterFence == [] [( - \A n \in Nodes: state[n] = Fenced => accepted' = accepted -)]_vars -ReleaseAfterDurability == [] (released => - /\ ~retained - /\ ~recovery_required - /\ \A c \in Commands: ~accepted[c] \/ acknowledged[c]) -RetainedRecoveryObligation == [] (recovery_required => - retained \/ published > 0) -DrainOrRecoveryReady == [] (released => owner = NoOwner) - -(* Under the normal (non-fault) bounded provider contract, every admitted - publication/acknowledgement step and the terminal release step eventually - get a scheduling opportunity. This is deliberately a bounded liveness - claim: it does not model provider timing or unbounded membership. *) -FairProgress == - /\ WF_vars(\E n \in Nodes: Prepare(n)) - /\ WF_vars(\E n \in Nodes: Publish(n)) - /\ WF_vars(\E c \in Commands: Ack(c)) - /\ WF_vars(\E n \in Nodes: Takeover(n)) - /\ WF_vars(\E n \in Nodes: Release(n)) - -DrainCompletes == - []((quiescing /\ owner # NoOwner) ~> (owner = NoOwner \/ released)) - -Spec == Init /\ [][Next]_vars - -(* Liveness is checked under an explicit stable-provider profile. Fence and - Crash remain in the safety model, but a liveness claim cannot promise - completion while an unbounded sequence of new owner failures is allowed. *) -StableNext == - \/ \E n \in Nodes, c \in Commands: Admit(n, c) - \/ \E n \in Nodes: Prepare(n) - \/ \E n \in Nodes: Publish(n) - \/ \E c \in Commands: Ack(c) - \/ \E n \in Nodes: Renew(n) - \/ \E n \in Nodes: Drain(n) - \/ \E n \in Nodes: Takeover(n) - \/ \E n \in Nodes: Release(n) - -LivenessSpec == Init /\ [][StableNext]_vars /\ FairProgress - -THEOREM Spec => []EpochIsNatural -======================================================================================== diff --git a/crates/crab-cell-runtime/model/CellCoordinationBrokenAck.cfg b/crates/crab-cell-runtime/model/CellCoordinationBrokenAck.cfg deleted file mode 100644 index 4489b5a8b..000000000 --- a/crates/crab-cell-runtime/model/CellCoordinationBrokenAck.cfg +++ /dev/null @@ -1,8 +0,0 @@ -SPECIFICATION Spec -CONSTANTS - Nodes = {"a", "b"} - Commands = {"c1", "c2"} - FaultMode = "early_ack" -INVARIANTS - AckHasDurability -CHECK_DEADLOCK FALSE diff --git a/crates/crab-cell-runtime/model/CellCoordinationBrokenOwner.cfg b/crates/crab-cell-runtime/model/CellCoordinationBrokenOwner.cfg deleted file mode 100644 index d80bf6141..000000000 --- a/crates/crab-cell-runtime/model/CellCoordinationBrokenOwner.cfg +++ /dev/null @@ -1,8 +0,0 @@ -SPECIFICATION Spec -CONSTANTS - Nodes = {"a", "b"} - Commands = {"c1", "c2"} - FaultMode = "dual_owner" -INVARIANTS - NoDualServing -CHECK_DEADLOCK FALSE diff --git a/crates/crab-cell-runtime/model/CellCoordinationBrokenRelease.cfg b/crates/crab-cell-runtime/model/CellCoordinationBrokenRelease.cfg deleted file mode 100644 index a67fa89cd..000000000 --- a/crates/crab-cell-runtime/model/CellCoordinationBrokenRelease.cfg +++ /dev/null @@ -1,8 +0,0 @@ -SPECIFICATION Spec -CONSTANTS - Nodes = {"a", "b"} - Commands = {"c1", "c2"} - FaultMode = "early_release" -INVARIANTS - RetainedHasOwner -CHECK_DEADLOCK FALSE diff --git a/crates/crab-cell-runtime/model/CellCoordinationBrokenWinner.cfg b/crates/crab-cell-runtime/model/CellCoordinationBrokenWinner.cfg deleted file mode 100644 index eefe4ac3d..000000000 --- a/crates/crab-cell-runtime/model/CellCoordinationBrokenWinner.cfg +++ /dev/null @@ -1,8 +0,0 @@ -SPECIFICATION Spec -CONSTANTS - Nodes = {"a", "b"} - Commands = {"c1", "c2"} - FaultMode = "different_winner" -INVARIANTS - OwnerIsLive -CHECK_DEADLOCK FALSE diff --git a/crates/crab-cell-runtime/model/CellCoordinationLiveness.cfg b/crates/crab-cell-runtime/model/CellCoordinationLiveness.cfg deleted file mode 100644 index 930fddb42..000000000 --- a/crates/crab-cell-runtime/model/CellCoordinationLiveness.cfg +++ /dev/null @@ -1,8 +0,0 @@ -SPECIFICATION LivenessSpec -CONSTANTS - Nodes = {"a", "b"} - Commands = {"c1", "c2"} - FaultMode = "none" -PROPERTIES - DrainCompletes -CHECK_DEADLOCK FALSE diff --git a/crates/crab-cell-runtime/model/DELTA.md b/crates/crab-cell-runtime/model/DELTA.md deleted file mode 100644 index 38470d137..000000000 --- a/crates/crab-cell-runtime/model/DELTA.md +++ /dev/null @@ -1,40 +0,0 @@ -# Rust-to-model delta ledger - -The model checks the coordination protocol, not SQLite bytes, object-store -providers, cryptographic signatures, or HTTP. Each collapse is intentional and -must be reviewed when `src/coordination.rs` changes. - -| Rust surface | Model surface | Abstraction | Safety consequence | -| --- | --- | --- | --- | -| `CoordinationState::lifecycle` | `state` and `owner` | Serving/fenced/draining is represented by serving/fenced/idle plus owner | The model still rejects dual serving and fenced admission | -| `CoordinationState::shutdown_requested` | `state`/`accepted` | A shutdown request is folded into the model's idle transition while queued work remains represented by the accepted set | Accepted work cannot be stranded behind terminal admission closure | -| `CoordinationState::publications` | `retained`, `published` | Byte counts and roots collapse to one retained obligation and a monotonic count | Early acknowledgement/release remains observable | -| `CoordinationState::residency` | omitted | Sparse/hydrating/resident is a local acceleration state with no authority effect | Hydration cannot make a fenced owner serve or change durability | -| `CoordinationState::busy` | `accepted`/`retained` action guards | One in-flight SQL/effect slot is represented by the bounded command and publication actions | Release cannot overtake accepted work or a retained cut | -| `CoordinationState::renewing` | `Renew` | Renewal in-flight identity is collapsed to one owner-preserving action | A late renewal cannot create a second owner | -| `CoordinationState::publisher_ready` | `state`/`retained` guards | Publisher availability is represented by the serving and publication preconditions | Release remains blocked while a publication obligation exists | -| `CoordinationState::pending_effects` | action interleavings | Stable effect IDs and stale completions are abstracted as independently schedulable actions | Duplicate/reordered completion cannot advance a publication twice | -| `CoordinationState::next_effect_id` | omitted | Numeric identity has no protocol meaning beyond exact membership | Stale-ID protection is checked in Rust and the simulator | -| `CoordinationInput::Admit` | `Admit` | Request IDs/digests collapse to finite command names | Admission-after-fence remains observable | -| `CoordinationInput::CallerCancel` | stuttering action | Caller cancellation does not mutate accepted durable work | Accepted work still reaches publication/terminal resolution | -| `CoordinationInput::FollowerProof` | `Publish`/`Ack` | An accepted follower proof is a bounded alternate durability witness | A follower proof can satisfy publication durability without weakening fencing | -| `CoordinationInput::Lookup` | `state`/`owner` predicates | Read-only local eligibility is represented by the serving-owner guard; the model does not return a handle | A fenced, idle, or quiescing owner cannot be modeled as a safe local hit | -| `BeginWork`/`FinishWork` | `Admit` plus the surrounding action boundary | SQL/query execution and caller channels collapse to a bounded accepted command; work completion has no durable state of its own | Publication and acknowledgement still cannot overtake the accepted command | -| `BeginEffect`/`CompleteEffect` | Individual TLA action interleavings | Monotonic local effect IDs and duplicate completion bookkeeping are abstracted because each action is independently schedulable | A completion cannot create a new owner, root, or acknowledgement; adapter stale-ID handling is covered by the simulator | -| `BeginDrain`/`BeginShutdown`/`FinishMigration` | `Fence`, `Release`, and `Takeover` | Runtime drain phases collapse to the authority-visible idle/fenced boundary | Release cannot overtake a retained publication, and takeover still requires a fenced/idle owner | -| `BeginDrain` | `Drain` in the stable-provider profile | The liveness configuration exposes the explicit quiesce event while omitting new failures | Fair publication, acknowledgement, and release steps must eventually complete after a drain request | -| `BeginHydration`/`FinishHydration` | No durable model action | Residency is local cache state and does not change ownership, roots, or acknowledgement | Hydration cannot weaken the modeled authority and durability predicates | -| Migration admission while a durable cut is pending | `Admit` may leave work in the accepted set while `Prepare`/`Publish` remains in flight | The Rust actor queues successor work behind the migration publication; the model does not execute that queued work until the serving owner returns | Migration cannot reject already accepted successor work, and publication still precedes acknowledgement | -| `BeginPublication`/`FinishPublication` | `Prepare`/`Publish` | Upload and CAS effects collapse to one atomic publication step | Acknowledgement still requires publication | -| `Fence` and takeover | `Fence`/`Takeover` | Lease/session signatures collapse to owner identity | Different-winner and dual-owner faults remain observable | -| `CoordinationInput::FinishRenewal` | `Renew`/`Fence` | Provider success/fence outcome collapses to owner-preserving renewal or fenced state | Late renewal cannot revive a fenced owner | -| `CoordinationInput::BeginPublication`/`FinishPublication` | `Prepare`/`Publish` | Publication result and ambiguous CAS collapse to retained/published transitions | Exact winner can complete; a different winner is a fault | -| `CoordinationInput::FinishHydration` | omitted | Hydration completion is local and has no modeled authority transition | Incomplete/stale hydration cannot be a durability proof | -| `CoordinationInput::FinishMigration` | `Takeover`/`Release` boundary | Migration code/schema details collapse to the authority handoff boundary | Release remains ordered after accepted work | -| `CoordinationDecision::ResolveUnknown` | `Ack` omitted from accepted set | Resolve outcomes do not mutate the durable command state | Unknown resolution cannot acknowledge a command | -| Renewal inputs | `Renew` | Provider response details are omitted | Renewal cannot create a second serving owner | -| Async task IDs/completions | TLA action interleavings | Every completion is modeled as an independently schedulable action | Reordering is not hidden by a combined action | - -The model does not prove provider semantics, byte integrity, or Rust adapter -correctness. Those claims remain covered by the deterministic simulator and -the runtime/LTX qualification suites. diff --git a/crates/crab-cell-runtime/model/README.md b/crates/crab-cell-runtime/model/README.md deleted file mode 100644 index f5948c22a..000000000 --- a/crates/crab-cell-runtime/model/README.md +++ /dev/null @@ -1,58 +0,0 @@ -# Cell coordination model - -This directory contains the bounded TLA+ safety model for the private -coordination kernel in `../src/coordination.rs`. It is a reviewable protocol -model, not a model of SQLite, provider behavior, cryptographic signing, HTTP, -or the primitive schemas. - -## Revisions and commands - -The model is maintained with the Rust kernel at the current source revision. -The CI workflow records that revision alongside every broad result. The TLC -runner downloads only the official `v1.8.0` artifact named in `toolchain.env`, -verifies its manifest provenance (`Implementation-Title`, vendor, and the -`tlc2/TLC.class` entry), and fails closed on a mismatch. Upstream rebuilds and -re-uploads that asset in place, so the runner logs the exact build revision -instead of pinning mutable bytes; set `CRAB_CELL_TLC_SHA256` to pin an exact -digest where reproducibility matters. - -```text -crates/crab-cell-runtime/model/check.sh fast -crates/crab-cell-runtime/model/check.sh negative -crates/crab-cell-runtime/model/check.sh broad -``` - -`fast` checks the positive bounded state space and the stable-provider liveness -profile. `negative` runs four deliberately broken configurations and requires -the named invariant violation. `broad` uses the same safety and liveness models -with a deeper bound and is scheduled/manual CI evidence. - -## Modeled transitions - -The actions correspond to the kernel's admission, publication, acknowledgement, -renewal, fence/crash, takeover, and release decisions. The positive configuration -checks single ownership, admission gates, acknowledgement durability, release -ordering, retained recovery obligations, and natural/monotonic epoch and -publication watermarks. The fair drain-completion property is checked by -`CellCoordinationLiveness.cfg`; it intentionally excludes new fence/crash -events so the claim is eventual provider response, not an unbounded-failure -guarantee. The temporal admission/release properties are kept in the -`PROPERTIES` section because TLC distinguishes state predicates from -action formulas. `retained` is the bounded -publication obligation, `published` is the monotonic publication watermark, -and `owner`/`state` represent the single authoritative serving owner. The -mapping and intentional reductions are recorded in [DELTA.md](DELTA.md). - -The Rust coordination simulator covers adapter-level schedules that are not -useful in this small model: task cancellation, duplicate or lost completions, -resource pressure, richer membership, and provider failures. Conversely, TLC -enumerates every bounded action interleaving; it does not prove the Rust -adapter or any external service. - -## Expected evidence - -Positive configurations must finish without an invariant violation and report -their state counts. Negative configurations must identify the configured -invariant by name. A passing model run is protocol-assurance evidence only; -release and Celld comparison claims still require the qualification receipts -described in `../docs/delivery.md`. diff --git a/crates/crab-cell-runtime/model/check.sh b/crates/crab-cell-runtime/model/check.sh deleted file mode 100755 index 0f8b08cf5..000000000 --- a/crates/crab-cell-runtime/model/check.sh +++ /dev/null @@ -1,128 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -mode="${1:-fast}" -model_dir="$(cd "$(dirname "$0")" && pwd)" -cache_dir="${CRAB_CELL_TLC_CACHE:-${TMPDIR:-/tmp}/crab-cell-tlc}" -toolchain_file="$model_dir/toolchain.env" -source "$toolchain_file" - -if ! command -v java >/dev/null 2>&1; then - echo "error: Java 11 or newer is required for TLC" >&2 - exit 1 -fi -if ! command -v curl >/dev/null 2>&1; then - echo "error: curl is required to obtain the pinned TLC artifact" >&2 - exit 1 -fi -if ! command -v unzip >/dev/null 2>&1; then - echo "error: unzip is required to verify the TLC artifact" >&2 - exit 1 -fi - -mkdir -p "$cache_dir" -jar="$cache_dir/tla2tools-${TLA_VERSION}.jar" - -# The upstream release job rebuilds tla2tools.jar from master and overwrites the -# asset in place, so a byte digest cannot be pinned to the tag without failing -# on every upstream rebuild. Verify provenance instead, log the exact build, and -# let CRAB_CELL_TLC_SHA256 pin bytes where reproducibility matters. -verify_jar() { - local candidate="$1" - if [ -n "${CRAB_CELL_TLC_SHA256:-}" ]; then - local actual - actual="$(shasum -a 256 "$candidate" | awk '{print $1}')" - if [ "$actual" != "$CRAB_CELL_TLC_SHA256" ]; then - echo "error: TLC artifact does not match CRAB_CELL_TLC_SHA256 (got $actual)" >&2 - return 1 - fi - fi - local manifest - if ! manifest="$(unzip -p "$candidate" META-INF/MANIFEST.MF 2>/dev/null)"; then - echo "error: TLC artifact is not a readable jar" >&2 - return 1 - fi - local expected - for expected in \ - "Implementation-Title: $TLA_MANIFEST_TITLE" \ - "Implementation-Vendor: $TLA_MANIFEST_VENDOR"; do - if ! grep -Fq "$expected" <<<"$manifest"; then - echo "error: TLC artifact manifest is missing '$expected'" >&2 - return 1 - fi - done - if ! unzip -l "$candidate" tlc2/TLC.class >/dev/null 2>&1; then - echo "error: TLC artifact does not contain tlc2/TLC.class" >&2 - return 1 - fi - local revision build - revision="$(sed -n 's/^X-Git-Revision: //p' <<<"$manifest" | tr -d '\r')" - build="$(sed -n 's/^Build-TimeStamp: //p' <<<"$manifest" | tr -d '\r')" - echo "ok: TLC artifact build ${build:-unknown} (revision ${revision:-unknown})" - return 0 -} - -if [ ! -f "$jar" ] || ! verify_jar "$jar"; then - temporary="$jar.tmp.$$" - trap 'rm -f "$temporary"' EXIT - curl --fail --location --silent --show-error "$TLA_URL" --output "$temporary" - if ! verify_jar "$temporary"; then - exit 1 - fi - mv "$temporary" "$jar" - trap - EXIT -fi - -run_model() { - local config="$1" - local expected="$2" - local output - local meta_dir="$cache_dir/meta-negative/${config%.cfg}" - mkdir -p "$meta_dir" - output="$(mktemp "${TMPDIR:-/tmp}/crab-cell-tlc.XXXXXX")" - trap 'rm -f "$output"' RETURN - set +e - java -cp "$jar" tlc2.TLC -workers 1 -noGenerateSpecTE \ - -metadir "$meta_dir" -config "$model_dir/$config" \ - "$model_dir/CellCoordination.tla" | tee "$output" - local status=${PIPESTATUS[0]} - set -e - if [ "$status" -eq 0 ]; then - echo "error: expected TLC violation for $config" >&2 - return 1 - fi - if ! grep -Fq "Invariant $expected is violated." "$output"; then - echo "error: $config violated an unexpected invariant (expected $expected)" >&2 - return 1 - fi - return 0 -} - -case "$mode" in - fast) - java -cp "$jar" tlc2.TLC -workers 1 -depth 6 -nowarning -noGenerateSpecTE \ - -metadir "$cache_dir/meta-fast" \ - -config "$model_dir/CellCoordination.cfg" "$model_dir/CellCoordination.tla" - java -cp "$jar" tlc2.TLC -workers 1 -depth 6 -nowarning -noGenerateSpecTE \ - -metadir "$cache_dir/meta-liveness-fast" \ - -config "$model_dir/CellCoordinationLiveness.cfg" "$model_dir/CellCoordination.tla" - ;; - broad) - java -cp "$jar" tlc2.TLC -workers 1 -depth 10 -nowarning -noGenerateSpecTE \ - -metadir "$cache_dir/meta-broad" \ - -config "$model_dir/CellCoordination.cfg" "$model_dir/CellCoordination.tla" - java -cp "$jar" tlc2.TLC -workers 1 -depth 10 -nowarning -noGenerateSpecTE \ - -metadir "$cache_dir/meta-liveness-broad" \ - -config "$model_dir/CellCoordinationLiveness.cfg" "$model_dir/CellCoordination.tla" - ;; - negative) - run_model CellCoordinationBrokenAck.cfg AckHasDurability - run_model CellCoordinationBrokenOwner.cfg NoDualServing - run_model CellCoordinationBrokenWinner.cfg OwnerIsLive - run_model CellCoordinationBrokenRelease.cfg RetainedHasOwner - ;; - *) - echo "usage: $0 {fast|broad|negative}" >&2 - exit 2 - ;; -esac diff --git a/crates/crab-cell-runtime/model/toolchain.env b/crates/crab-cell-runtime/model/toolchain.env deleted file mode 100644 index f92299c25..000000000 --- a/crates/crab-cell-runtime/model/toolchain.env +++ /dev/null @@ -1,8 +0,0 @@ -TLA_VERSION=v1.8.0 -TLA_URL=https://github.com/tlaplus/tlaplus/releases/download/v1.8.0/tla2tools.jar -# The upstream release job rebuilds from master and overwrites this asset in -# place (observed re-uploads on 2026-09-22 and 2026-09-23), so no byte digest -# is stable for the tag. check.sh verifies manifest provenance instead and -# honours CRAB_CELL_TLC_SHA256 for callers that need an exact byte pin. -TLA_MANIFEST_TITLE="TLA+ Tools" -TLA_MANIFEST_VENDOR="Microsoft Corp." diff --git a/crates/crab-cell-runtime/qualification/README.md b/crates/crab-cell-runtime/qualification/README.md deleted file mode 100644 index 0cdc8d316..000000000 --- a/crates/crab-cell-runtime/qualification/README.md +++ /dev/null @@ -1,257 +0,0 @@ -# Cell-runtime qualification profiles - -For a local worker-scheduling diagnostic with real RustFS and enforced one-vCPU, -one-GiB limits, use [the Compose worker profile](worker-profile.md). Its raw -measurements inform performance work; they do not produce protected receipts. - -The JSON files in `profiles/` are canonical, versioned threshold inputs. The -receipt signer binds the profile digest; changing a threshold therefore makes -old evidence unusable. Profile schema 2 also binds workload thresholds to the -measured environment: protected profiles carry required provider and topology -labels, minimum throughput, peak RSS/local-disk/file-descriptor ceilings, and -an object-store call ceiling. Matrix verification requires the corresponding -`cells`, `operations`, `duration_secs`, `p99_latency_ms`, -`peak_local_disk_bytes`, and `peak_file_descriptors` metrics when a profile -sets those limits; a missing metric is a failed gate, not an assumed zero. -The release CLI additionally rejects protected receipts finished more than -seven days ago or more than five minutes ahead of its verifier clock. The -historical library verifier remains timestamp-neutral; use the fresh protected -matrix entry point for release decisions. - -Generate and verify a deterministic mixed primitive workload with: - -```text -cargo run --locked -p crab-cell-runtime --bin qualification_receipt -- \ - workload workload.json profiles/pr-contract-v1.json 7 -cargo run --locked -p crab-cell-runtime --bin qualification_receipt -- \ - verify-workload workload.json profiles/pr-contract-v1.json -``` - -After a real harness has written one receipt and one or more raw artifacts for -each matrix workload, build the canonical manifest from that evidence tree: - -```text -evidence/ -├── receipts/.json -└── artifacts// - -cargo run --locked -p crab-cell-runtime --bin qualification_receipt -- \ - manifest evidence/qualification-matrix.json evidence/ -``` - -The builder emits schema-2 rows in the required order and rejects missing, -symlinked, or non-file evidence entries. The manifest output must be directly -under the evidence directory so its relative paths remain verifiable. It only -indexes files; it does not create or sign receipts. Run `verify-matrix` with the -protected profile and pinned attestation key before treating the resulting -manifest as release evidence. - -The protected release bundle has one canonical fresh-process verifier. It -requires all nine provider, scale, compatibility, and fault profiles, rejects -symlinks anywhere below the bundle, checks each profile name and protected -threshold contract, and verifies every matrix against the same source, image, -and pinned signer: - -```text -cargo run --locked -p crab-cell-runtime --bin qualification_receipt -- \ - verify-protected-bundle protected/ \ - -``` - -The workflow and release gate call this command directly; they only retain -their shell-side byte-for-byte comparison between each supplied profile and -the tracked profile from the exact source checkout. - -Typed qualification adapters may use the bounded `QualificationWorkload::run_concurrent` -entry point when scheduled operations are independent or idempotent. The serial -`run` entry point remains the safe choice for workloads with application-level -ordering dependencies; both paths retain the same streaming schedule, counters, -latency histogram, and logical outcome digest. Protected adapters should use -`run_with_case_coverage` or `run_concurrent_with_case_coverage`, which require -each result to bind the lifecycle case it exercised. - -After a protected adapter has captured a verified run artifact, write the -non-secret execution identity and fault/ownership observations with the -`QualificationExecutionEvidence` schema, then bind the receipt with the -trusted signing key held outside the evidence directory: - -```text -cargo run --locked -p crab-cell-runtime --bin qualification_receipt -- \ - bind-protected receipt.json profile.json \ - execution-evidence.json /run/secrets/qualification-signing-key \ - run-artifact.json workload.json raw/provider-events.json -``` - -The command rejects local profiles, non-canonical evidence, mismatched -workload/run artifacts, missing protected resource measurements, and a signing -key symlink. It never generates a protected receipt from the synthetic -`emit` path; the resulting receipt must still pass `verify-matrix` with the -pinned public key before release packaging. - -Named provider profiles also require one canonical -`QualificationProviderEvidence` raw artifact beside the run and workload -artifacts. It binds the provider/profile digest and workload seed, and records -successful conditional, range, and multipart checks. Missing, duplicated, -partial, or mismatched provider semantics are rejected by both the binder and -the fresh-process matrix verifier. - -The protected `scale-v1` primitives row also requires one canonical raw JSON -artifact with `schema_version: 1`, `profile`, the 32-byte `profile_digest`, -`workload_seed`, and `cell_samples`. It has exactly fifteen samples, in this -order: `empty`, `sparse`, `resident`, `pending-publication`, and `churned`, each -at 1,000, 5,000, and 10,000 open Cells. Every sample records `state`, -`target_cells`, and `before`, `after`, and `peak` snapshots. Each snapshot -contains `active_cells`, `rss_bytes`, `allocator_bytes`, `threads`, -`file_descriptors`, `sqlite_cache_bytes`, `admitted_resident_bytes`, -`admitted_file_descriptors`, `retained_bytes`, `local_disk_reserved_bytes`, -and `local_disk_bytes`. The protected harness must capture OS process counters -and runtime admission counters from the same isolated sample, starting with -zero active Cells and observing the requested count after opening them. -`peak` records each field's high-water mark over that sample; its fields need -not come from one instant. The -verifier rejects missing, duplicate, partial, or false-open samples, requires -the peak snapshot to cover before and after, checks the admission ledger's -minimum active-Cell charges, and requires the measured run's RSS, disk, and -descriptor peaks to cover every sample. This artifact makes the per-Cell -slopes calculable from raw before/after counters; signing a scheduled -10,000-Cell workload alone is insufficient evidence that those Cells opened. -Each sample's ledger increase must cover its open Cells. Within each workload -state, the verifier compares the net `after` minus `before` growth at 1,000, -5,000, and 10,000 Cells. Growth between those points in RSS and allocator -bytes must fit the additional native and SQLite page-cache reservations plus -retained-byte charges. SQLite cache and descriptor growth must fit their own -per-Cell reservations. This comparison cancels fixed process overhead rather -than charging it repeatedly to every Cell. The protected report must still -compare that measured fixed overhead with the node's separate process reserve. - -The default workload contains a deterministic, seed-bound case schedule for each -primitive: `happy`, `retry`, `duplicate`, `expiry`, `cancellation`, `owner-loss`, -and `recovery`. Adapters inspect `QualificationOperation::case()` (or its -bounded hint accessors) to drive the corresponding primitive-specific behavior; -the schedule is a case plan, not evidence that an external provider or owner -fault actually occurred. The workload's outcome counts are forecasts used to -describe that schedule. A measured run binds the scheduled attempt count for -each primitive and records actual acknowledgements, rejections, ambiguity, -retries, and verification independently; it need not reproduce those -forecasts. A PR wiring smoke marks only the primitive/case pairs it actually -exercises and checks; other scheduled pairs stay unmarked. - -The measured summary emits `throughput_ops_per_sec` using the artifact's -whole-millisecond elapsed time, rounded up to seconds. Sub-millisecond fractions -are discarded consistently before calculating duration and throughput; the -minimum-duration gate still uses the recorded milliseconds. Protected profiles -also verify their independent resource counters. Protected `primitives` receipts -must also match the measured run artifact's cells, operations, duration, throughput, -and p50/p95/p99/max latency metrics in non-decreasing order; a signed receipt -with substituted threshold values is rejected. -Measured run artifacts use schema 4 and include a bounded primitive/case bitset; -named provider/topology profiles reject artifacts missing any lifecycle case. -Those protected profiles also require every acknowledged operation to have an -independent verification result; a partial verification count cannot be -promoted by the receipt binder. The local observed smoke remains allowed to -report partial coverage and is not release evidence. -The PR profile is a correctness gate. -`local-provider-v1`, `fault-v1`, `provider-v1`, `compatibility-v1`, and -`scale-v1` are release-candidate inputs only. The provider-specific -`provider-{s3,gcs,azure}-v1` and `fault-{s3,gcs,azure}-v1` profiles bind the -object-store provider and Kubernetes fault topology explicitly; they do not -claim provider or Kubernetes qualification until a protected run records -matching receipts and artifacts signed by the pinned qualification attestation -key. Profile verification also rejects those non-PR profiles when -the receipt is local, has no measured object-store/RSS counters, has an -environment label that does not match the profile, or has no ownership -watermark proof. The command-line `emit` helper only creates threshold metrics -for the PR correctness profile; protected evidence must come from the real -provider/fault/scale harness so it cannot be promoted from a synthetic local -receipt. - -## Release handoff - -The HTTP server release uses three runs so the protected receipt and the -promoted image share one immutable digest: - -1. Push an annotated `crab-http-server-v*` tag reachable from `origin/main`. - `.github/workflows/http-server-release.yml` builds and Compose-qualifies a - candidate, records its source, image reference, and digest in the - `http-server-candidate--` artifact, and stops before - promotion. Keep this candidate run ID. -2. Dispatch `.github/workflows/cell-runtime-protected-qualification.yml` from - that exact tag/commit with the candidate artifact's source as `source_ref` - and digest as `image_digest`. Wait for its successful protected evidence run - and keep that run ID. -3. Manually dispatch `.github/workflows/http-server-release.yml` **from that - exact tag ref** with `tag`, `candidate_run_id`, and - `cell_runtime_evidence_run_id`. The workflow rejects a dispatch SHA/ref - different from the tag so its provenance cannot bind to another commit. - The release reloads - the original candidate instead of rebuilding it, repeats Compose - qualification on that digest, verifies the complete protected bundle, and - only then promotes the image and chart. An optional - `cell_runtime_evidence_artifact` selects a non-default artifact name. - -Use the same tag ref for both manual dispatches (with values read from the -candidate and protected run artifacts): - -```sh -gh workflow run cell-runtime-protected-qualification.yml --ref "$tag" \ - -f source_ref="$source_sha" -f image_digest="$candidate_digest" -gh workflow run http-server-release.yml --ref "$tag" \ - -f tag="$tag" -f candidate_run_id="$candidate_run_id" \ - -f cell_runtime_evidence_run_id="$protected_run_id" -``` - -The protected workflow rejects a dispatch ref different from `source_ref` and uses a -protected self-hosted runner labelled `crab-cell-runtime-protected`, an exact -source commit and image digest, and the -operator-installed executable -`/opt/crab/bin/crab-cell-runtime-protected-qualifier`. That executable is the -provider/Kubernetes boundary; it must run the real workload and write the -complete evidence tree, and the workflow fails when it is absent. There is no -local, emulator, or synthetic fallback. - -The protected workflow must be named `Cell runtime protected qualification`, -run against the exact release commit, and upload an artifact named -`cell-runtime-protected--`. The artifact contains one -`protected/` directory with the tracked profiles, verified matrix manifests, -receipts, and raw artifacts. The workflow independently verifies every required -provider, scale, compatibility, and fault matrix with the pinned signer before -uploading it. The release job checks the run status, workflow name, -manual-dispatch event, run ID, attempt, and commit before moving that directory -beside the exact-source Compose receipt; the existing pinned-signer and -image/profile-bound matrix verifier remains authoritative. - -The protected executable receives these arguments and must not print secrets: - -```text -crab-cell-runtime-protected-qualifier \ - --source-sha <40-hex-commit> \ - --image-digest sha256:<64-hex> \ - --signing-key-file /run/secrets/crab-cell-runtime-qualification-signing-key \ - --output -``` - -It is responsible for isolated provider prefixes/namespaces, fault injection, -resource sampling, and writing the signed matrices. The workflow verifies the -result in a fresh Cargo process; it does not turn a command that merely claims -to have run a workload into release evidence. - -The manual release verifies that the candidate artifact came from a successful -candidate job in this workflow on the exact source commit and tag-push run. -Missing, failed, stale, wrong-workflow, wrong-commit, symlinked, or malformed -candidate or protected evidence fails closed. No synthetic receipt is accepted -as a substitute. - -The public host's local process-fault smoke can be run without credentials: - -```text -cargo test --locked -p crab-http-server --test public_cell_process_fault \ - filesystem_owner_kill_ -- \ - --nocapture -``` - -It starts an owner, successor, and independent observer as separate processes, -kills the owner before writes, after leases, and after settlements, and -verifies all primitive outcomes through typed `CellNode` handles. The filter -selects all three local lifecycle-boundary tests. The -filesystem CAS backend is a deterministic lifecycle regression fixture only; -it is not a provider, Kubernetes, or large-scale qualification receipt. diff --git a/crates/crab-cell-runtime/qualification/profiles/compatibility-v1.json b/crates/crab-cell-runtime/qualification/profiles/compatibility-v1.json deleted file mode 100644 index 66a74d13d..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/compatibility-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"compatibility-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"rustfs","topology":"rolling"} diff --git a/crates/crab-cell-runtime/qualification/profiles/fault-azure-v1.json b/crates/crab-cell-runtime/qualification/profiles/fault-azure-v1.json deleted file mode 100644 index fb073f76c..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/fault-azure-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"fault-azure-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"azure","topology":"kubernetes"} diff --git a/crates/crab-cell-runtime/qualification/profiles/fault-gcs-v1.json b/crates/crab-cell-runtime/qualification/profiles/fault-gcs-v1.json deleted file mode 100644 index 47278a5d7..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/fault-gcs-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"fault-gcs-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"gcs","topology":"kubernetes"} diff --git a/crates/crab-cell-runtime/qualification/profiles/fault-s3-v1.json b/crates/crab-cell-runtime/qualification/profiles/fault-s3-v1.json deleted file mode 100644 index 6ed84d63a..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/fault-s3-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"fault-s3-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"s3","topology":"kubernetes"} diff --git a/crates/crab-cell-runtime/qualification/profiles/fault-v1.json b/crates/crab-cell-runtime/qualification/profiles/fault-v1.json deleted file mode 100644 index 23cd9b7ab..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/fault-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"fault-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"rustfs","topology":"kubernetes"} diff --git a/crates/crab-cell-runtime/qualification/profiles/local-provider-v1.json b/crates/crab-cell-runtime/qualification/profiles/local-provider-v1.json deleted file mode 100644 index 0ead7348a..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/local-provider-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"local-provider-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"rustfs","topology":"three-process"} diff --git a/crates/crab-cell-runtime/qualification/profiles/pr-contract-v1.json b/crates/crab-cell-runtime/qualification/profiles/pr-contract-v1.json deleted file mode 100644 index f671faaeb..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/pr-contract-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"pr-contract-v1","minimum_cells":1,"minimum_operations":1,"minimum_duration_secs":1,"maximum_p99_latency_ms":5000,"minimum_throughput_ops_per_sec":0,"maximum_peak_rss_bytes":0,"maximum_local_disk_bytes":0,"maximum_file_descriptors":0,"maximum_bucket_calls":0,"provider":"","topology":""} diff --git a/crates/crab-cell-runtime/qualification/profiles/provider-azure-v1.json b/crates/crab-cell-runtime/qualification/profiles/provider-azure-v1.json deleted file mode 100644 index 2aab5d161..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/provider-azure-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"provider-azure-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"azure","topology":"three-process"} diff --git a/crates/crab-cell-runtime/qualification/profiles/provider-gcs-v1.json b/crates/crab-cell-runtime/qualification/profiles/provider-gcs-v1.json deleted file mode 100644 index fce5220ae..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/provider-gcs-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"provider-gcs-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"gcs","topology":"three-process"} diff --git a/crates/crab-cell-runtime/qualification/profiles/provider-s3-v1.json b/crates/crab-cell-runtime/qualification/profiles/provider-s3-v1.json deleted file mode 100644 index e046f4e56..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/provider-s3-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"provider-s3-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"s3","topology":"three-process"} diff --git a/crates/crab-cell-runtime/qualification/profiles/provider-v1.json b/crates/crab-cell-runtime/qualification/profiles/provider-v1.json deleted file mode 100644 index a9b1bb0cb..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/provider-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"provider-v1","minimum_cells":256,"minimum_operations":1000000,"minimum_duration_secs":60,"maximum_p99_latency_ms":1000,"minimum_throughput_ops_per_sec":1,"maximum_peak_rss_bytes":8589934592,"maximum_local_disk_bytes":21474836480,"maximum_file_descriptors":10000,"maximum_bucket_calls":10000000,"provider":"provider-matrix","topology":"three-process"} diff --git a/crates/crab-cell-runtime/qualification/profiles/scale-v1.json b/crates/crab-cell-runtime/qualification/profiles/scale-v1.json deleted file mode 100644 index 202883d84..000000000 --- a/crates/crab-cell-runtime/qualification/profiles/scale-v1.json +++ /dev/null @@ -1 +0,0 @@ -{"schema_version":2,"name":"scale-v1","minimum_cells":10000,"minimum_operations":10000000,"minimum_duration_secs":3600,"maximum_p99_latency_ms":500,"minimum_throughput_ops_per_sec":2777,"maximum_peak_rss_bytes":34359738368,"maximum_local_disk_bytes":214748364800,"maximum_file_descriptors":100000,"maximum_bucket_calls":100000000,"provider":"rustfs","topology":"dedicated-hosts"} diff --git a/crates/crab-cell-runtime/qualification/run-worker-profile.sh b/crates/crab-cell-runtime/qualification/run-worker-profile.sh deleted file mode 100644 index 04ebf150e..000000000 --- a/crates/crab-cell-runtime/qualification/run-worker-profile.sh +++ /dev/null @@ -1,44 +0,0 @@ -#!/bin/sh -set -eu - -case "${1:-}" in - single) selected=rustfs_single_worker_reports_sparse_read_interference ;; - paired) selected=rustfs_sparse_reads_report_worker_interference ;; - *) printf 'usage: run-worker-profile.sh single|paired\n' >&2; exit 2 ;; -esac - -binary= -for candidate in /target/release/deps/crab_cell_runtime-*; do - if [ -f "$candidate" ] && [ -x "$candidate" ]; then - if [ -n "$binary" ]; then - printf 'multiple runtime test binaries; use a fresh qualification directory\n' >&2 - exit 1 - fi - binary=$candidate - fi -done -test -n "$binary" -sha256sum "$binary" > "/evidence/$1-binary.sha256" - -snapshot() { - for counter in cpu.max cpu.stat memory.max memory.swap.max memory.peak memory.events; do - printf '%s\n' "$counter" - cat "/sys/fs/cgroup/$counter" - done -} -snapshot > "/evidence/$1-kernel-before.txt" -read -r quota period < /sys/fs/cgroup/cpu.max -test "$quota" -eq "$period" -test "$(cat /sys/fs/cgroup/memory.max)" -eq 1073741824 -test "$(cat /sys/fs/cgroup/memory.swap.max)" -eq 0 - -set +e -"$binary" --ignored --exact "cell::worker::tests::$selected" --nocapture \ - > "/evidence/$1.log" 2>&1 -result=$? -set -e -snapshot > "/evidence/$1-kernel-after.txt" -cat "/evidence/$1.log" -test "$result" -eq 0 -grep -Eq 'test result: ok\. 1 passed; 0 failed;' "/evidence/$1.log" -grep -q 'worker-interference {' "/evidence/$1.log" diff --git a/crates/crab-cell-runtime/qualification/worker-profile.compose.yaml b/crates/crab-cell-runtime/qualification/worker-profile.compose.yaml deleted file mode 100644 index e2388cdb4..000000000 --- a/crates/crab-cell-runtime/qualification/worker-profile.compose.yaml +++ /dev/null @@ -1,83 +0,0 @@ -x-rust-image: &rust-image rust:1.97-bookworm@sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97 -x-storage: &storage - AWS_ACCESS_KEY_ID: crab - AWS_SECRET_ACCESS_KEY: crab - AWS_DEFAULT_REGION: us-east-1 - AWS_EC2_METADATA_DISABLED: "true" - -services: - build: - image: *rust-image - working_dir: /source - environment: - CARGO_TARGET_DIR: /target - CARGO_HOME: /cargo-home - CARGO_BUILD_JOBS: "2" - # This is a native Linux build; the repository's ARM cross-linker is absent. - CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER: cc - volumes: - - ${CRAB_WORKER_STATE:?set an external state directory}/source:/source:ro - - ${CRAB_WORKER_STATE:?set an external state directory}/target-linux:/target - - cargo-home:/cargo-home - command: [cargo, test, --release, --locked, -p, crab-cell-runtime, --lib, --no-run] - cpus: 2.0 - mem_limit: 4g - memswap_limit: 4g - restart: "no" - - rustfs: - image: ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858 - environment: - RUSTFS_ACCESS_KEY: crab - RUSTFS_SECRET_KEY: crab - RUSTFS_CONSOLE_ENABLE: "false" - RUSTFS_OBS_LOG_DIRECTORY: /data/logs - volumes: - - rustfs-data:/data - healthcheck: - test: [CMD, curl, --fail, --silent, http://127.0.0.1:9000/health/ready] - interval: 3s - timeout: 3s - retries: 30 - start_period: 5s - restart: "no" - - bucket-init: - image: public.ecr.aws/aws-cli/aws-cli:2.27.41@sha256:1c2d7a51b1ff4f460bc3f1e1ee46a6cef47c7429ad9089ef99012a95e472a1a5 - environment: *storage - command: [--endpoint-url, http://rustfs:9000, s3api, create-bucket, --bucket, crab-worker-profile] - depends_on: - rustfs: - condition: service_healthy - restart: "no" - - worker: - image: *rust-image - environment: - <<: *storage - CRAB_LTX_TEST_BUCKET: crab-worker-profile - CRAB_LTX_TEST_ENDPOINT: http://rustfs:9000 - TMPDIR: /scratch - volumes: - - ${CRAB_WORKER_STATE:?set an external state directory}/source:/source:ro - - ${CRAB_WORKER_STATE:?set an external state directory}/target-linux:/target:ro - - ${CRAB_WORKER_STATE:?set an external state directory}/evidence:/evidence - - worker-local:/scratch - entrypoint: [/bin/sh, /source/crates/crab-cell-runtime/qualification/run-worker-profile.sh] - command: [single] - cpus: 1.0 - mem_limit: 1g - memswap_limit: 1g - pids_limit: 256 - read_only: true - cap_drop: [ALL] - security_opt: [no-new-privileges:true] - depends_on: - bucket-init: - condition: service_completed_successfully - restart: "no" - -volumes: - cargo-home: - rustfs-data: - worker-local: diff --git a/crates/crab-cell-runtime/qualification/worker-profile.md b/crates/crab-cell-runtime/qualification/worker-profile.md deleted file mode 100644 index 3e6fc46cf..000000000 --- a/crates/crab-cell-runtime/qualification/worker-profile.md +++ /dev/null @@ -1,114 +0,0 @@ -# Sparse reads under a one-vCPU worker limit - -This diagnostic runs the runtime's existing sparse-read fixture against real -RustFS 1.0 GA. Each measured process has a Linux CPU quota of one vCPU, a -1 GiB memory limit and no swap. The `single` case uses one SQL worker; `paired` -uses two under the same limits. It measures interference between three Cells, -checks payload integrity and retains raw timings and kernel counters. - -It exercises private worker scheduling. Public `CellNode` actions, traffic -through the load balancer, durable confirmations, first activation and recovery -still need the [application fleet qualification](../../crab-http-server/deploy/cell-issue-fleet/README.md). -This diagnostic does not generate a protected receipt or establish a supported -throughput or latency limit. - -## Run from a committed source revision - -Use a dedicated Docker context with cgroup v2 and `memory.peak` support. On -macOS, use the existing Colima context. Allow the VM at least four CPUs and -8 GiB memory; the builder has a separate two-CPU/four-GiB limit. Finish building -before measuring. Stop other workloads in that context during measurement. - -From the repository root, choose a **new** state directory and project name for -each run. The bind mounts must resolve on the Docker host; on Colima, the mounted -Workspace volume must be shared with the VM. The example uses a per-checkout -directory on that volume: - -```sh -export CRAB_WORKER_CONTEXT=colima-ga-8bc8 -worker_target=$(cd "$HOME/Workspace/crabbuild-target/crab-8bc8" && pwd -P) -export CRAB_WORKER_STATE="$worker_target/worker-profile-$(date -u +%Y%m%dT%H%M%SZ)" -worker_project="crab-worker-$(date -u +%Y%m%d%H%M%S)" -test -d "$HOME/Workspace" && test -w "$HOME/Workspace" -test ! -e "$CRAB_WORKER_STATE" -mkdir -p "$CRAB_WORKER_STATE"/{source,target-linux,evidence} -git archive HEAD | tar -x -C "$CRAB_WORKER_STATE/source" -git rev-parse HEAD > "$CRAB_WORKER_STATE/evidence/source-sha.txt" - -worker_compose() { - docker --context "$CRAB_WORKER_CONTEXT" compose \ - --project-name "$worker_project" \ - -f "$CRAB_WORKER_STATE/source/crates/crab-cell-runtime/qualification/worker-profile.compose.yaml" "$@" -} -worker_compose config --quiet -worker_compose config > "$CRAB_WORKER_STATE/evidence/compose.yaml" -docker --context "$CRAB_WORKER_CONTEXT" info --format '{{json .}}' \ - > "$CRAB_WORKER_STATE/evidence/docker-info.json" -``` - -On Colima, share this state directory as writable before the build. Add -`--mount "$CRAB_WORKER_STATE:w"` when starting the task's idle profile, preserving -its existing mounts. An unshared host path can appear as an empty directory -inside Docker. Check the source mount, then build: - -```sh -worker_compose run --rm --no-deps --entrypoint test build -f /source/Cargo.toml -worker_compose run --no-deps --name "$worker_project-build" build \ - > "$CRAB_WORKER_STATE/evidence/build.log" 2>&1 -``` - -Check the build exit status and log before continuing. The image and RustFS -versions are pinned by digest. RustFS credentials `crab` / `crab` are confined -to this disposable network; no ports are published. Start it and initialize the -fresh bucket once: - -```sh -worker_compose up -d --wait rustfs -worker_compose run --no-deps --name "$worker_project-bucket-init" bucket-init -``` - -Run the cases serially. Keep each exit status; inspect even a failed container. -The entrypoint rejects a missing or ambiguous test binary, incorrect kernel -limits, a failed test or a filter that ran zero tests. It records the binary -SHA-256, cgroup CPU/memory counters before and after, and the complete test log. - -```sh -worker_compose run --no-deps --name "$worker_project-single" worker single -docker --context "$CRAB_WORKER_CONTEXT" inspect "$worker_project-single" \ - > "$CRAB_WORKER_STATE/evidence/single-container.json" -worker_compose run --no-deps --name "$worker_project-paired" worker paired -docker --context "$CRAB_WORKER_CONTEXT" inspect "$worker_project-paired" \ - > "$CRAB_WORKER_STATE/evidence/paired-container.json" -docker --context "$CRAB_WORKER_CONTEXT" inspect "$(worker_compose ps -q rustfs)" \ - > "$CRAB_WORKER_STATE/evidence/rustfs-container.json" -worker_compose logs --no-color rustfs > "$CRAB_WORKER_STATE/evidence/rustfs.log" -worker_compose stop -``` - -Keep the state directory, containers and volumes until the evidence is reviewed. -For another independent process pair, use another fresh directory/project. -Avoid reusing a build directory containing multiple runtime test binaries. - -## Interpret the measurements - -Each `worker-interference` JSON record (schema 3) includes worker assignments, -the runtime's detected CPU parallelism, the authenticated cold root and payload -digest, query admission/queue time, SQLite callback time, and provider reads. -With one SQL worker all three Cells share it. With two, the cold and same-worker -Cells share worker zero; the comparison Cell uses worker one. - -Each process creates a random 4 MiB payload, then reopens that exact root into -six fresh sparse files with added GET delays `0, 20, 20, 0, 0, 20` ms. Provider -and metadata caches remain warm. The two resident Cells each hold 256 KiB. -A repeated full-payload query must return the original digest with zero new -origin reads. A separate background-hydration phase adds 500 ms per GET. -Delays wrap the real provider; they do not replace it with memory storage. - -Compare medians within each process and retain every sample. Three samples per -delay cannot establish p99. Separate processes create different payloads/roots; -one versus two workers is a scheduling experiment, not a byte-identical A/B -throughput comparison. Kernel `memory.peak` includes charged cache and is not -process RSS; CPU throttling counters cover setup as well as query phases. -RustFS and workers share one VM, so this does not prove independent failure -domains. Docker's [CPU and memory limits](https://docs.docker.com/engine/containers/resource_constraints/) -cap consumption; they do not reserve a dedicated physical core. diff --git a/crates/crab-cell-runtime/src/bin/cell_movement_probe.rs b/crates/crab-cell-runtime/src/bin/cell_movement_probe.rs deleted file mode 100644 index 8de6dbf55..000000000 --- a/crates/crab-cell-runtime/src/bin/cell_movement_probe.rs +++ /dev/null @@ -1,164 +0,0 @@ -// A panic in a filter process or FUSE path corrupts a worktree, so production -// builds deny unwrap, expect, panic, todo, and unimplemented; test builds keep -// them available. -#![cfg_attr( - not(test), - deny( - clippy::unwrap_used, - clippy::expect_used, - clippy::panic, - clippy::todo, - clippy::unimplemented - ) -)] - -use std::{ - env, - path::{Path as FilePath, PathBuf}, - sync::Arc, - time::Duration, -}; - -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CellCatalog; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{ApplicationId, CellTarget, NamespaceId, SessionId, TenantId}; -use crab_cell_runtime::test_support::FilesystemCasStore; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::Store; -use object_store::path::Path; - -#[tokio::main] -async fn main() -> Result<(), Box> { - let mut args = env::args().skip(1); - let store_root = required(&mut args, "store root")?; - let partition = required(&mut args, "partition")?; - let session = decode_fixed::<16>(&required(&mut args, "session")?)?; - let destination = PathBuf::from(required(&mut args, "destination")?); - let hold_ms = required(&mut args, "hold milliseconds")?.parse::()?; - let mode = match args.next().as_deref() { - None | Some("drain") => ProbeMode::Drain, - Some("crash") => ProbeMode::Crash, - Some("lost-release") => ProbeMode::LostRelease, - Some("fail-receiver") => ProbeMode::FailReceiver, - Some(_) => { - return Err("mode must be drain, crash, lost-release, or fail-receiver".into()); - } - }; - if args.next().is_some() { - return Err( - "usage: cell_movement_probe [drain|crash|lost-release|fail-receiver]" - .into(), - ); - } - - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([3; 16]), - NamespaceId::from_bytes([6; 16]), - partition.as_bytes(), - )?; - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([2; 16]); - let object_store = FilesystemCasStore::new(FilePath::new(&store_root))?; - let shared_store = object_store.clone(); - let layout = CellStorageLayout::new( - Store::new(Arc::new(shared_store)), - Path::from("runtime"), - [3; 16], - ); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .lookup(cell) - .await? - .ok_or("cell catalog entry is missing")?; - let authority = CellAuthority::new(layout.clone()); - let observed = authority - .load(cell) - .await? - .ok_or("cell control is missing")?; - let replica = CellReplica::new( - layout, - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - )?; - let session = SessionId::from_bytes(session); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1)?, 8 * 1024 * 1024, session)?; - let owner = Owner { - session, - endpoint: "https://process-movement-probe.invalid".into(), - }; - let acquisition = runtime - .acquire_idle_restored(proof, replica, authority, observed, destination, owner) - .await; - if matches!(mode, ProbeMode::FailReceiver) { - match acquisition { - Ok(handle) => { - handle.drain().await?; - runtime.shutdown().await?; - return Err("receiver-failure probe unexpectedly activated the Cell".into()); - } - Err(_) => { - runtime.shutdown().await?; - return Ok(()); - } - } - } - let handle = match acquisition { - Ok(handle) => handle, - Err(error) => { - runtime.shutdown().await?; - return Err(error.into()); - } - }; - if matches!(mode, ProbeMode::LostRelease) { - object_store.drop_next_update_response(); - } - tokio::time::sleep(Duration::from_millis(hold_ms)).await; - if matches!(mode, ProbeMode::Crash) { - std::process::exit(0); - } - handle.drain().await?; - if matches!(mode, ProbeMode::LostRelease) && !object_store.dropped_update_response() { - return Err("lost-release probe did not inject a committed response loss".into()); - } - runtime.shutdown().await?; - Ok(()) -} - -#[derive(Clone, Copy)] -enum ProbeMode { - Drain, - Crash, - LostRelease, - FailReceiver, -} - -fn required(args: &mut impl Iterator, name: &str) -> Result { - args.next().ok_or_else(|| format!("missing {name}")) -} - -fn decode_fixed(value: &str) -> Result<[u8; N], String> { - if value.len() != N * 2 { - return Err(format!("expected {} hexadecimal characters", N * 2)); - } - let mut bytes = [0_u8; N]; - for (index, pair) in value.as_bytes().as_chunks::<2>().0.iter().enumerate() { - bytes[index] = (decode_nibble(pair[0])? << 4) | decode_nibble(pair[1])?; - } - Ok(bytes) -} - -fn decode_nibble(value: u8) -> Result { - match value { - b'0'..=b'9' => Ok(value - b'0'), - b'a'..=b'f' => Ok(value - b'a' + 10), - b'A'..=b'F' => Ok(value - b'A' + 10), - _ => Err("session must be hexadecimal".into()), - } -} diff --git a/crates/crab-cell-runtime/src/bin/qualification_receipt.rs b/crates/crab-cell-runtime/src/bin/qualification_receipt.rs deleted file mode 100644 index 60f467f2d..000000000 --- a/crates/crab-cell-runtime/src/bin/qualification_receipt.rs +++ /dev/null @@ -1,1141 +0,0 @@ -// A panic in a filter process or FUSE path corrupts a worktree, so production -// builds deny unwrap, expect, panic, todo, and unimplemented; test builds keep -// them available. -#![cfg_attr( - not(test), - deny( - clippy::unwrap_used, - clippy::expect_used, - clippy::panic, - clippy::todo, - clippy::unimplemented - ) -)] - -use std::{ - env, fs, - path::{Component, Path, PathBuf}, - process::ExitCode, - time::{SystemTime, UNIX_EPOCH}, -}; - -use crab_cell_runtime::identity::Digest; -use crab_cell_runtime::qualification::cluster::validate_cluster_receipt; -use crab_cell_runtime::qualification::{ - QUALIFICATION_MATRIX_ROWS, QualificationExecutionEvidence, QualificationMatrixEntry, - QualificationMatrixManifest, QualificationMetric, QualificationOwnership, QualificationReceipt, - QualificationRunArtifact, QualificationRunner, -}; -use crab_cell_runtime::qualification::{QualificationProfile, QualificationWorkload}; -use ed25519_dalek::SigningKey; -use rand::Rng; - -const PROTECTED_BUNDLE_PROFILES: &[(&str, &str)] = &[ - ( - "local-provider-v1", - "qualification-matrix-local-provider.json", - ), - ("scale-v1", "qualification-matrix.json"), - ( - "compatibility-v1", - "qualification-matrix-compatibility.json", - ), - ("provider-s3-v1", "qualification-matrix-provider-s3.json"), - ("provider-gcs-v1", "qualification-matrix-provider-gcs.json"), - ( - "provider-azure-v1", - "qualification-matrix-provider-azure.json", - ), - ("fault-s3-v1", "qualification-matrix-fault-s3.json"), - ("fault-gcs-v1", "qualification-matrix-fault-gcs.json"), - ("fault-azure-v1", "qualification-matrix-fault-azure.json"), -]; - -fn main() -> ExitCode { - match run() { - Ok(()) => ExitCode::SUCCESS, - Err(error) => { - eprintln!("qualification receipt: {error}"); - ExitCode::FAILURE - } - } -} - -fn run() -> Result<(), String> { - let mut args = env::args().skip(1); - match args.next().as_deref() { - Some("emit") => { - let output = required(&mut args, "output")?; - let source = required(&mut args, "source revision")?; - let image = parse_digest(&required(&mut args, "image digest")?)?; - let artifact = required(&mut args, "artifact")?; - let provider = args.next().unwrap_or_else(|| "github-actions".into()); - let workload = args.next().unwrap_or_else(|| "cell-runtime-release".into()); - let fault = args.next().unwrap_or_else(|| "none".into()); - let profile = match args.next() { - Some(path) => QualificationProfile::decode( - &fs::read(path).map_err(|error| format!("read profile: {error}"))?, - ) - .map_err(|error| error.to_string())?, - None => QualificationProfile::pr_contract(), - }; - if profile.requires_protected_evidence() { - return Err( - "emit cannot create protected evidence; use bind-protected with a measured run artifact" - .into(), - ); - } - if args.next().is_some() { - return Err(usage()); - } - let artifact = - fs::read(&artifact).map_err(|error| format!("read artifact: {error}"))?; - let workload_seed = if workload == "primitives" { - primitive_workload_seed(&artifact, &profile)? - } else { - 0 - }; - let now = unix_millis()?; - let mut key_bytes = [0_u8; 32]; - rand::rng().fill(&mut key_bytes); - let mut metrics = vec![ - QualificationMetric::new( - "artifact_bytes".into(), - artifact.len() as u64, - "bytes".into(), - ) - .map_err(|error| error.to_string())?, - ]; - if profile == QualificationProfile::pr_contract() { - for (name, value, unit) in [ - ("cells", profile.minimum_cells(), "cells"), - ("operations", profile.minimum_operations(), "operations"), - ("duration_secs", profile.minimum_duration_secs(), "seconds"), - ("p99_latency_ms", 1, "ms"), - ] { - metrics.push( - QualificationMetric::new(name.into(), value, unit.into()) - .map_err(|error| error.to_string())?, - ); - } - } - let receipt = QualificationRunner::new(SigningKey::from_bytes(&key_bytes)) - .emit_with_profile_and_evidence( - &profile, - source, - image, - provider, - workload, - fault.clone(), - metrics, - &artifact, - true, - ( - "rustc".into(), - "release".into(), - "published-image".into(), - workload_seed, - 0, - 0, - false, - ), - now, - now, - fault.as_bytes(), - vec![Digest::from_bytes(*blake3::hash(&artifact).as_bytes())], - Vec::::new(), - ) - .map_err(|error| error.to_string())?; - let encoded = receipt.encode().map_err(|error| error.to_string())?; - fs::write(output, encoded).map_err(|error| format!("write receipt: {error}"))?; - Ok(()) - } - Some("bind-protected") => bind_protected(&mut args), - Some("profile") => { - let output = required(&mut args, "output")?; - let tier = args.next().unwrap_or_else(|| "pr-contract".into()); - if args.next().is_some() { - return Err(usage()); - } - let profile = match tier.as_str() { - "pr-contract" => QualificationProfile::pr_contract(), - "local-provider" => QualificationProfile::local_provider(), - "scale" => QualificationProfile::scale(), - "fault" => QualificationProfile::fault(), - "fault-s3" => QualificationProfile::fault_s3(), - "fault-gcs" => QualificationProfile::fault_gcs(), - "fault-azure" => QualificationProfile::fault_azure(), - "provider" => QualificationProfile::provider(), - "provider-s3" => QualificationProfile::provider_s3(), - "provider-gcs" => QualificationProfile::provider_gcs(), - "provider-azure" => QualificationProfile::provider_azure(), - "compatibility" => QualificationProfile::compatibility(), - _ => { - return Err( - "profile must be pr-contract, local-provider, scale, fault, fault-s3, fault-gcs, fault-azure, provider, provider-s3, provider-gcs, provider-azure, or compatibility".into(), - ); - } - }; - fs::write(output, profile.encode().map_err(|error| error.to_string())?) - .map_err(|error| format!("write profile: {error}"))?; - Ok(()) - } - Some("workload") => { - let output = required(&mut args, "output")?; - let profile_path = required(&mut args, "profile")?; - let seed = parse_u64(&required(&mut args, "seed")?, "seed")?; - let profile = QualificationProfile::decode( - &fs::read(profile_path).map_err(|error| format!("read profile: {error}"))?, - ) - .map_err(|error| error.to_string())?; - let workload = match (args.next(), args.next(), args.next()) { - (None, None, None) => QualificationWorkload::generate(&profile, seed), - (Some(cells), Some(operations), Some(duration_secs)) => { - QualificationWorkload::generate_with_size( - &profile, - seed, - parse_u64(&cells, "cells")?, - parse_u64(&operations, "operations")?, - parse_u64(&duration_secs, "duration_secs")?, - ) - } - _ => return Err(usage()), - } - .map_err(|error| error.to_string())?; - fs::write( - output, - workload.encode().map_err(|error| error.to_string())?, - ) - .map_err(|error| format!("write workload: {error}"))?; - Ok(()) - } - Some("verify-workload") => { - let workload_path = required(&mut args, "workload")?; - let profile_path = required(&mut args, "profile")?; - if args.next().is_some() { - return Err(usage()); - } - let workload = QualificationWorkload::decode( - &fs::read(workload_path).map_err(|error| format!("read workload: {error}"))?, - ) - .map_err(|error| error.to_string())?; - let profile = QualificationProfile::decode( - &fs::read(profile_path).map_err(|error| format!("read profile: {error}"))?, - ) - .map_err(|error| error.to_string())?; - workload - .verify_for_profile(&profile) - .map_err(|error| error.to_string()) - } - Some("manifest") => { - let output = PathBuf::from(required(&mut args, "manifest output")?); - let evidence_dir = PathBuf::from(required(&mut args, "evidence directory")?); - if args.next().is_some() { - return Err(usage()); - } - require_manifest_output(&output, &evidence_dir)?; - let manifest = build_manifest(&evidence_dir)?; - fs::write( - output, - manifest.encode().map_err(|error| error.to_string())?, - ) - .map_err(|error| format!("write matrix manifest: {error}"))?; - Ok(()) - } - Some("verify") => { - let receipt_path = required(&mut args, "receipt")?; - let source = required(&mut args, "source revision")?; - let image = parse_digest(&required(&mut args, "image digest")?)?; - let artifact = required(&mut args, "artifact")?; - let profile = match args.next() { - Some(path) => Some( - QualificationProfile::decode( - &fs::read(path).map_err(|error| format!("read profile: {error}"))?, - ) - .map_err(|error| error.to_string())?, - ), - None => None, - }; - let trusted_signer = args.next().map(|value| parse_signer(&value)).transpose()?; - if args.next().is_some() { - return Err(usage()); - } - let requires_freshness = profile - .as_ref() - .is_some_and(QualificationProfile::requires_protected_evidence); - require_trusted_signer(profile.as_ref(), trusted_signer.as_ref())?; - let receipt = QualificationReceipt::decode( - &fs::read(receipt_path).map_err(|error| format!("read receipt: {error}"))?, - ) - .map_err(|error| error.to_string())?; - if !receipt.passed() { - return Err("qualification receipt is not passed".into()); - } - let artifact = fs::read(artifact).map_err(|error| format!("read artifact: {error}"))?; - let result = match profile { - Some(profile) => match trusted_signer { - Some(trusted_signer) => receipt - .verify_for_profile_with_signer( - &source, - image, - &profile, - &[&artifact], - trusted_signer, - ) - .map_err(|error| error.to_string()), - None => receipt - .verify_for_profile(&source, image, &profile, &[&artifact]) - .map_err(|error| error.to_string()), - }, - None => match trusted_signer { - Some(trusted_signer) => receipt - .verify_for_trusted_signer(&source, image, &artifact, trusted_signer) - .map_err(|error| error.to_string()), - None => receipt - .verify_for(&source, image, &artifact) - .map_err(|error| error.to_string()), - }, - }; - result?; - if requires_freshness { - receipt - .verify_fresh_at(unix_millis()?) - .map_err(|error| error.to_string())?; - } - Ok(()) - } - Some("verify-matrix") => verify_matrix(&mut args), - Some("verify-protected-bundle") => verify_protected_bundle(&mut args), - Some("validate-cluster") => validate_cluster(&mut args), - _ => Err(usage()), - } -} - -fn bind_protected(args: &mut impl Iterator) -> Result<(), String> { - let output = PathBuf::from(required(args, "output")?); - let source = required(args, "source revision")?; - let image = parse_digest(&required(args, "image digest")?)?; - let profile = QualificationProfile::decode(&read_regular_file( - Path::new(&required(args, "profile")?), - "profile", - )?) - .map_err(|error| error.to_string())?; - if !profile.requires_protected_evidence() { - return Err("bind-protected requires a protected qualification profile".into()); - } - let evidence = QualificationExecutionEvidence::decode(&read_regular_file( - Path::new(&required(args, "execution evidence")?), - "execution evidence", - )?) - .map_err(|error| error.to_string())?; - let signing_key = read_signing_key(Path::new(&required(args, "signing key file")?))?; - let run_bytes = read_regular_file(Path::new(&required(args, "run artifact")?), "run artifact")?; - let run = QualificationRunArtifact::decode(&run_bytes) - .map_err(|error| format!("decode run artifact: {error}"))?; - let workload_bytes = read_regular_file( - Path::new(&required(args, "workload artifact")?), - "workload artifact", - )?; - let mut artifact_bytes = vec![run_bytes, workload_bytes]; - for path in args { - artifact_bytes.push(read_regular_file(Path::new(&path), "raw artifact")?); - } - let artifact_refs = artifact_bytes.iter().map(Vec::as_slice).collect::>(); - let receipt = QualificationRunner::new(signing_key) - .emit_protected_run(&profile, source, image, evidence, &run, &artifact_refs) - .map_err(|error| error.to_string())?; - write_regular_file( - &output, - &receipt.encode().map_err(|error| error.to_string())?, - "protected receipt", - ) -} - -fn read_regular_file(path: &Path, field: &str) -> Result, String> { - let metadata = fs::symlink_metadata(path).map_err(|error| format!("{field}: {error}"))?; - if metadata.file_type().is_symlink() || !metadata.is_file() { - return Err(format!("{field} must be a regular, non-symlink file")); - } - fs::read(path).map_err(|error| format!("read {field}: {error}")) -} - -fn write_regular_file(path: &Path, bytes: &[u8], field: &str) -> Result<(), String> { - if let Ok(metadata) = fs::symlink_metadata(path) - && (metadata.file_type().is_symlink() || !metadata.is_file()) - { - return Err(format!("{field} must be a regular, non-symlink file")); - } - fs::write(path, bytes).map_err(|error| format!("write {field}: {error}")) -} - -fn read_signing_key(path: &Path) -> Result { - let bytes = read_regular_file(path, "signing key")?; - let key_bytes = if bytes.len() == 32 { - bytes - } else { - let text = std::str::from_utf8(&bytes) - .map_err(|_| "signing key must be 32 raw bytes or 64 lowercase hex characters")? - .trim(); - if text.len() != 64 || !text.bytes().all(|byte| byte.is_ascii_hexdigit()) { - return Err("signing key must be 32 raw bytes or 64 lowercase hex characters".into()); - } - text.as_bytes() - .as_chunks::<2>() - .0 - .iter() - .map(|pair| Ok((hex(pair[0])? << 4) | hex(pair[1])?)) - .collect::, String>>()? - }; - let key_bytes: [u8; 32] = key_bytes - .try_into() - .map_err(|_| "signing key must contain exactly 32 bytes".to_owned())?; - if key_bytes.iter().all(|byte| *byte == 0) { - return Err("signing key must not be zero".into()); - } - Ok(SigningKey::from_bytes(&key_bytes)) -} - -fn primitive_workload_seed(artifact: &[u8], profile: &QualificationProfile) -> Result { - if let Ok(workload) = QualificationWorkload::decode(artifact) { - workload - .verify_for_profile(profile) - .map_err(|error| format!("verify primitive workload: {error}"))?; - return Ok(workload.seed()); - } - let run = QualificationRunArtifact::decode(artifact) - .map_err(|error| format!("decode primitive workload or run artifact: {error}"))?; - run.verify_for_profile(profile) - .map_err(|error| format!("verify primitive run artifact: {error}"))?; - Ok(run.workload().seed()) -} - -fn validate_cluster(args: &mut impl Iterator) -> Result<(), String> { - let receipt_path = required(args, "cluster receipt")?; - let source = required(args, "source revision")?; - let image = parse_digest(&required(args, "image digest")?)?; - let mode = args.next().unwrap_or_else(|| "source-only".into()); - if args.next().is_some() || !matches!(mode.as_str(), "release" | "source-only") { - return Err(usage()); - } - let receipt = - fs::read(receipt_path).map_err(|error| format!("read cluster receipt: {error}"))?; - validate_cluster_receipt(&receipt, &source, image, mode == "release") - .map_err(|error| error.to_string()) -} - -fn verify_matrix(args: &mut impl Iterator) -> Result<(), String> { - let manifest_path = PathBuf::from(required(args, "matrix manifest")?); - let source = required(args, "source revision")?; - let image = parse_digest(&required(args, "image digest")?)?; - let profile = match args.next() { - Some(path) => Some( - QualificationProfile::decode( - &fs::read(path).map_err(|error| format!("read profile: {error}"))?, - ) - .map_err(|error| error.to_string())?, - ), - None => None, - }; - let trusted_signer = args.next().map(|value| parse_signer(&value)).transpose()?; - if args.next().is_some() { - return Err(usage()); - } - verify_matrix_files( - &manifest_path, - &source, - image, - profile.as_ref(), - trusted_signer.as_ref(), - )?; - println!("qualification matrix verified"); - Ok(()) -} - -fn verify_protected_bundle(args: &mut impl Iterator) -> Result<(), String> { - let protected_dir = PathBuf::from(required(args, "protected evidence directory")?); - let source = required(args, "source revision")?; - let image = parse_digest(&required(args, "image digest")?)?; - let trusted_signer = parse_signer(&required(args, "trusted signer")?)?; - if args.next().is_some() { - return Err(usage()); - } - require_directory(&protected_dir, "protected evidence directory")?; - reject_symlinks(&protected_dir, "protected evidence")?; - - for (profile_name, matrix_name) in PROTECTED_BUNDLE_PROFILES { - let profile_path = protected_dir.join(format!("{profile_name}.json")); - let matrix_path = protected_dir.join(matrix_name); - let profile = QualificationProfile::decode(&read_regular_file( - &profile_path, - "protected qualification profile", - )?) - .map_err(|error| error.to_string())?; - if profile.name() != *profile_name { - return Err(format!( - "protected profile name mismatch: expected {profile_name}, got {}", - profile.name() - )); - } - if !profile.requires_protected_evidence() { - return Err(format!( - "protected profile {profile_name} does not require protected evidence" - )); - } - require_file(&matrix_path, "protected qualification matrix")?; - verify_matrix_files( - &matrix_path, - &source, - image, - Some(&profile), - Some(&trusted_signer), - )?; - } - println!("qualification protected bundle verified"); - Ok(()) -} - -fn verify_matrix_files( - manifest_path: &Path, - source: &str, - image: Digest, - profile: Option<&QualificationProfile>, - trusted_signer: Option<&[u8; 32]>, -) -> Result<(), String> { - require_trusted_signer(profile, trusted_signer)?; - let manifest = - QualificationMatrixManifest::decode(&read_regular_file(manifest_path, "matrix manifest")?) - .map_err(|error| error.to_string())?; - let base = manifest_base(manifest_path); - let mut receipts = Vec::with_capacity(manifest.entries().len()); - let mut artifacts = Vec::with_capacity(manifest.entries().len()); - for entry in manifest.entries() { - let receipt_path = resolve_manifest_path(base, entry.receipt())?; - receipts.push( - QualificationReceipt::decode( - &fs::read(receipt_path).map_err(|error| format!("read matrix receipt: {error}"))?, - ) - .map_err(|error| error.to_string())?, - ); - let mut row = Vec::with_capacity(entry.artifacts().len()); - for artifact in entry.artifacts() { - let artifact_path = resolve_manifest_path(base, artifact)?; - row.push( - fs::read(artifact_path) - .map_err(|error| format!("read matrix artifact: {error}"))?, - ); - } - artifacts.push(row); - } - let artifact_views = artifacts - .iter() - .map(|row| row.iter().map(Vec::as_slice).collect::>()) - .collect::>(); - let evidence = manifest - .entries() - .iter() - .enumerate() - .map(|(index, entry)| { - ( - entry.workload(), - &receipts[index], - artifact_views[index].as_slice(), - ) - }) - .collect::>(); - match (profile, trusted_signer) { - (Some(profile), Some(trusted_signer)) => { - if profile.requires_protected_evidence() { - QualificationReceipt::verify_matrix_for_profile_with_signer_fresh_at( - source, - image, - profile, - &evidence, - *trusted_signer, - unix_millis()?, - ) - .map_err(|error| error.to_string())?; - } else { - QualificationReceipt::verify_matrix_for_profile_with_signer( - source, - image, - profile, - &evidence, - *trusted_signer, - ) - .map_err(|error| error.to_string())?; - } - } - (Some(profile), None) => { - QualificationReceipt::verify_matrix_for_profile(source, image, profile, &evidence) - .map_err(|error| error.to_string())?; - } - (None, Some(_)) => return Err("a trusted signer requires a profile".into()), - (None, None) => { - QualificationReceipt::verify_matrix(source, image, &evidence) - .map_err(|error| error.to_string())?; - } - } - Ok(()) -} - -fn reject_symlinks(path: &Path, field: &str) -> Result<(), String> { - let metadata = fs::symlink_metadata(path).map_err(|error| format!("{field}: {error}"))?; - if metadata.file_type().is_symlink() { - return Err(format!("{field} must not contain symlinks")); - } - if !metadata.is_dir() { - return Ok(()); - } - for entry in fs::read_dir(path).map_err(|error| format!("read {field}: {error}"))? { - let entry = entry.map_err(|error| format!("read {field} entry: {error}"))?; - reject_symlinks(&entry.path(), field)?; - } - Ok(()) -} - -fn build_manifest(evidence_dir: &Path) -> Result { - require_directory(evidence_dir, "evidence directory")?; - let receipts_dir = evidence_dir.join("receipts"); - let artifacts_dir = evidence_dir.join("artifacts"); - require_directory(&receipts_dir, "matrix receipts directory")?; - require_directory(&artifacts_dir, "matrix artifacts directory")?; - - let entries = QUALIFICATION_MATRIX_ROWS - .iter() - .map(|workload| { - let receipt_name = format!("{workload}.json"); - require_file(&receipts_dir.join(&receipt_name), "matrix receipt")?; - let workload_dir = artifacts_dir.join(workload); - require_directory(&workload_dir, "matrix workload artifact directory")?; - let mut artifact_names = fs::read_dir(&workload_dir) - .map_err(|error| format!("read matrix workload artifacts: {error}"))? - .map(|entry| { - let entry = entry.map_err(|error| format!("read matrix artifact: {error}"))?; - let name = entry - .file_name() - .into_string() - .map_err(|_| "matrix artifact name is not UTF-8".to_owned())?; - validate_component(&name, "matrix artifact name")?; - require_file(&entry.path(), "matrix artifact")?; - Ok(name) - }) - .collect::, String>>()?; - artifact_names.sort_unstable(); - if artifact_names.is_empty() { - return Err("matrix workload has no artifacts".into()); - } - let artifacts = artifact_names - .into_iter() - .map(|name| format!("artifacts/{workload}/{name}")) - .collect(); - QualificationMatrixEntry::new( - (*workload).to_owned(), - format!("receipts/{receipt_name}"), - artifacts, - ) - .map_err(|error| error.to_string()) - }) - .collect::, String>>()?; - QualificationMatrixManifest::new(entries).map_err(|error| error.to_string()) -} - -fn require_manifest_output(output: &Path, evidence_dir: &Path) -> Result<(), String> { - let parent = output - .parent() - .filter(|path| !path.as_os_str().is_empty()) - .unwrap_or_else(|| Path::new(".")); - let evidence_root = fs::canonicalize(evidence_dir) - .map_err(|error| format!("canonicalize evidence directory: {error}"))?; - let output_parent = fs::canonicalize(parent) - .map_err(|error| format!("canonicalize matrix manifest parent: {error}"))?; - if output_parent != evidence_root { - return Err("matrix manifest output must be directly under the evidence directory".into()); - } - if let Ok(metadata) = fs::symlink_metadata(output) - && metadata.file_type().is_symlink() - { - return Err("matrix manifest output must not be a symlink".into()); - } - Ok(()) -} - -fn require_directory(path: &Path, field: &str) -> Result<(), String> { - let metadata = fs::symlink_metadata(path).map_err(|error| format!("{field}: {error}"))?; - if metadata.file_type().is_symlink() || !metadata.is_dir() { - return Err(format!("{field} must be a real directory")); - } - Ok(()) -} - -fn require_file(path: &Path, field: &str) -> Result<(), String> { - let metadata = fs::symlink_metadata(path).map_err(|error| format!("{field}: {error}"))?; - if metadata.file_type().is_symlink() || !metadata.is_file() { - return Err(format!("{field} must be a real file")); - } - Ok(()) -} - -fn validate_component(value: &str, field: &str) -> Result<(), String> { - if value.is_empty() - || value.len() > 256 - || !value.is_ascii() - || value.bytes().any(|byte| byte.is_ascii_control()) - || matches!(value, "." | "..") - { - return Err(format!("{field} is invalid")); - } - Ok(()) -} - -fn resolve_manifest_path(base: &Path, value: &str) -> Result { - let path = Path::new(value); - if path.is_absolute() - || path - .components() - .any(|component| matches!(component, Component::ParentDir)) - { - return Err("matrix paths must be relative and stay within the manifest directory".into()); - } - let canonical_base = fs::canonicalize(base) - .map_err(|error| format!("resolve matrix manifest directory: {error}"))?; - let mut candidate = canonical_base.clone(); - for component in path.components() { - candidate.push(component.as_os_str()); - let metadata = fs::symlink_metadata(&candidate) - .map_err(|error| format!("inspect matrix artifact path: {error}"))?; - if metadata.file_type().is_symlink() { - return Err("matrix paths must not be symlinks".into()); - } - } - let resolved = fs::canonicalize(&candidate) - .map_err(|error| format!("resolve matrix artifact path: {error}"))?; - if !resolved.starts_with(&canonical_base) { - return Err("matrix paths must stay within the manifest directory".into()); - } - Ok(resolved) -} - -fn require_trusted_signer( - profile: Option<&QualificationProfile>, - trusted_signer: Option<&[u8; 32]>, -) -> Result<(), String> { - if profile.is_some_and(QualificationProfile::requires_protected_evidence) - && trusted_signer.is_none() - { - return Err("protected qualification profiles require a trusted signer".into()); - } - Ok(()) -} - -fn manifest_base(path: &Path) -> &Path { - path.parent() - .filter(|parent| !parent.as_os_str().is_empty()) - .unwrap_or_else(|| Path::new(".")) -} - -fn required(args: &mut impl Iterator, name: &str) -> Result { - args.next() - .ok_or_else(|| format!("missing {name}\n{}", usage())) -} - -fn parse_digest(value: &str) -> Result { - let value = value.strip_prefix("sha256:").unwrap_or(value); - if value.len() != 64 || !value.bytes().all(|byte| byte.is_ascii_hexdigit()) { - return Err("image digest must be 64 lowercase hexadecimal characters".into()); - } - let mut bytes = [0_u8; 32]; - for (index, pair) in value.as_bytes().as_chunks::<2>().0.iter().enumerate() { - bytes[index] = (hex(pair[0])? << 4) | hex(pair[1])?; - } - Ok(Digest::from_bytes(bytes)) -} - -fn parse_u64(value: &str, name: &str) -> Result { - value - .parse::() - .map_err(|_| format!("{name} must be an unsigned integer")) -} - -fn parse_signer(value: &str) -> Result<[u8; 32], String> { - if value.len() != 64 || !value.bytes().all(|byte| byte.is_ascii_hexdigit()) { - return Err("trusted signer must be 64 lowercase hexadecimal characters".into()); - } - let mut bytes = [0_u8; 32]; - for (index, pair) in value.as_bytes().as_chunks::<2>().0.iter().enumerate() { - bytes[index] = (hex(pair[0])? << 4) | hex(pair[1])?; - } - if bytes.iter().all(|byte| *byte == 0) { - return Err("trusted signer must not be zero".into()); - } - Ok(bytes) -} - -fn hex(value: u8) -> Result { - match value { - b'0'..=b'9' => Ok(value - b'0'), - b'a'..=b'f' => Ok(value - b'a' + 10), - _ => Err("image digest must use lowercase hexadecimal characters".into()), - } -} - -fn unix_millis() -> Result { - SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_err(|error| format!("system clock: {error}")) - .and_then(|duration| { - u64::try_from(duration.as_millis()).map_err(|_| "timestamp overflow".into()) - }) -} - -fn usage() -> String { - "usage: qualification_receipt profile [pr-contract|local-provider|scale|fault|fault-s3|fault-gcs|fault-azure|provider|provider-s3|provider-gcs|provider-azure|compatibility]\n qualification_receipt workload [cells operations duration_secs]\n qualification_receipt verify-workload \n qualification_receipt manifest \n qualification_receipt emit [provider workload fault profile.json]\n qualification_receipt bind-protected [raw-artifact ...]\n qualification_receipt verify [profile.json [trusted-signer-hex]]\n qualification_receipt verify-matrix [profile.json [trusted-signer-hex]]\n qualification_receipt verify-protected-bundle \n qualification_receipt validate-cluster [release|source-only]".into() -} - -#[cfg(test)] -mod tests { - use super::{ - PROTECTED_BUNDLE_PROFILES, bind_protected, build_manifest, manifest_base, read_signing_key, - reject_symlinks, require_manifest_output, require_trusted_signer, resolve_manifest_path, - verify_protected_bundle, - }; - use crab_cell_runtime::identity::Digest; - use crab_cell_runtime::qualification::{ - QUALIFICATION_MATRIX_ROWS, QualificationExecutionEvidence, QualificationOwnership, - QualificationProviderEvidence, QualificationReceipt, - }; - use crab_cell_runtime::qualification::{ - QualificationExecution, QualificationOperation, QualificationOperationExecutor, - QualificationProfile, QualificationWorkload, - }; - use ed25519_dalek::Signer; - use std::{fs, future::Future, path::Path, pin::Pin, time::Duration}; - - struct BinderExecutor; - - impl QualificationOperationExecutor for BinderExecutor { - type Future<'a> = Pin< - Box> + Send + 'a>, - >; - - fn execute<'a>(&'a mut self, _operation: QualificationOperation) -> Self::Future<'a> { - Box::pin(async { - tokio::time::sleep(Duration::from_millis(125)).await; - Ok(QualificationExecution::acknowledged(true)) - }) - } - } - - #[test] - fn relative_manifest_uses_the_current_directory() { - assert_eq!( - manifest_base(Path::new("qualification-matrix.json")), - Path::new(".") - ); - } - - #[test] - fn nested_manifest_uses_its_parent_directory() { - assert_eq!( - manifest_base(Path::new("evidence/qualification-matrix.json")), - Path::new("evidence") - ); - } - - #[test] - fn protected_profiles_require_a_pinned_signer() { - let protected = QualificationProfile::scale(); - assert!(require_trusted_signer(Some(&protected), None).is_err()); - assert!(require_trusted_signer(Some(&QualificationProfile::pr_contract()), None).is_ok()); - assert!(require_trusted_signer(None, None).is_ok()); - } - - #[test] - fn protected_bundle_profile_contract_covers_every_release_profile() { - assert_eq!(PROTECTED_BUNDLE_PROFILES.len(), 9); - for (name, _) in PROTECTED_BUNDLE_PROFILES { - let profile = match *name { - "local-provider-v1" => QualificationProfile::local_provider(), - "scale-v1" => QualificationProfile::scale(), - "compatibility-v1" => QualificationProfile::compatibility(), - "provider-s3-v1" => QualificationProfile::provider_s3(), - "provider-gcs-v1" => QualificationProfile::provider_gcs(), - "provider-azure-v1" => QualificationProfile::provider_azure(), - "fault-s3-v1" => QualificationProfile::fault_s3(), - "fault-gcs-v1" => QualificationProfile::fault_gcs(), - "fault-azure-v1" => QualificationProfile::fault_azure(), - _ => panic!("unexpected protected profile"), - }; - assert_eq!(profile.name(), *name); - assert!(profile.requires_protected_evidence()); - } - } - - #[test] - fn protected_bundle_rejects_a_noncanonical_profile_before_matrix_reads() { - let directory = tempfile::tempdir().expect("bundle directory"); - fs::write( - directory.path().join("local-provider-v1.json"), - QualificationProfile::pr_contract() - .encode() - .expect("profile encoding"), - ) - .expect("profile"); - let mut args = vec![ - directory.path().display().to_string(), - "0123456789abcdef0123456789abcdef01234567".into(), - "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa".into(), - "0102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f20".into(), - ] - .into_iter(); - let error = verify_protected_bundle(&mut args).expect_err("noncanonical profile"); - assert!(error.contains("protected profile name mismatch")); - } - - #[test] - fn signing_key_reader_accepts_hex_and_rejects_symlinks() { - let directory = tempfile::tempdir().expect("key directory"); - let key_path = directory.path().join("key"); - fs::write( - &key_path, - "0102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f20", - ) - .expect("key"); - let key = read_signing_key(&key_path).expect("hex key"); - assert_eq!( - key.to_bytes(), - [ - 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, - 24, 25, 26, 27, 28, 29, 30, 31, 32, - ] - ); - let message = b"qualification"; - assert!(key.sign(message).to_bytes().iter().any(|byte| *byte != 0)); - - #[cfg(unix)] - { - let linked = directory.path().join("linked-key"); - std::os::unix::fs::symlink(&key_path, &linked).expect("key symlink"); - assert!(read_signing_key(&linked).is_err()); - } - } - - #[tokio::test] - async fn bind_protected_cli_binds_and_verifies_a_measured_run() { - let directory = tempfile::tempdir().expect("binder directory"); - let mut profile_value = serde_json::to_value( - QualificationProfile::new("cli-binder".into(), 1, 8, 1, 5_000).expect("binder profile"), - ) - .expect("binder profile value"); - profile_value["provider"] = serde_json::Value::String("rustfs".into()); - let profile: QualificationProfile = - serde_json::from_value(profile_value).expect("named binder profile"); - let workload = QualificationWorkload::generate_with_size(&profile, 31, 1, 8, 1) - .expect("binder workload"); - let mut executor = BinderExecutor; - let summary = workload.run(&mut executor).await.expect("binder run"); - let elapsed_ms = u64::try_from(summary.elapsed().as_millis()) - .expect("binder elapsed duration") - .max(1); - let resources = [ - crab_cell_runtime::qualification::QualificationMetric::new( - "peak_rss_bytes".into(), - 19, - "bytes".into(), - ) - .expect("RSS metric"), - crab_cell_runtime::qualification::QualificationMetric::new( - "peak_local_disk_bytes".into(), - 29, - "bytes".into(), - ) - .expect("disk metric"), - crab_cell_runtime::qualification::QualificationMetric::new( - "peak_file_descriptors".into(), - 39, - "count".into(), - ) - .expect("FD metric"), - crab_cell_runtime::qualification::QualificationMetric::new( - "bucket_calls".into(), - 49, - "count".into(), - ) - .expect("bucket metric"), - ]; - let run = summary - .artifact_with_resource_metrics(&workload, &resources) - .expect("binder run artifact"); - let profile_path = directory.path().join("profile.json"); - let evidence_path = directory.path().join("execution-evidence.json"); - let key_path = directory.path().join("signing-key"); - let run_path = directory.path().join("run-artifact.json"); - let workload_path = directory.path().join("workload.json"); - let provider_path = directory.path().join("provider-evidence.json"); - let output_path = directory.path().join("receipt.json"); - fs::write(&profile_path, profile.encode().expect("profile encoding")).expect("profile"); - fs::write( - &evidence_path, - QualificationExecutionEvidence { - provider: "rustfs".into(), - workload: "primitives".into(), - fault: "none".into(), - toolchain: "rustc-test".into(), - execution_profile: "release".into(), - topology: "three-process".into(), - started_at_ms: 1, - finished_at_ms: 1u64.saturating_add(elapsed_ms), - fault_schedule: b"none".to_vec(), - ownership: vec![QualificationOwnership::new( - 1, - 1, - Digest::from_bytes([7; 32]), - )], - dirty: false, - } - .encode() - .expect("evidence encoding"), - ) - .expect("evidence"); - let signing_key = ed25519_dalek::SigningKey::from_bytes(&[91; 32]); - fs::write(&key_path, signing_key.to_bytes()).expect("signing key"); - let run_bytes = run.encode().expect("run encoding"); - let workload_bytes = workload.encode().expect("workload encoding"); - let provider_bytes = - QualificationProviderEvidence::new(&profile, workload.seed(), true, true, true) - .expect("provider evidence") - .encode() - .expect("provider evidence encoding"); - fs::write(&run_path, &run_bytes).expect("run artifact"); - fs::write(&workload_path, &workload_bytes).expect("workload artifact"); - fs::write(&provider_path, &provider_bytes).expect("provider evidence"); - - let mut args = vec![ - output_path.display().to_string(), - "cli-source".into(), - "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa".into(), - profile_path.display().to_string(), - evidence_path.display().to_string(), - key_path.display().to_string(), - run_path.display().to_string(), - workload_path.display().to_string(), - provider_path.display().to_string(), - ] - .into_iter(); - bind_protected(&mut args).expect("bind protected receipt"); - - let receipt = QualificationReceipt::decode(&fs::read(&output_path).expect("receipt")) - .expect("receipt decoding"); - receipt - .verify_for_profile_with_signer( - "cli-source", - Digest::from_bytes([171; 32]), - &profile, - &[&run_bytes, &workload_bytes, &provider_bytes], - signing_key.verifying_key().to_bytes(), - ) - .expect_err("wrong image must be rejected"); - receipt - .verify_for_profile_with_signer( - "cli-source", - Digest::from_bytes([170; 32]), - &profile, - &[&run_bytes, &workload_bytes, &provider_bytes], - signing_key.verifying_key().to_bytes(), - ) - .expect("receipt verification"); - } - - #[test] - fn manifest_builder_emits_canonical_rows_and_rejects_invalid_output_or_missing_files() { - let directory = tempfile::tempdir().expect("matrix directory"); - fs::create_dir(directory.path().join("receipts")).expect("receipts"); - fs::create_dir(directory.path().join("artifacts")).expect("artifacts"); - for workload in QUALIFICATION_MATRIX_ROWS { - fs::write( - directory - .path() - .join("receipts") - .join(format!("{workload}.json")), - b"receipt", - ) - .expect("receipt"); - let workload_dir = directory.path().join("artifacts").join(workload); - fs::create_dir(&workload_dir).expect("workload artifacts"); - fs::write(workload_dir.join("z.json"), b"z").expect("z artifact"); - fs::write(workload_dir.join("a.json"), b"a").expect("a artifact"); - } - - let manifest = build_manifest(directory.path()).expect("canonical manifest"); - assert_eq!( - manifest - .entries() - .iter() - .map(|entry| entry.workload()) - .collect::>(), - QUALIFICATION_MATRIX_ROWS - ); - assert_eq!( - manifest.entries()[0].artifacts(), - &[ - "artifacts/protocol/a.json".to_owned(), - "artifacts/protocol/z.json".to_owned() - ] - ); - - assert!( - require_manifest_output( - &directory.path().join("nested/qualification-matrix.json"), - directory.path() - ) - .is_err() - ); - fs::remove_file(directory.path().join("receipts/protocol.json")).expect("remove receipt"); - assert!(build_manifest(directory.path()).is_err()); - } - - #[cfg(unix)] - #[test] - fn manifest_output_rejects_symlinks() { - let directory = tempfile::tempdir().expect("matrix directory"); - let target = directory.path().join("target.json"); - let output = directory.path().join("qualification-matrix.json"); - fs::write(&target, b"existing").expect("target"); - std::os::unix::fs::symlink(&target, &output).expect("manifest symlink"); - assert!(require_manifest_output(&output, directory.path()).is_err()); - } - - #[cfg(unix)] - #[test] - fn matrix_paths_reject_symlinks() { - let directory = tempfile::tempdir().expect("matrix directory"); - fs::write(directory.path().join("artifact.json"), b"artifact").expect("artifact"); - std::os::unix::fs::symlink( - directory.path().join("artifact.json"), - directory.path().join("linked.json"), - ) - .expect("symlink"); - fs::create_dir(directory.path().join("nested")).expect("nested directory"); - std::os::unix::fs::symlink( - directory.path().join("nested"), - directory.path().join("linked-directory"), - ) - .expect("directory symlink"); - fs::write(directory.path().join("nested/artifact.json"), b"artifact") - .expect("nested artifact"); - assert!(resolve_manifest_path(directory.path(), "linked.json").is_err()); - assert!(resolve_manifest_path(directory.path(), "linked-directory/artifact.json").is_err()); - assert!(resolve_manifest_path(directory.path(), "artifact.json").is_ok()); - } - - #[cfg(unix)] - #[test] - fn protected_bundle_rejects_nested_symlinks() { - let directory = tempfile::tempdir().expect("bundle directory"); - let nested = directory.path().join("nested"); - fs::create_dir(&nested).expect("nested directory"); - fs::write(nested.join("artifact"), b"artifact").expect("artifact"); - std::os::unix::fs::symlink(nested.join("artifact"), nested.join("linked")) - .expect("nested symlink"); - assert!(reject_symlinks(directory.path(), "protected evidence").is_err()); - } -} diff --git a/crates/crab-cell-runtime/src/cell.rs b/crates/crab-cell-runtime/src/cell.rs deleted file mode 100644 index 2c765d5d7..000000000 --- a/crates/crab-cell-runtime/src/cell.rs +++ /dev/null @@ -1,10 +0,0 @@ -//! One Cell: activation, admission, execution, catalog, and schema. - -pub mod actor; -pub mod application; -pub mod catalog; -pub mod due; -pub mod executor; -pub(crate) mod resume; -pub mod schema; -pub mod worker; diff --git a/crates/crab-cell-runtime/src/cell/actor.rs b/crates/crab-cell-runtime/src/cell/actor.rs deleted file mode 100644 index 041b33ba0..000000000 --- a/crates/crab-cell-runtime/src/cell/actor.rs +++ /dev/null @@ -1,412 +0,0 @@ -//! One Cell's actor: dispatcher, admission, lifecycle, requests, and runtime administration. -use std::{ - collections::{HashMap, HashSet, VecDeque}, - path::{Path, PathBuf}, - sync::{ - Arc, OnceLock, - atomic::{AtomicBool, AtomicU64, Ordering}, - }, -}; - -use tokio::{ - sync::{Semaphore, broadcast, mpsc, oneshot}, - task::JoinSet, -}; - -mod handle; -use state::*; -use task::*; -mod acquire; -mod admission; -mod lifecycle; -mod requests; -mod runtime; -mod state; -mod task; -mod tasks; - -use lifecycle::*; -use requests::*; - -use handle::{CellAdmission, WorkAdmission}; -pub use handle::{CellHandle, DueResident}; - -use crate::Error; -use crate::cell::catalog::{CatalogEntry, CatalogProof, CatalogRole}; -use crate::cell::executor::{MAX_PENDING_PUBLICATIONS, PENDING_PUBLICATION_HIGH_WATER_BYTES}; -use crate::cell::executor::{ - MigrationOutcome, MutationIdentity, PendingCommit, Resolution, StoredOutcome, -}; -use crate::cell::worker::{CellReservation, Handler, Initializer, QueryHandler, WorkerState}; -use crate::cell::worker::{HydrationStep, SqlDeadline, SqlWorkerPool, WorkerExecution}; -use crate::control::authority::{CellAuthority, VersionedControl}; -use crate::control::{Owner, Transition}; -use crate::coordination::{ - AdmissionKind, CoordinationDecision, CoordinationEffect, CoordinationInput, CoordinationState, - RejectReason, Residency, -}; -use crate::fleet::eviction::{EvictionObservation, EvictionState, select_victims}; -use crate::fleet::pressure::{ - MovementBudget, MovementPermit, PressureClassifier, PressureSample, PressureState, -}; -use crate::fleet::resource::{ - ACTIVE_CELL_NATIVE_BYTES as ACTIVE_CELL_NATIVE_BYTES_USIZE, LedgerDiskAdmission, - LedgerHostResourceAdmission, ResourceCost, ResourceLedger, ResourceReservation, - ResourceSnapshot, -}; -use crate::identity::{ApplicationId, CellId, CellTarget, Digest, SessionId}; -use crate::node::durability::NodeDurability; -use crate::node::lease::NodeLeaseGuard; -use crate::primitives::effects::InboxDelivery; -use crate::publication::CellPublisher; -use crate::publication::{CellDurabilitySubmitter, NodeDurabilitySlot, PendingDurability}; -use crate::registry::MigrationPlan; - -const INGRESS_REQUESTS: usize = 1_024; -const PUBLICATION_NOTIFICATIONS: usize = 64; -pub(crate) const CELL_REQUESTS: usize = 64; -// A repository attribution read may reserve a 1 MiB result plus its encoded -// request. Allow a normal burst of concurrent reads; node-wide retained-byte -// admission remains the aggregate safety ceiling. -pub(crate) const CELL_BYTES: usize = 16 * 1024 * 1024; -const RENEWAL_SCAN: std::time::Duration = std::time::Duration::from_millis(100); -const MAX_RENEWALS_IN_FLIGHT: usize = 32; -const SQL_WALL_DEADLINE: std::time::Duration = std::time::Duration::from_secs(5); -const HYDRATION_TICK: std::time::Duration = std::time::Duration::from_millis(100); -const HYDRATION_RETRY: std::time::Duration = std::time::Duration::from_secs(1); -// Four samples per second. The classifier still needs its 1000 ms dwell, so a -// brief spike above the soft reserve cannot evict a Cell. -const PRESSURE_SAMPLE: std::time::Duration = std::time::Duration::from_millis(250); -const COMPACTION_QUIET: std::time::Duration = std::time::Duration::from_millis(250); -const COMPACTION_RETRY: std::time::Duration = std::time::Duration::from_secs(1); -const HYDRATION_PAGES_PER_STEP: u32 = 64; -const FLEET_PUBLICATION_GRACE: std::time::Duration = std::time::Duration::from_secs(10); - -const fn bounded_u32(value: usize) -> u32 { - if value > u32::MAX as usize { - u32::MAX - } else { - value as u32 - } -} - -fn unix_millis() -> i64 { - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .map(|duration| i64::try_from(duration.as_millis()).unwrap_or(i64::MAX)) - .unwrap_or(0) -} - -/// Conservative per-active-Cell reservation for actor state and native tasks. -/// -/// This is admission accounting rather than an RSS guarantee. Embedders must -/// qualify the estimate against their compiled registry and workload. -pub const ACTIVE_CELL_NATIVE_BYTES: u64 = ACTIVE_CELL_NATIVE_BYTES_USIZE as u64; - -/// Node-wide dispatcher for bounded per-Cell command mailboxes. -#[derive(Clone)] -pub struct CellRuntime { - inner: Arc, -} - -/// Opaque node-wide byte reservation held until it is dropped. -#[must_use = "dropping the reservation immediately releases its capacity"] -pub struct NodeByteReservation { - _reservation: ResourceReservation, -} - -/// Opaque node-wide worker-job reservation held until a primitive job exits. -#[must_use = "dropping the reservation immediately releases its capacity"] -pub struct NodeJobReservation { - // Return accounting before waking the next waiter on the admission gate. - _reservation: ResourceReservation, - _permit: tokio::sync::OwnedSemaphorePermit, -} - -/// Point-in-time node admission usage for one embedded Cell runtime. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct CellRuntimeStats { - active_cells: usize, - active_cell_capacity: usize, - resident_bytes: usize, - resident_capacity_bytes: usize, - file_descriptors: usize, - file_descriptor_capacity: usize, - retained_bytes: usize, - retained_capacity_bytes: usize, - worker_jobs: usize, - worker_job_capacity: usize, - primitive_jobs: usize, - primitive_job_capacity: usize, - hydration_jobs: usize, - hydration_job_capacity: usize, - io_slots: usize, - io_slot_capacity: usize, - blocking_jobs: usize, - blocking_job_capacity: usize, - recovery_jobs: usize, - recovery_job_capacity: usize, - dirty_jobs: usize, - dirty_job_capacity: usize, - scratch_units: usize, - scratch_unit_capacity: usize, - local_disk_reserved_bytes: u64, - local_disk_capacity_bytes: u64, - unpublished_node_log_bytes: u64, -} - -impl CellRuntimeStats { - /// Returns the number of admitted Cell actors. - #[must_use] - pub const fn active_cells(self) -> usize { - self.active_cells - } - - /// Returns the maximum number of Cell actors admitted by this runtime. - #[must_use] - pub const fn active_cell_capacity(self) -> usize { - self.active_cell_capacity - } - - /// Returns resident native bytes reserved by active Cells. - #[must_use] - pub const fn resident_bytes(self) -> usize { - self.resident_bytes - } - - /// Returns the resident native-byte ceiling. - #[must_use] - pub const fn resident_capacity_bytes(self) -> usize { - self.resident_capacity_bytes - } - - /// Returns file descriptors reserved by active Cells in the shared ledger. - #[must_use] - pub const fn file_descriptors(self) -> usize { - self.file_descriptors - } - - /// Returns the active-Cell file-descriptor ceiling in the shared ledger. - #[must_use] - pub const fn file_descriptor_capacity(self) -> usize { - self.file_descriptor_capacity - } - - /// Returns the bytes currently reserved by node-wide native work. - #[must_use] - pub const fn retained_bytes(self) -> usize { - self.retained_bytes - } - - /// Returns the node-wide native-work byte capacity. - #[must_use] - pub const fn retained_capacity_bytes(self) -> usize { - self.retained_capacity_bytes - } - - /// Returns SQL jobs currently admitted by the shared ledger. - #[must_use] - pub const fn worker_jobs(self) -> usize { - self.worker_jobs - } - - /// Returns the SQL-job ceiling in the shared ledger. - #[must_use] - pub const fn worker_job_capacity(self) -> usize { - self.worker_job_capacity - } - - /// Returns primitive jobs currently admitted by the shared ledger. - #[must_use] - pub const fn primitive_jobs(self) -> usize { - self.primitive_jobs - } - - /// Returns the primitive-job ceiling in the shared ledger. - #[must_use] - pub const fn primitive_job_capacity(self) -> usize { - self.primitive_job_capacity - } - - /// Returns background hydration jobs currently admitted by the shared ledger. - #[must_use] - pub const fn hydration_jobs(self) -> usize { - self.hydration_jobs - } - - /// Returns the background hydration-job ceiling in the shared ledger. - #[must_use] - pub const fn hydration_job_capacity(self) -> usize { - self.hydration_job_capacity - } - - /// Returns the active-Cell count in the bounded placement wire shape. - #[must_use] - pub const fn placement_active_cells(self) -> u32 { - bounded_u32(self.active_cells) - } - - /// Returns the active-Cell ceiling in the bounded placement wire shape. - #[must_use] - pub const fn placement_active_cell_capacity(self) -> u32 { - bounded_u32(self.active_cell_capacity) - } - - /// Returns aggregate admitted worker, primitive, and hydration jobs. - #[must_use] - pub const fn placement_running_jobs(self) -> u32 { - bounded_u32( - self.worker_jobs - .saturating_add(self.primitive_jobs) - .saturating_add(self.hydration_jobs), - ) - } - - /// Returns aggregate worker, primitive, and hydration job capacity. - #[must_use] - pub const fn placement_job_capacity(self) -> u32 { - bounded_u32( - self.worker_job_capacity - .saturating_add(self.primitive_job_capacity) - .saturating_add(self.hydration_job_capacity), - ) - } - - /// Returns bounded object-store I/O operations currently admitted. - #[must_use] - pub const fn io_slots(self) -> usize { - self.io_slots - } - - /// Returns the object-store I/O operation ceiling. - #[must_use] - pub const fn io_slot_capacity(self) -> usize { - self.io_slot_capacity - } - - /// Returns blocking host jobs currently admitted. - #[must_use] - pub const fn blocking_jobs(self) -> usize { - self.blocking_jobs - } - - /// Returns the blocking host-job ceiling. - #[must_use] - pub const fn blocking_job_capacity(self) -> usize { - self.blocking_job_capacity - } - - /// Returns full recovery cohorts currently admitted. - #[must_use] - pub const fn recovery_jobs(self) -> usize { - self.recovery_jobs - } - - /// Returns the full recovery-cohort ceiling. - #[must_use] - pub const fn recovery_job_capacity(self) -> usize { - self.recovery_job_capacity - } - - /// Returns dirty-memory cohorts currently admitted. - #[must_use] - pub const fn dirty_jobs(self) -> usize { - self.dirty_jobs - } - - /// Returns the dirty-memory cohort ceiling. - #[must_use] - pub const fn dirty_job_capacity(self) -> usize { - self.dirty_job_capacity - } - - /// Returns temporary scratch units currently admitted. - #[must_use] - pub const fn scratch_units(self) -> usize { - self.scratch_units - } - - /// Returns the temporary scratch-unit ceiling. - #[must_use] - pub const fn scratch_unit_capacity(self) -> usize { - self.scratch_unit_capacity - } - - /// Returns the bytes currently reserved in the local replica cache. - #[must_use] - pub const fn local_disk_reserved_bytes(self) -> u64 { - self.local_disk_reserved_bytes - } - - /// Returns the local replica-cache byte capacity. - #[must_use] - pub const fn local_disk_capacity_bytes(self) -> u64 { - self.local_disk_capacity_bytes - } - - /// Returns owner bytes submitted to node logs but not covered by object roots. - #[must_use] - pub const fn unpublished_node_log_bytes(self) -> u64 { - self.unpublished_node_log_bytes - } -} - -/// New capability and publication receipt returned by one schema migration. -pub struct MigratedCell { - /// Handle that owns the migrated Cell. - pub handle: CellHandle, - /// Outcome the migration published. - pub outcome: MigrationOutcome, -} - -impl CellRuntime { - /// Resolves a verified resident owner without reading catalog or authority objects. - pub async fn resident_handle( - &self, - target: &CellTarget, - role: CatalogRole, - ) -> crate::Result> { - self.ensure_running()?; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Lookup { - cell: target.cell_id(), - require_resident: true, - reply, - }) - .await - .map_err(|_| Error::RuntimeClosed)?; - let local = match response.await { - Ok(local) => local, - Err(_) => { - self.inner - .telemetry - .resident_route(crate::fleet::telemetry::ResidentRouteOutcome::Refused); - return Err(Error::RuntimeClosed); - } - }; - let Some(local) = local else { - self.inner - .telemetry - .resident_route(crate::fleet::telemetry::ResidentRouteOutcome::Miss); - return Ok(None); - }; - self.inner - .telemetry - .resident_route(crate::fleet::telemetry::ResidentRouteOutcome::Hit); - let entry = CatalogEntry::new(target, role, local.code, local.schema)?; - Ok(Some(CellHandle { - cell: target.cell_id(), - incarnation: local.incarnation, - code: local.code, - schema: local.schema, - catalog: CatalogProof::local(entry, target), - inner: self.inner.clone(), - admission: local.admission, - })) - } -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/cell/actor/acquire.rs b/crates/crab-cell-runtime/src/cell/actor/acquire.rs deleted file mode 100644 index 5cf0ca24d..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/acquire.rs +++ /dev/null @@ -1,731 +0,0 @@ -//! Resolving, acquiring, activating, and taking over a local Cell. -//! -//! Every entry point here turns a catalog proof, a restored root, or a -//! takeover proof into a running local handle, and refuses when the -//! authority, capacity ledger, or node lease does not agree. - -use super::*; - -impl CellRuntime { - /// Resolves an active local owner without exposing the dispatcher's Cell map. - pub async fn local_handle( - &self, - catalog: CatalogProof, - control: &VersionedControl, - ) -> crate::Result> { - self.ensure_running()?; - let value = control.value(); - if catalog.entry().cell() != value.cell { - return Err(Error::Control("scheduler catalog and control differ")); - } - if value - .owner - .as_ref() - .is_none_or(|owner| owner.session != self.inner.session) - || value.root.is_none() - { - return Ok(None); - } - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Lookup { - cell: value.cell, - require_resident: false, - reply, - }) - .await - .map_err(|_| Error::RuntimeClosed)?; - let Some(local) = response.await.map_err(|_| Error::RuntimeClosed)? else { - return Ok(None); - }; - if local.incarnation != value.incarnation - || local.code != value.code - || local.schema != value.schema - { - return Ok(None); - } - Ok(Some(CellHandle { - cell: value.cell, - incarnation: value.incarnation, - code: value.code, - schema: value.schema, - catalog, - inner: self.inner.clone(), - admission: local.admission, - })) - } - - /// Creates, initializes and publishes a new Cell before returning a handle. - pub async fn bootstrap( - &self, - catalog: CatalogProof, - replica: crab_ltx::CellReplica, - authority: CellAuthority, - observed: VersionedControl, - destination: PathBuf, - initialize: F, - ) -> crate::Result - where - F: for<'connection> FnOnce( - &crab_ltx::rusqlite::Transaction<'connection>, - ) -> crate::Result<()> - + Send - + 'static, - { - self.ensure_acquiring()?; - self.check_application_limits(&catalog, replica.limits())?; - self.activation_cell(&catalog, &observed)?; - if observed.value().state != crate::control::ControlState::Recovering - || observed.value().root.is_some() - { - return Err(Error::Control("bootstrap requires an unpublished control")); - } - let replica = self - .replica_with_directory_cache(replica, &destination) - .await?; - let reservation = self.inner.pool.reserve_activation()?; - let incarnation = observed.value().incarnation; - let schema = observed.value().schema; - self.activate_inner( - catalog, - Activation::Bootstrap(Box::new(BootstrapActivation { - replica: replica.clone(), - destination, - incarnation, - schema, - initialize: Box::new(initialize), - reservation, - })), - replica, - authority, - observed, - ) - .await - } - - /// Takes over an unchanged unpublished owner, initializes and publishes the Cell. - #[expect( - clippy::too_many_arguments, - reason = "the takeover boundary keeps every authority and activation input explicit" - )] - pub async fn takeover_unpublished( - &self, - catalog: CatalogProof, - replica: crab_ltx::CellReplica, - authority: CellAuthority, - mut observed: VersionedControl, - takeover: crate::node::NodeTakeoverProof, - destination: PathBuf, - owner: Owner, - initialize: F, - ) -> crate::Result - where - F: for<'connection> FnOnce( - &crab_ltx::rusqlite::Transaction<'connection>, - ) -> crate::Result<()> - + Send - + 'static, - { - self.ensure_acquiring()?; - self.check_application_limits(&catalog, replica.limits())?; - let cell = self.claiming_cell(&catalog, &observed, &owner)?; - if owner.session != takeover.claimant() { - return Err(Error::Fenced); - } - let replica = self - .replica_with_directory_cache(replica, &destination) - .await?; - loop { - if observed.value().state != crate::control::ControlState::Recovering - || observed.value().owner.is_none() - || observed.value().root.is_some() - { - return Err(Error::Control( - "unpublished takeover requires an active rootless control", - )); - } - if observed.value().owner.as_ref().map(|owner| owner.session) - != Some(takeover.session()) - { - return Err(Error::Fenced); - } - self.ensure_running()?; - let current = authority.load(cell).await?.ok_or(Error::Fenced)?; - if current.value() != observed.value() { - self.claiming_cell(&catalog, ¤t, &owner)?; - observed = current; - continue; - } - let reservation = self.inner.pool.reserve_activation()?; - let successor = current.value().takeover(owner.clone())?; - let claimed = match authority - .transition(¤t, successor.clone(), Transition::Takeover) - .await - { - Ok(claimed) => claimed, - Err(error) => { - let latest = authority.load(cell).await?.ok_or(Error::Fenced)?; - if latest.value() == &successor { - latest - } else if matches!( - &error, - Error::Storage(crab_storage::StorageError::StateConflict { .. }) - ) { - observed = latest; - continue; - } else { - return Err(error); - } - } - }; - let incarnation = claimed.value().incarnation; - let schema = claimed.value().schema; - return self - .activate_inner( - catalog, - Activation::Bootstrap(Box::new(BootstrapActivation { - replica: replica.clone(), - destination, - incarnation, - schema, - initialize: Box::new(initialize), - reservation, - })), - replica, - authority, - claimed, - ) - .await; - } - } - - /// Cold-opens the exact authoritative root on the Cell's SQL worker. - pub async fn activate_restored( - &self, - catalog: CatalogProof, - replica: crab_ltx::CellReplica, - authority: CellAuthority, - observed: VersionedControl, - recovery_store: crate::recovery::manifest::RecoveryManifestStore, - destination: PathBuf, - ) -> crate::Result { - self.ensure_acquiring()?; - self.check_application_limits(&catalog, replica.limits())?; - self.check_application_limits(&catalog, recovery_store.limits())?; - self.activation_cell(&catalog, &observed)?; - let replica = self - .replica_with_directory_cache(replica, &destination) - .await?; - let observed = self - .publish_attached_recovery(&replica, &authority, observed, &recovery_store) - .await?; - let reservation = self.inner.pool.reserve_activation()?; - self.activate_restored_reserved( - catalog, - replica, - authority, - observed, - destination, - reservation, - ) - .await - } - - /// Acquires an idle published Cell and restores its exact immutable root. - pub async fn acquire_idle_restored( - &self, - catalog: CatalogProof, - replica: crab_ltx::CellReplica, - authority: CellAuthority, - observed: VersionedControl, - destination: PathBuf, - owner: Owner, - ) -> crate::Result { - self.ensure_acquiring()?; - self.check_application_limits(&catalog, replica.limits())?; - let rollback_node_lease = self.inner.node_lease.guard()?; - self.claiming_cell(&catalog, &observed, &owner)?; - if observed.value().state != crate::control::ControlState::Idle - || observed.value().owner.is_some() - || observed.value().root.is_none() - { - return Err(Error::Control( - "idle acquisition requires a published idle control", - )); - } - let replica = self - .replica_with_directory_cache(replica, &destination) - .await?; - let reservation = self.inner.pool.reserve_activation()?; - let successor = observed.value().takeover(owner)?; - let ownership_started = std::time::Instant::now(); - let claimed = match authority - .transition(&observed, successor.clone(), Transition::Takeover) - .await - { - Ok(claimed) => claimed, - Err(error) => { - let current = authority - .load(observed.value().cell) - .await? - .ok_or(Error::Fenced)?; - if current.value() != &successor { - return Err(error); - } - current - } - }; - self.inner.telemetry.activation_phase( - crate::fleet::telemetry::ActivationPhase::Ownership, - ownership_started.elapsed(), - ); - let rollback_authority = authority.clone(); - let rollback_claim = claimed.clone(); - let rollback_replica = replica.clone(); - match self - .activate_restored_reserved( - catalog, - replica, - authority, - claimed, - destination, - reservation, - ) - .await - { - Ok(handle) => Ok(handle), - Err(error) => { - match rollback_failed_acquisition( - &rollback_authority, - &rollback_claim, - &rollback_replica, - rollback_node_lease, - ) - .await - { - Ok(()) => Err(error), - Err(cleanup) => Err(cleanup), - } - } - } - } - - /// Takes over an unchanged owner after its exact node session is fenced. - #[expect( - clippy::too_many_arguments, - reason = "the takeover boundary keeps every authority, recovery and activation input explicit" - )] - pub async fn takeover_restored( - &self, - catalog: CatalogProof, - replica: crab_ltx::CellReplica, - authority: CellAuthority, - mut observed: VersionedControl, - takeover: crate::node::NodeTakeoverProof, - recovery_store: crate::recovery::manifest::RecoveryManifestStore, - destination: PathBuf, - owner: Owner, - ) -> crate::Result { - self.ensure_acquiring()?; - self.check_application_limits(&catalog, replica.limits())?; - self.check_application_limits(&catalog, recovery_store.limits())?; - let replica = self - .replica_with_directory_cache(replica, &destination) - .await?; - let rollback_node_lease = self.inner.node_lease.guard()?; - let rollback_authority = authority.clone(); - let rollback_replica = replica.clone(); - let cell = self.claiming_cell(&catalog, &observed, &owner)?; - if owner.session != takeover.claimant() { - return Err(Error::Fenced); - } - loop { - if !matches!( - observed.value().state, - crate::control::ControlState::Recovering | crate::control::ControlState::Serving - ) || observed.value().owner.is_none() - || observed.value().root.is_none() - { - return Err(Error::Control( - "takeover requires a published control with an active owner", - )); - } - if observed.value().owner.as_ref().map(|owner| owner.session) - != Some(takeover.session()) - { - return Err(Error::Fenced); - } - self.ensure_running()?; - let current = authority.load(cell).await?.ok_or(Error::Fenced)?; - if current.value() != observed.value() { - self.claiming_cell(&catalog, ¤t, &owner)?; - observed = current; - continue; - } - let reservation = self.inner.pool.reserve_activation()?; - let successor = current.value().takeover(owner.clone())?; - let claimed = match authority - .transition(¤t, successor.clone(), Transition::Takeover) - .await - { - Ok(claimed) => claimed, - Err(error) => { - let latest = authority.load(cell).await?.ok_or(Error::Fenced)?; - if latest.value() == &successor { - latest - } else if matches!( - &error, - Error::Storage(crab_storage::StorageError::StateConflict { .. }) - ) { - observed = latest; - continue; - } else { - return Err(error); - } - } - }; - let recovery_rollback_claim = claimed.clone(); - let claimed = match self - .publish_attached_recovery(&replica, &authority, claimed, &recovery_store) - .await - { - Ok(claimed) => claimed, - Err(error) => { - match rollback_failed_acquisition( - &rollback_authority, - &recovery_rollback_claim, - &rollback_replica, - rollback_node_lease.clone(), - ) - .await - { - Ok(()) => return Err(error), - Err(cleanup) => return Err(cleanup), - } - } - }; - let rollback_claim = claimed.clone(); - return match self - .activate_restored_reserved( - catalog, - replica, - authority, - claimed, - destination, - reservation, - ) - .await - { - Ok(handle) => Ok(handle), - Err(error) => { - match rollback_failed_acquisition( - &rollback_authority, - &rollback_claim, - &rollback_replica, - rollback_node_lease, - ) - .await - { - Ok(()) => Err(error), - Err(cleanup) => Err(cleanup), - } - } - }; - } - } - - async fn publish_attached_recovery( - &self, - replica: &crab_ltx::CellReplica, - authority: &CellAuthority, - observed: VersionedControl, - recovery_store: &crate::recovery::manifest::RecoveryManifestStore, - ) -> crate::Result { - let Some(recovery) = observed.value().recovery.as_ref() else { - return Ok(observed); - }; - let overlay = recovery_store - .load_overlay( - observed.value().cell, - observed.value().incarnation, - recovery, - ) - .await?; - let prepared = replica - .prepare_recovered_overlay(&overlay, observed.value().schema) - .await?; - self.ensure_running()?; - let successor = observed - .value() - .publish_recovery(&prepared, observed.value().next_due_ms)?; - match authority - .transition(&observed, successor.clone(), Transition::PublishRecovery) - .await - { - Ok(published) => Ok(published), - Err(error) => { - let current = authority - .load(observed.value().cell) - .await? - .ok_or(Error::Fenced)?; - if current.value() == &successor { - Ok(current) - } else { - Err(error) - } - } - } - } - - async fn activate_restored_reserved( - &self, - catalog: CatalogProof, - replica: crab_ltx::CellReplica, - authority: CellAuthority, - observed: VersionedControl, - destination: PathBuf, - reservation: CellReservation, - ) -> crate::Result { - // Acquisition installed one cache owner before recovery. Keep that - // replica through root verification, SQLite and publisher activation. - let cell = self.activation_cell(&catalog, &observed)?; - let control = observed.value(); - let root = control - .ltx_root() - .ok_or(Error::Control("activation requires a published root"))?; - let incarnation = control.incarnation; - let schema = control.schema; - // A resume record this node wrote on a clean release still names this - // exact root, so the local image can be continued instead of restored. - // The record is an accelerator: every failure falls back to the origin. - let database = match crate::cell::resume::take_matching(&destination, control, &replica) { - Some(source) => { - let resumed_started = std::time::Instant::now(); - match replica.open_resumed(&source, &destination) { - Ok(db) => { - self.inner.telemetry.activation_phase( - crate::fleet::telemetry::ActivationPhase::Resume, - resumed_started.elapsed(), - ); - crate::cell::worker::RestoredDatabase::Local(Box::new(db)) - } - Err(error) => { - tracing::debug!(error = %error, "Cell resume record did not continue"); - let _ = replica.discard_resumed(&destination); - let _ = replica.discard_resumed(&source); - self.restore_exact(&replica, &root, schema, &destination) - .await? - } - } - } - None => { - self.restore_exact(&replica, &root, schema, &destination) - .await? - } - }; - let current = authority.load(cell).await?.ok_or(Error::Fenced)?; - if !current.value().is_same_or_pure_renewal_of(observed.value()) { - return Err(Error::Fenced); - } - let activate_started = std::time::Instant::now(); - let handle = self - .activate_inner( - catalog, - Activation::Restored(Box::new(RestoredActivation { - database, - destination, - incarnation, - schema, - root, - reservation, - })), - replica, - authority, - current, - ) - .await?; - self.inner.telemetry.activation_phase( - crate::fleet::telemetry::ActivationPhase::Activate, - activate_started.elapsed(), - ); - Ok(handle) - } - - /// Verifies the immutable root and materializes it into the destination. - async fn restore_exact( - &self, - replica: &crab_ltx::CellReplica, - root: &crab_ltx::RootRef, - schema: u32, - destination: &Path, - ) -> crate::Result { - let root_open_started = std::time::Instant::now(); - let verified = replica.open_root(root).await?; - self.inner.telemetry.activation_phase( - crate::fleet::telemetry::ActivationPhase::RootOpen, - root_open_started.elapsed(), - ); - if verified.schema() != schema { - return Err(Error::Control( - "immutable root schema does not match control", - )); - } - let restore_started = std::time::Instant::now(); - let database = verified.paged().prepare_writable(destination).await?; - self.inner.telemetry.activation_phase( - crate::fleet::telemetry::ActivationPhase::Restore, - restore_started.elapsed(), - ); - Ok(crate::cell::worker::RestoredDatabase::Paged(Box::new( - database, - ))) - } - - fn claiming_cell( - &self, - catalog: &CatalogProof, - observed: &VersionedControl, - owner: &Owner, - ) -> crate::Result { - let cell = catalog.entry().cell(); - if observed.value().cell != cell { - return Err(Error::Control("ownership control changed Cell")); - } - if owner.session != self.inner.session { - return Err(Error::Fenced); - } - if observed - .value() - .owner - .as_ref() - .is_some_and(|current| current.session == owner.session) - { - return Err(Error::CellAlreadyActive); - } - Ok(cell) - } - - fn check_application_limits( - &self, - catalog: &CatalogProof, - limits: crab_ltx::Limits, - ) -> crate::Result<()> { - let Some(application_limits) = self.inner.application_limits.get() else { - return Ok(()); - }; - let Some(&(database, capture)) = application_limits.get(&catalog.entry().namespace()) - else { - return Err(Error::Control( - "Cell namespace is not declared by application", - )); - }; - if limits.max_database_bytes != database || limits.max_capture_bytes != capture { - return Err(Error::Control( - "Cell storage limits differ from application", - )); - } - Ok(()) - } - - fn activation_cell( - &self, - catalog: &CatalogProof, - observed: &VersionedControl, - ) -> crate::Result { - let cell = catalog.entry().cell(); - if observed.value().cell != cell { - return Err(Error::Control("activation control changed Cell")); - } - if observed - .value() - .owner - .as_ref() - .is_none_or(|owner| owner.session != self.inner.session) - { - return Err(Error::Fenced); - } - Ok(cell) - } - - pub(crate) fn ensure_running(&self) -> crate::Result<()> { - if self.inner.shutting_down.load(Ordering::Acquire) { - return Err(Error::RuntimeClosed); - } - self.inner.node_lease.check() - } - - fn ensure_acquiring(&self) -> crate::Result<()> { - self.ensure_running()?; - if !self.inner.accepting_cells.load(Ordering::Acquire) { - return Err(Error::CellDraining); - } - Ok(()) - } - - async fn activate_inner( - &self, - catalog: CatalogProof, - activation: Activation, - replica: crab_ltx::CellReplica, - authority: CellAuthority, - observed: VersionedControl, - ) -> crate::Result { - let cell = self.activation_cell(&catalog, &observed)?; - let incarnation = observed.value().incarnation; - let code = observed.value().code; - let schema = observed.value().schema; - let scratch_directory = match &activation { - Activation::Restored(activation) => activation.destination.parent(), - Activation::Bootstrap(activation) => activation.destination.parent(), - } - .ok_or(Error::Control("Cell activation destination has no parent"))? - .to_owned(); - let (reply, response) = oneshot::channel(); - let mut publisher = CellPublisher::new(replica, authority, observed, scratch_directory); - if let Some(node_lease) = self.inner.node_lease.guard()? { - publisher = publisher.with_node_lease(node_lease); - } - publisher = publisher.with_node_durability_slot(Arc::clone(&self.inner.node_durability)); - publisher = publisher.with_telemetry(self.inner.telemetry.clone()); - self.inner - .sender - .send(Message::Activate { - cell, - role: catalog.entry().role(), - catalog: catalog.clone(), - activation, - publisher: Box::new(publisher), - reply, - }) - .await - .map_err(|_| Error::RuntimeClosed)?; - let admission = response.await.map_err(|_| Error::RuntimeClosed)??; - Ok(CellHandle { - cell, - incarnation, - code, - schema, - catalog, - inner: self.inner.clone(), - admission, - }) - } - - async fn replica_with_directory_cache( - &self, - replica: crab_ltx::CellReplica, - destination: &Path, - ) -> crate::Result { - let scratch_directory = destination - .parent() - .ok_or(Error::Control("Cell activation destination has no parent"))?; - let host = self - .inner - .replica_host - .clone() - .with_directory_cache(scratch_directory.join(".crab-cell-directory-cache")) - .await?; - Ok(replica.with_host(host)) - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/admission.rs b/crates/crab-cell-runtime/src/cell/actor/admission.rs deleted file mode 100644 index ddbdd7764..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/admission.rs +++ /dev/null @@ -1,182 +0,0 @@ -//! Admission, fencing, completion, and reply helpers. - -use super::*; - -pub(super) fn fail_shutdown(shutdown: &mut Option, error: Error) { - if let Some(state) = shutdown.as_mut() - && state.error.is_none() - { - state.error = Some(error); - } -} - -pub(super) fn finish_shutdown(shutdown: &mut Option) { - let Some(mut state) = shutdown.take() else { - return; - }; - let result = state.error.take().map_or(Ok(()), Err); - let _ = state.reply.send(result); -} - -pub(super) fn subtract_unpublished_bytes(total: &AtomicU64, bytes: u64) { - let _ = total.fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { - Some(current.saturating_sub(bytes)) - }); -} - -pub(super) fn rejection_error(reason: RejectReason) -> Error { - match reason { - RejectReason::NotActive => Error::CellNotActive, - RejectReason::Fenced => Error::Fenced, - RejectReason::Draining => Error::CellDraining, - RejectReason::Busy => Error::CellDraining, - RejectReason::PublicationPending => Error::PendingPublication, - } -} - -pub(super) fn finish_work(active: &mut ActiveCell, fenced: bool) -> CoordinationDecision { - active.last_work_at = std::time::Instant::now(); - let decision = active - .coordination - .step(CoordinationInput::FinishWork { fenced }); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - } - decision -} - -pub(super) fn finish_migration(active: &mut ActiveCell, fenced: bool) -> CoordinationDecision { - active.last_work_at = std::time::Instant::now(); - let decision = active - .coordination - .step(CoordinationInput::FinishMigration { fenced }); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - } - decision -} - -pub(super) fn fence_active(active: &mut ActiveCell) { - active.coordination.step(CoordinationInput::Fence); - fence_admission(&active.admission); - if let Some(transfer) = active.transfer.take() { - let _ = transfer.reply.send(Err(Error::Fenced)); - } - active.inventory_refreshing = false; - while let Some(publication) = active.publications.pop_front() { - active - .coordination - .step(CoordinationInput::FinishPublication { - fenced: true, - succeeded: false, - }); - active.publication_bytes = active - .publication_bytes - .saturating_sub(publication.pending.retained_bytes()); - let _ = publication.proof.send(Err(Error::Fenced)); - } - while let Some(queued) = active.queue.pop_front() { - match queued { - QueuedWork::Command(mut command) => { - send_command_reply(&mut command, Err(Error::Fenced)); - } - QueuedWork::Query(mut query) => { - send_query_reply(&mut query, Err(Error::Fenced)); - } - QueuedWork::Resolve(mut resolve) => { - send_resolve_reply(&mut resolve, Ok(Resolution::Unknown)); - } - QueuedWork::Migration(mut migration) => { - send_migration_reply(&mut migration, Err(Error::Fenced)); - } - } - } -} - -pub(super) fn fence_admission(admission: &CellAdmission) { - admission.fenced.store(true, Ordering::Release); - admission.draining.store(true, Ordering::Release); - admission.requests.close(); - admission.bytes.close(); -} - -pub(super) fn new_cell_admission() -> Arc { - Arc::new(CellAdmission { - requests: Arc::new(Semaphore::new(CELL_REQUESTS)), - bytes: Arc::new(Semaphore::new(CELL_BYTES)), - draining: AtomicBool::new(false), - fenced: AtomicBool::new(false), - }) -} - -pub(super) fn send_command_reply( - command: &mut QueuedCommand, - result: crate::Result, -) { - if let Some(reply) = command.reply.take() { - let sequence = result.as_ref().ok().map(StoredOutcome::commit_sequence); - if reply.send(result).is_ok() - && let Some(commit_sequence) = sequence - { - use crate::fleet::telemetry::CommandResponseSource; - use crate::node::log::DurabilitySource; - - let (source, confirmation) = match command.response_proof { - Some((DurabilitySource::Fleet, elapsed)) => (CommandResponseSource::Fleet, elapsed), - Some((DurabilitySource::Object, elapsed)) => { - (CommandResponseSource::Object, elapsed) - } - None => (CommandResponseSource::Recorded, std::time::Duration::ZERO), - }; - // Final admission may refuse a proven command. Observe at the one - // reply boundary so failed or later object proofs cannot add winners. - let elapsed = command.queued_at.elapsed(); - tracing::debug!( - target: "crab_cell_runtime::action", - parent: &command.trace, - event = "cell_command_response", - commit_sequence, - source = ?source, - response_us = elapsed.as_micros(), - confirmation_us = confirmation.as_micros(), - "Cell command response released" - ); - command - .telemetry - .command_response(source, elapsed, confirmation); - } - } -} - -pub(super) fn send_command_task_reply( - command: &mut QueuedCommand, - result: crate::Result, -) { - let result = match result { - Ok(CommandTaskResult::Recorded(outcome)) => Ok(outcome), - Ok(CommandTaskResult::Pending { .. }) => Err(command.operation.unknown(Error::Fenced)), - Err(error) => Err(error), - }; - send_command_reply(command, result); -} - -pub(super) fn send_query_reply(query: &mut QueuedQuery, result: crate::Result>) { - if let Some(reply) = query.reply.take() { - let _ = reply.send(result); - } -} - -pub(super) fn send_resolve_reply(resolve: &mut QueuedResolve, result: crate::Result) { - if let Some(reply) = resolve.reply.take() { - let _ = reply.send(result); - } -} - -pub(super) fn send_migration_reply( - migration: &mut QueuedMigration, - result: crate::Result, -) { - if let Some(reply) = migration.reply.take() { - let _ = reply.send(result); - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/handle.rs b/crates/crab-cell-runtime/src/cell/actor/handle.rs deleted file mode 100644 index 19394ee1a..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/handle.rs +++ /dev/null @@ -1,507 +0,0 @@ -use std::sync::{ - Arc, - atomic::{AtomicBool, Ordering}, -}; - -use tokio::sync::{OwnedSemaphorePermit, Semaphore, TryAcquireError, oneshot}; - -use super::admission::new_cell_admission; -use super::state::{ - Message, QueuedCommand, QueuedMigration, QueuedOperation, QueuedQuery, QueuedResolve, - ResolveOperation, RuntimeInner, -}; - -use crate::Error; -use crate::cell::actor::MigratedCell; -use crate::cell::catalog::CatalogProof; -use crate::cell::catalog::CatalogRole; -use crate::cell::executor::StoredOutcome; -use crate::cell::executor::{MutationIdentity, Resolution}; -use crate::fleet::resource::{ResourceCost, ResourceReservation}; -use crate::identity::IncarnationId; -use crate::identity::{CellId, Digest}; -use crate::primitives::effects::InboxDelivery; -use crate::primitives::maintenance::PersistedWorkInventory; -use crate::registry::MigrationPlan; - -pub(super) const MAX_OPERATION_BYTES: usize = crate::codec::MAX_WIRE_BYTES; -pub(super) const MAX_RESULT_BYTES: usize = crate::codec::MAX_WIRE_BYTES; - -/// Cloneable capability for one activated Cell. -#[derive(Clone)] -pub struct CellHandle { - pub(super) cell: CellId, - pub(super) incarnation: IncarnationId, - pub(super) code: Digest, - pub(super) schema: u32, - pub(super) catalog: CatalogProof, - pub(super) inner: Arc, - pub(super) admission: Arc, -} - -pub(super) struct CellAdmission { - pub(super) requests: Arc, - pub(super) bytes: Arc, - pub(super) draining: AtomicBool, - pub(super) fenced: AtomicBool, -} - -pub(super) struct WorkAdmission { - pub(super) _request: OwnedSemaphorePermit, - pub(super) _cell_bytes: OwnedSemaphorePermit, - pub(super) _node_bytes: ResourceReservation, -} - -/// One resident Cell whose published due time has passed. -/// -/// A scheduler ticks such a Cell through the handle it already holds, so due -/// work does not wait for a fleet scan to reach it. `expected_commit_sequence` -/// is what the last authoritative publication named: a Tick built from it is a -/// no-op when the Cell committed again in the meantime. -pub struct DueResident { - handle: CellHandle, - expected_commit_sequence: u64, - next_due_ms: i64, -} - -impl DueResident { - pub(super) const fn new( - handle: CellHandle, - expected_commit_sequence: u64, - next_due_ms: i64, - ) -> Self { - Self { - handle, - expected_commit_sequence, - next_due_ms, - } - } - - /// Returns the resident handle to dispatch the Tick through. - #[must_use] - pub const fn handle(&self) -> &CellHandle { - &self.handle - } - - /// Returns the commit sequence the last publication named. - #[must_use] - pub const fn expected_commit_sequence(&self) -> u64 { - self.expected_commit_sequence - } - - /// Returns the due time this Cell published. - #[must_use] - pub const fn next_due_ms(&self) -> i64 { - self.next_due_ms - } -} - -impl CellHandle { - /// Returns the catalog proof that authorized this activation. - #[must_use] - pub const fn catalog(&self) -> &CatalogProof { - &self.catalog - } - - /// Returns the Cell this handle addresses. - #[must_use] - pub const fn cell_id(&self) -> CellId { - self.cell - } - - /// Returns the incarnation the activation belongs to. - #[must_use] - pub const fn incarnation(&self) -> IncarnationId { - self.incarnation - } - - /// Returns the application code digest the Cell serves. - #[must_use] - pub const fn code(&self) -> Digest { - self.code - } - - /// Returns the schema version the Cell serves. - #[must_use] - pub const fn schema(&self) -> u32 { - self.schema - } - - /// Runs and publishes one command while retaining admission after cancellation. - pub async fn execute( - &self, - identity: MutationIdentity, - operation_digest: Digest, - now_ms: i64, - operation_bytes: usize, - max_result_bytes: usize, - handler: F, - ) -> crate::Result - where - F: for<'connection> FnOnce( - &crab_ltx::rusqlite::Transaction<'connection>, - ) - -> crate::Result - + Send - + 'static, - { - let admission = self.reserve_work(operation_bytes, max_result_bytes)?; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Execute(Box::new(QueuedCommand { - trace: tracing::debug_span!( - target: "crab_cell_runtime::action", - "cell_execution", - cell = ?self.cell, - incarnation = ?self.incarnation, - owner_session = ?self.inner.session, - mutation_request_id = ?identity.request_id, - ), - telemetry: self.inner.telemetry.clone(), - queued_at: std::time::Instant::now(), - response_proof: None, - cell: self.cell, - admission: self.admission.clone(), - operation: QueuedOperation::Mutation { - identity, - operation_digest, - }, - now_ms, - max_result_bytes, - handler: Some(Box::new(handler)), - reply: Some(reply), - _work: admission, - }))) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::OutcomeUnknown { - request_id: identity.request_id, - operation_digest, - source: Box::new(Error::RuntimeClosed), - })? - } - - /// Applies or replays one private destination effect through durable publication. - pub async fn deliver_effect( - &self, - delivery: InboxDelivery, - now_ms: i64, - operation_bytes: usize, - max_result_bytes: usize, - handler: F, - ) -> crate::Result - where - F: for<'connection> FnOnce( - &crab_ltx::rusqlite::Transaction<'connection>, - ) - -> crate::Result - + Send - + 'static, - { - let admission = self.reserve_work(operation_bytes, max_result_bytes)?; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Execute(Box::new(QueuedCommand { - trace: tracing::debug_span!( - target: "crab_cell_runtime::action", - "cell_effect_execution", - cell = ?self.cell, - incarnation = ?self.incarnation, - owner_session = ?self.inner.session, - effect_id = ?delivery.effect_id, - ), - telemetry: self.inner.telemetry.clone(), - queued_at: std::time::Instant::now(), - response_proof: None, - cell: self.cell, - admission: self.admission.clone(), - operation: QueuedOperation::Effect { delivery }, - now_ms, - max_result_bytes, - handler: Some(Box::new(handler)), - reply: Some(reply), - _work: admission, - }))) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::EffectOutcomeUnknown { - effect_id: delivery.effect_id, - operation_digest: delivery.operation_digest, - source: Box::new(Error::RuntimeClosed), - })? - } - - /// Runs one FIFO-ordered bounded read after all preceding writes publish. - pub async fn query( - &self, - operation_bytes: usize, - max_result_bytes: usize, - handler: F, - ) -> crate::Result> - where - F: FnOnce(&crab_ltx::rusqlite::Connection) -> crate::Result> + Send + 'static, - { - let admission = self.reserve_work(operation_bytes, max_result_bytes)?; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Query(Box::new(QueuedQuery { - cell: self.cell, - admission: self.admission.clone(), - max_result_bytes, - handler: Some(Box::new(handler)), - reply: Some(reply), - _work: admission, - }))) - .await - .map_err(|_| Error::RuntimeClosed)?; - let result = response.await.map_err(|_| Error::RuntimeClosed)?; - self.inner.node_lease.check()?; - result - } - - /// Reads durable work that can retain a removed release contract. - pub async fn persisted_work_inventory( - &self, - role: CatalogRole, - ) -> crate::Result { - let encoded = self - .query(1, 1, move |connection| { - crate::primitives::maintenance::inspect_persisted_work(connection, role) - .map(PersistedWorkInventory::encode) - }) - .await?; - PersistedWorkInventory::decode(&encoded) - } - - /// Resolves a prior mutation without rerunning its handler. - pub async fn resolve( - &self, - identity: MutationIdentity, - operation_digest: Digest, - now_ms: i64, - max_result_bytes: usize, - ) -> crate::Result { - if identity.expired(now_ms)? { - return Ok(Resolution::Expired); - } - if self.admission.fenced.load(Ordering::Acquire) { - return Ok(Resolution::Unknown); - } - let admission = match self.reserve_work(48, max_result_bytes) { - Ok(admission) => admission, - Err(Error::Fenced) => return Ok(Resolution::Unknown), - Err(error) => return Err(error), - }; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Resolve(Box::new(QueuedResolve { - cell: self.cell, - admission: self.admission.clone(), - operation: ResolveOperation::Mutation { - identity, - operation_digest, - }, - now_ms, - max_result_bytes, - reply: Some(reply), - _work: admission, - }))) - .await - .map_err(|_| Error::RuntimeClosed)?; - match response.await { - Ok(result) => result, - Err(_) => Ok(Resolution::Unknown), - } - } - - /// Resolves one private destination effect without executing its handler again. - pub async fn resolve_effect( - &self, - delivery: InboxDelivery, - now_ms: i64, - max_result_bytes: usize, - ) -> crate::Result { - if delivery.expires_at_ms <= now_ms { - return Ok(Resolution::Expired); - } - if self.admission.fenced.load(Ordering::Acquire) { - return Ok(Resolution::Unknown); - } - let admission = match self.reserve_work(72, max_result_bytes) { - Ok(admission) => admission, - Err(Error::Fenced) => return Ok(Resolution::Unknown), - Err(error) => return Err(error), - }; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Resolve(Box::new(QueuedResolve { - cell: self.cell, - admission: self.admission.clone(), - operation: ResolveOperation::Effect { delivery }, - now_ms, - max_result_bytes, - reply: Some(reply), - _work: admission, - }))) - .await - .map_err(|_| Error::RuntimeClosed)?; - match response.await { - Ok(result) => result, - Err(_) => Ok(Resolution::Unknown), - } - } - - /// Drains the old capability and durably proves one registry-verified schema step. - pub async fn migrate(&self, plan: MigrationPlan, now_ms: i64) -> crate::Result { - if plan.from_code() != self.code - || plan.from_schema() != self.schema - || plan.operation_bytes() > MAX_OPERATION_BYTES - { - return Err(Error::Registry("migration plan does not match Cell handle")); - } - let work = self.reserve_work(plan.operation_bytes(), 0)?; - if self - .admission - .draining - .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) - .is_err() - { - return Err(Error::CellDraining); - } - self.admission.requests.close(); - self.admission.bytes.close(); - let successor_admission = new_cell_admission(); - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Migrate(Box::new(QueuedMigration { - cell: self.cell, - admission: self.admission.clone(), - plan, - now_ms, - reply: Some(reply), - successor_admission, - _work: work, - }))) - .await - .map_err(|_| Error::RuntimeClosed)?; - let migrated = response.await.map_err(|_| Error::Fenced)??; - Ok(MigratedCell { - handle: Self { - cell: self.cell, - incarnation: self.incarnation, - code: migrated.outcome.code, - schema: migrated.outcome.schema, - catalog: self.catalog.clone(), - inner: self.inner.clone(), - admission: migrated.admission, - }, - outcome: migrated.outcome, - }) - } - - /// Stops admission, publishes accepted commands, then closes the SQLite handle. - pub async fn drain(&self) -> crate::Result<()> { - if self - .admission - .draining - .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) - .is_err() - { - return Err(Error::CellDraining); - } - self.admission.requests.close(); - self.admission.bytes.close(); - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Drain { - cell: self.cell, - admission: self.admission.clone(), - reply, - }) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::RuntimeClosed)? - } - - pub(super) fn reserve_work( - &self, - operation_bytes: usize, - max_result_bytes: usize, - ) -> crate::Result { - if self.inner.shutting_down.load(Ordering::Acquire) { - return Err(Error::RuntimeClosed); - } - self.inner.node_lease.check()?; - if operation_bytes > MAX_OPERATION_BYTES || max_result_bytes > MAX_RESULT_BYTES { - return Err(Error::Capacity("operation or result bytes")); - } - let reservation_bytes = operation_bytes - .checked_add(max_result_bytes) - .filter(|bytes| *bytes != 0) - .ok_or(Error::Capacity("mailbox bytes"))?; - if self.admission.fenced.load(Ordering::Acquire) { - return Err(Error::Fenced); - } - if self.admission.draining.load(Ordering::Acquire) { - return Err(Error::CellDraining); - } - let admission = WorkAdmission { - _request: try_one(self.admission.requests.clone(), "Cell mailbox requests")?, - _cell_bytes: try_many( - self.admission.bytes.clone(), - reservation_bytes, - "Cell mailbox bytes", - )?, - _node_bytes: self - .inner - .resources - .try_reserve(ResourceCost::zero().with_retained_bytes(reservation_bytes)) - .map_err(|error| match error { - Error::Capacity(_) => Error::Capacity("node retained bytes"), - error => error, - })?, - }; - if self.admission.fenced.load(Ordering::Acquire) { - return Err(Error::Fenced); - } - if self.admission.draining.load(Ordering::Acquire) { - return Err(Error::CellDraining); - } - self.inner.node_lease.check()?; - Ok(admission) - } -} - -pub(super) fn try_one( - semaphore: Arc, - resource: &'static str, -) -> crate::Result { - semaphore - .try_acquire_owned() - .map_err(|error| admission_error(error, resource)) -} - -pub(super) fn try_many( - semaphore: Arc, - permits: usize, - resource: &'static str, -) -> crate::Result { - let permits = u32::try_from(permits).map_err(|_| Error::Capacity(resource))?; - semaphore - .try_acquire_many_owned(permits) - .map_err(|error| admission_error(error, resource)) -} - -pub(super) fn admission_error(error: TryAcquireError, resource: &'static str) -> Error { - match error { - TryAcquireError::Closed => Error::CellDraining, - TryAcquireError::NoPermits => Error::Capacity(resource), - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/lifecycle.rs b/crates/crab-cell-runtime/src/cell/actor/lifecycle.rs deleted file mode 100644 index c39fdc9e1..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/lifecycle.rs +++ /dev/null @@ -1,18 +0,0 @@ -//! Cell lifecycle scheduling for one Cell actor. -//! -//! Activation, bootstrap, background hydration/inventory/compaction, eviction, -//! renewals, transfer inspection, and deactivation all schedule the next step -//! the loop should take. - -use super::admission::{fence_active, finish_migration, send_migration_reply}; -use super::*; - -mod activation; -mod background; -mod eviction; -mod scheduling; - -pub(super) use activation::*; -pub(super) use background::*; -pub(super) use eviction::*; -pub(super) use scheduling::*; diff --git a/crates/crab-cell-runtime/src/cell/actor/lifecycle/activation.rs b/crates/crab-cell-runtime/src/cell/actor/lifecycle/activation.rs deleted file mode 100644 index b9fc0aac0..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/lifecycle/activation.rs +++ /dev/null @@ -1,171 +0,0 @@ -//! Activation, bootstrap publication, and failure cleanup for one Cell. - -use super::*; - -pub(in crate::cell::actor) async fn activate_restored_and_publish( - cell: CellId, - pool: &SqlWorkerPool, - publisher: &mut CellPublisher, - activation: RestoredActivation, -) -> crate::Result> { - let RestoredActivation { - database, - destination, - incarnation, - schema, - root, - reservation, - } = activation; - pool.activate_restored( - cell, - database, - destination, - incarnation, - schema, - root, - reservation, - ) - .await?; - if let Err(error) = publisher.activate().await { - return match pool.deactivate(cell).await { - Ok(()) => Err(error), - Err(cleanup) => Err(cleanup), - }; - } - pool.hydration(cell).await -} - -pub(in crate::cell::actor) async fn bootstrap_and_publish( - cell: CellId, - pool: &SqlWorkerPool, - publisher: &mut CellPublisher, - activation: BootstrapActivation, -) -> crate::Result> { - let BootstrapActivation { - replica, - destination, - incarnation, - schema, - initialize, - reservation, - } = activation; - let bootstrap = pool.bootstrap( - cell, - replica, - destination, - incarnation, - schema, - initialize, - reservation, - ); - tokio::pin!(bootstrap); - let mut renewal_error = None; - let bootstrap = loop { - tokio::select! { - result = &mut bootstrap => break result, - _ = tokio::time::sleep_until(tokio::time::Instant::from_std(publisher.renewal_at())) => { - if let Err(error) = publisher.renew().await { - renewal_error = Some(error); - break bootstrap.await; - } - } - } - }; - if let Some(error) = renewal_error { - return match bootstrap { - Ok(_) => match pool.deactivate(cell).await { - Ok(()) => Err(error), - Err(cleanup) => Err(cleanup), - }, - Err(bootstrap) => Err(bootstrap), - }; - } - let bootstrap = bootstrap?; - let publication = async { - let prepared = publisher.prepare_initial(&bootstrap.cuts).await?; - publisher - .publish_prepared(&prepared, bootstrap.next_due_ms) - .await?; - pool.confirm_bootstrap_published(cell, bootstrap.cuts) - .await?; - Ok(()) - } - .await; - if let Err(error) = publication { - return match pool.deactivate(cell).await { - Ok(()) => Err(error), - Err(cleanup) => Err(cleanup), - }; - } - Ok(None) -} - -pub(in crate::cell::actor) async fn cleanup_failed_activation( - cell: CellId, - pool: &SqlWorkerPool, - publisher: &mut CellPublisher, -) -> crate::Result<()> { - match pool.deactivate(cell).await { - Ok(()) | Err(Error::CellNotActive) => {} - Err(error) => return Err(error), - } - if publisher.control().value().root.is_some() { - publisher.release().await - } else { - Ok(()) - } -} - -pub(in crate::cell::actor) async fn rollback_failed_acquisition( - authority: &CellAuthority, - claimed: &VersionedControl, - replica: &crab_ltx::CellReplica, - node_lease: Option, -) -> crate::Result<()> { - let current = authority - .load(claimed.value().cell) - .await? - .ok_or(Error::Fenced)?; - if current.value().state == crate::control::ControlState::Idle - && current.value().owner.is_none() - { - return Ok(()); - } - if current.value().epoch != claimed.value().epoch - || current.value().owner != claimed.value().owner - || current.value().root != claimed.value().root - || current.value().recovery != claimed.value().recovery - || current.value().code != claimed.value().code - || current.value().schema != claimed.value().schema - || current.value().recovery.is_some() - { - return Ok(()); - } - let mut publisher = - CellPublisher::new(replica.clone(), authority.clone(), current, PathBuf::new()); - if let Some(node_lease) = node_lease { - publisher = publisher.with_node_lease(node_lease); - } - match publisher.release().await { - Ok(_) => Ok(()), - Err(error) => { - let latest = authority - .load(claimed.value().cell) - .await? - .ok_or(Error::Fenced)?; - if (latest.value().state == crate::control::ControlState::Idle - && latest.value().owner.is_none()) - || latest.value().epoch != claimed.value().epoch - || latest.value().owner != claimed.value().owner - || latest.value().root != claimed.value().root - || latest.value().recovery != claimed.value().recovery - || latest.value().code != claimed.value().code - || latest.value().schema != claimed.value().schema - { - Ok(()) - } else { - Err(error) - } - } - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/lifecycle/background.rs b/crates/crab-cell-runtime/src/cell/actor/lifecycle/background.rs deleted file mode 100644 index 1d4703c21..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/lifecycle/background.rs +++ /dev/null @@ -1,185 +0,0 @@ -//! Background hydration, inventory, and compaction starters. - -use super::*; - -pub(in crate::cell::actor) fn start_background_hydration( - pool: &SqlWorkerPool, - cells: &mut HashMap, - tasks: &mut JoinSet, - node_lease: &RuntimeNodeLease, -) { - if node_lease.check().is_err() { - for active in cells.values_mut() { - active.coordination.step(CoordinationInput::Fence); - fence_active(active); - } - return; - } - let resources = pool.resource_ledger(); - let now = std::time::Instant::now(); - let candidates = cells - .iter_mut() - .filter_map(|(cell, active)| { - if now < active.hydration_retry_at { - return None; - } - let reservation = resources - .try_reserve(ResourceCost::zero().with_hydration_jobs(1)) - .ok()?; - match active.coordination.step(CoordinationInput::BeginHydration { - queue_empty: active.queue.is_empty(), - publication_idle: active.coordination.publication_count() == 0, - lease_live: node_lease.check().is_ok(), - }) { - CoordinationDecision::Started => {} - CoordinationDecision::Fence => { - drop(reservation); - fence_active(active); - return None; - } - _ => { - drop(reservation); - return None; - } - } - let effect_id = active.begin_task(CoordinationEffect::Hydration); - Some((*cell, active.generation, effect_id, reservation)) - }) - .collect::>(); - - for (cell, generation, effect_id, reservation) in candidates { - let pool = pool.clone(); - tasks.spawn(async move { - let _reservation = reservation; - let deadline = std::time::Instant::now() + SQL_WALL_DEADLINE; - // The worker distinguishes abandoned fetches from uncertain local - // installation. An outer timeout would erase that safety boundary. - let result = pool.hydrate(cell, HYDRATION_PAGES_PER_STEP, deadline).await; - if result.is_err() { - let _ = pool.fence(cell).await; - } - TaskResult::Hydrated { - cell, - generation, - effect_id, - result, - } - }); - } -} - -pub(in crate::cell::actor) fn start_background_inventory( - pool: &SqlWorkerPool, - cells: &mut HashMap, - tasks: &mut JoinSet, - node_lease: &RuntimeNodeLease, -) { - let candidates = cells - .iter_mut() - .filter_map(|(cell, active)| { - let decision = active.coordination.step(CoordinationInput::BeginInventory { - queue_empty: active.queue.is_empty(), - publication_idle: active.coordination.publication_count() == 0, - inventory_unknown: active.persisted_work.is_unknown(), - refreshing: active.inventory_refreshing, - lease_live: node_lease.check().is_ok(), - }); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - return None; - } - if !matches!(decision, CoordinationDecision::Started) { - return None; - } - let effect_id = active.begin_task(CoordinationEffect::Inventory); - active.inventory_refreshing = true; - Some((*cell, active.generation, active.role, effect_id)) - }) - .collect::>(); - - for (cell, generation, role, effect_id) in candidates { - let pool = pool.clone(); - tasks.spawn(async move { - let deadline = std::time::Instant::now() + SQL_WALL_DEADLINE; - let result = - tokio::time::timeout_at(deadline.into(), pool.persisted_work_inventory(cell, role)) - .await - .map_err(|_| Error::Deadline) - .and_then(|result| result); - TaskResult::InventoryRefreshed { - cell, - generation, - effect_id, - result, - } - }); - } -} - -pub(in crate::cell::actor) fn start_background_compaction( - pool: &SqlWorkerPool, - cells: &mut HashMap, - tasks: &mut JoinSet, - node_lease: &RuntimeNodeLease, -) { - let now = std::time::Instant::now(); - for (cell, active) in cells { - if now.duration_since(active.last_work_at) < COMPACTION_QUIET - || now < active.compaction_retry_at - { - continue; - } - let due = active - .publisher - .as_ref() - .is_some_and(CellPublisher::compaction_due); - let decision = active - .coordination - .step(CoordinationInput::BeginCompaction { - queue_empty: active.queue.is_empty(), - publication_idle: active.coordination.publication_count() == 0, - publisher_ready: active.publisher.is_some(), - due, - lease_live: node_lease.check().is_ok(), - }); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - continue; - } - if !matches!(decision, CoordinationDecision::Started) { - continue; - } - let Some(mut publisher) = active.publisher.take() else { - active - .coordination - .step(CoordinationInput::FinishCompaction { fenced: true }); - fence_active(active); - continue; - }; - let cell = *cell; - let generation = active.generation; - let effect_id = active.begin_task(CoordinationEffect::Compaction); - let pool = pool.clone(); - tasks.spawn(async move { - let started = std::time::Instant::now(); - let result = publisher.compact_one_quiet().await; - tracing::debug!( - elapsed_ms = started.elapsed().as_millis(), - promoted = matches!(result, Ok(Some(true))), - retry = matches!(result, Ok(None)), - succeeded = result.is_ok(), - "Cell LTX quiet compaction completed" - ); - if result.is_err() { - let _ = pool.fence(cell).await; - } - TaskResult::Compacted { - cell, - generation, - effect_id, - publisher: Box::new(publisher), - result, - } - }); - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/lifecycle/eviction.rs b/crates/crab-cell-runtime/src/cell/actor/lifecycle/eviction.rs deleted file mode 100644 index 003ccb534..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/lifecycle/eviction.rs +++ /dev/null @@ -1,131 +0,0 @@ -//! Idle eviction, drain, and transfer observations for one Cell actor. - -use super::*; - -pub(in crate::cell::actor) fn start_bounded_evictions( - limit: usize, - now_ms: i64, - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - movement: &mut MovementBudget, - movement_permits: &mut HashMap, -) -> usize { - let mut started = 0; - for _ in 0..limit { - let Ok(permit) = movement.try_start(now_ms) else { - break; - }; - let mut selected = begin_idle_evictions(1, pool, cells, transitioning, tasks); - let Some(cell) = selected.pop() else { - let mut permit = permit; - movement.complete(&mut permit); - break; - }; - movement_permits.insert(cell, permit); - started += 1; - } - started -} - -pub(in crate::cell::actor) fn begin_idle_evictions( - limit: usize, - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, -) -> Vec { - let observations = cells - .iter() - .map(|(cell, active)| eviction_observation(*cell, active)) - .collect::>(); - let victims = select_victims(&observations, limit); - let mut started = Vec::new(); - for cell in victims { - if begin_idle_cell_eviction(cell, pool, cells, transitioning, tasks, None) { - started.push(cell); - } - } - started -} - -pub(in crate::cell::actor) fn eviction_observation( - cell: CellId, - active: &ActiveCell, -) -> EvictionObservation { - EvictionObservation { - cell, - state: if active.draining() { - EvictionState::Quiescing - } else { - EvictionState::Idle - }, - last_used_ms: active.last_used_ms, - cost: ResourceCost::active_cell() - .with_retained_bytes(usize::try_from(active.publication_bytes).unwrap_or(usize::MAX)), - busy: active.busy() || active.renewing() || active.transfer.is_some(), - retained_obligation: active.coordination.publication_count() != 0, - migrating: active.publisher.is_none(), - backup_pinned: active.unpublished_node_logs != 0, - leased_work: !active.queue.is_empty(), - primitive_obligation: !active.persisted_work.is_transfer_settled(), - accounting_known: !active.persisted_work.is_unknown(), - } -} - -pub(in crate::cell::actor) fn transfer_candidate_observation( - cell: CellId, - active: &ActiveCell, -) -> EvictionObservation { - let mut observation = eviction_observation(cell, active); - if !active.persisted_work.is_unknown() { - observation.primitive_obligation = false; - } - observation -} - -pub(in crate::cell::actor) fn transfer_observation( - cell: CellId, - active: &ActiveCell, - inventory: crate::primitives::maintenance::TransferWorkInventory, -) -> EvictionObservation { - let mut observation = eviction_observation(cell, active); - observation.primitive_obligation = !inventory.is_settled(); - observation.accounting_known = true; - observation -} - -pub(in crate::cell::actor) fn begin_idle_cell_eviction( - cell: CellId, - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - reply: Option>>, -) -> bool { - let Some(active) = cells.get_mut(&cell) else { - if let Some(reply) = reply { - let _ = reply.send(Err(Error::CellNotActive)); - } - return false; - }; - let decision = active.coordination.step(CoordinationInput::BeginDrain); - if matches!(decision, CoordinationDecision::Reject(_)) { - if let Some(reply) = reply { - let _ = reply.send(Err(Error::CellDraining)); - } - return false; - } - active.admission.draining.store(true, Ordering::Release); - active.admission.requests.close(); - active.admission.bytes.close(); - active.drain = reply; - if matches!( - schedule(active, true), - CoordinationDecision::ReadyToDeactivate - ) { - start_deactivate(cell, pool, cells, transitioning, tasks); - } - true -} diff --git a/crates/crab-cell-runtime/src/cell/actor/lifecycle/scheduling.rs b/crates/crab-cell-runtime/src/cell/actor/lifecycle/scheduling.rs deleted file mode 100644 index 3cbb16421..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/lifecycle/scheduling.rs +++ /dev/null @@ -1,331 +0,0 @@ -//! Coordination decisions that schedule the loop's next step. - -use super::*; - -pub(in crate::cell::actor) fn schedule( - active: &mut ActiveCell, - lease_live: bool, -) -> CoordinationDecision { - let publication_blocked = active.queue.front().is_some_and(|work| { - matches!(work, QueuedWork::Command(_)) - && (active.coordination.publication_count() >= MAX_PENDING_PUBLICATIONS - || active.publication_bytes >= PENDING_PUBLICATION_HIGH_WATER_BYTES) - }); - active.coordination.step(CoordinationInput::Schedule { - queue_empty: active.queue.is_empty(), - publisher_ready: active.publisher.is_some(), - publication_blocked, - lease_live, - }) -} - -pub(in crate::cell::actor) fn start_next( - active: &mut ActiveCell, - pool: &SqlWorkerPool, - tasks: &mut JoinSet, - node_lease: &RuntimeNodeLease, -) { - let decision = schedule(active, node_lease.check().is_ok()); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - return; - } - if !matches!(decision, CoordinationDecision::StartQueuedWork) { - return; - } - let Some(work) = active.queue.pop_front() else { - return; - }; - if matches!(&work, QueuedWork::Command(_) | QueuedWork::Migration(_)) { - // Durable command outcomes, effects, Queue rows, and Workflow runs - // remain release obligations until a fresh inventory proves otherwise. - active.persisted_work = crate::primitives::maintenance::PersistedWorkInventory::unknown(); - } - let generation = active.generation; - active.last_used_ms = unix_millis(); - active.last_work_at = std::time::Instant::now(); - let kind = match &work { - QueuedWork::Command(_) => AdmissionKind::Command, - QueuedWork::Query(_) => AdmissionKind::Query, - QueuedWork::Resolve(_) => AdmissionKind::Resolve, - QueuedWork::Migration(_) => AdmissionKind::Migration, - }; - if !matches!( - active.coordination.step(CoordinationInput::BeginWork { - kind, - publisher_ready: active.publisher.is_some(), - }), - CoordinationDecision::Started - ) { - active.queue.push_front(work); - return; - } - let pool = pool.clone(); - let interrupt = active.interrupt.clone(); - match work { - QueuedWork::Command(command) => { - let durability = active.durability_submitter.clone(); - let effect_id = active.begin_task(CoordinationEffect::Work(kind)); - tasks.spawn(async move { - execute_command(pool, durability, command, interrupt, generation, effect_id).await - }); - } - QueuedWork::Query(query) => { - let effect_id = active.begin_task(CoordinationEffect::Work(kind)); - tasks.spawn(async move { - execute_query(pool, query, interrupt, generation, effect_id).await - }); - } - QueuedWork::Resolve(resolve) => { - let effect_id = active.begin_task(CoordinationEffect::Work(kind)); - tasks.spawn(async move { - execute_resolve(pool, resolve, interrupt, generation, effect_id).await - }); - } - QueuedWork::Migration(migration) => { - let Some(publisher) = active.publisher.take() else { - let mut migration = migration; - send_migration_reply(&mut migration, Err(Error::Fenced)); - finish_migration(active, true); - return; - }; - let effect_id = active.begin_task(CoordinationEffect::Work(kind)); - tasks.spawn(async move { - execute_migration( - pool, - Box::new(publisher), - migration, - interrupt, - generation, - effect_id, - ) - .await - }); - } - } -} -pub(in crate::cell::actor) fn start_transfer_inspection( - cell: CellId, - pool: &SqlWorkerPool, - cells: &mut HashMap, - tasks: &mut JoinSet, - node_lease: &RuntimeNodeLease, -) { - let Some(active) = cells.get_mut(&cell) else { - return; - }; - if active.transfer.is_none() - || active.inventory_refreshing - || !active.queue.is_empty() - || !active.coordination.can_deactivate() - { - return; - } - if node_lease.check().is_err() { - fence_active(active); - return; - } - let generation = active.generation; - let role = active.role; - let effect_id = active.begin_task(CoordinationEffect::Inventory); - active.inventory_refreshing = true; - let pool = pool.clone(); - tasks.spawn(async move { - let deadline = std::time::Instant::now() + SQL_WALL_DEADLINE; - let result = tokio::time::timeout_at( - deadline.into(), - pool.transfer_work_inventory(cell, role, unix_millis()), - ) - .await - .map_err(|_| Error::Deadline) - .and_then(|result| result); - TaskResult::TransferPreflight { - cell, - generation, - effect_id, - result, - } - }); -} - -pub(in crate::cell::actor) fn start_due_renewals( - pool: &SqlWorkerPool, - cells: &mut HashMap, - tasks: &mut JoinSet, - node_lease: &RuntimeNodeLease, -) { - let active_renewals = cells.values().filter(|active| active.renewing()).count(); - let mut available = MAX_RENEWALS_IN_FLIGHT.saturating_sub(active_renewals); - if available == 0 { - return; - } - let now = std::time::Instant::now(); - for (cell, active) in cells.iter_mut() { - if available == 0 { - break; - } - if active - .publisher - .as_ref() - .is_none_or(|publisher| !publisher.renewal_due(now)) - { - continue; - } - match active.coordination.step(CoordinationInput::BeginRenewal { - queue_empty: active.queue.is_empty(), - publication_idle: active.coordination.publication_count() == 0, - lease_live: node_lease.check().is_ok(), - }) { - CoordinationDecision::Started => {} - CoordinationDecision::Fence => { - fence_active(active); - continue; - } - _ => continue, - } - let Some(mut publisher) = active.publisher.take() else { - active - .coordination - .step(CoordinationInput::FinishRenewal { fenced: true }); - continue; - }; - let generation = active.generation; - let effect_id = active.begin_task(CoordinationEffect::Renewal); - available -= 1; - let cell = *cell; - let pool = pool.clone(); - tasks.spawn(async move { - let result = publisher.renew().await; - if result.is_err() { - let _ = pool.fence(cell).await; - } - TaskResult::Renewed { - cell, - generation, - effect_id, - publisher: Box::new(publisher), - result, - } - }); - } -} - -pub(in crate::cell::actor) fn start_deactivate( - cell: CellId, - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, -) { - let Some(active) = cells.remove(&cell) else { - return; - }; - transitioning.insert(cell); - let generation = active.generation; - let pool = pool.clone(); - tasks.spawn(async move { - let result = async { - let mut publisher = active.publisher.ok_or(Error::Fenced)?; - // A release that already published its root may leave a resume - // record, so the next same-node activation can continue the local - // database instead of restoring the exact root again. - let control = publisher.control().value().clone(); - match control.root { - Some(root) => pool.deactivate_resumable(cell, root, control.code).await?, - None => pool.deactivate(cell).await?, - } - publisher.release().await?; - // A released Cell that still has a deadline publishes one bounded - // hint key, so the scheduler finds it without scanning every shard. - // The hint is an accelerator: a failed write costs a later Tick - // through the backstop and never a missed deadline. - if let Some(due_ms) = control.next_due_ms - && let Err(error) = crate::cell::due::publish( - publisher.layout(), - control.cell, - due_ms, - unix_millis(), - ) - .await - { - tracing::debug!(error = %error, "Cell due hint was not published"); - } - Ok(()) - } - .await; - TaskResult::Deactivated { - cell, - generation, - reply: active.drain, - shutdown_drain: active.coordination.is_shutdown(), - result, - } - }); -} - -pub(in crate::cell::actor) fn start_fenced_deactivate( - cell: CellId, - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - preserve_owner: bool, -) { - let Some(active) = cells.remove(&cell) else { - return; - }; - transitioning.insert(cell); - let generation = active.generation; - let pool = pool.clone(); - tasks.spawn(async move { - let result = async { - pool.discard(cell).await?; - // An unpublished node-log cut must keep its owner record so - // takeover seals and replays it instead of treating the Cell as idle. - if preserve_owner { - return Ok(()); - } - let mut publisher = active.publisher.ok_or(Error::Fenced)?; - publisher.release_after_fence().await - } - .await; - TaskResult::Deactivated { - cell, - generation, - reply: active.drain, - shutdown_drain: active.coordination.is_shutdown(), - result, - } - }); -} - -pub(in crate::cell::actor) fn start_orphan_deactivate( - cell: CellId, - pool: &SqlWorkerPool, - mut publisher: CellPublisher, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - shutdown_drain: bool, - generation: u64, -) { - let pool = pool.clone(); - tasks.spawn(async move { - let result = async { - let control = publisher.control().value().clone(); - match control.root { - Some(root) => pool.deactivate_resumable(cell, root, control.code).await?, - None => pool.deactivate(cell).await?, - } - publisher.release().await - } - .await; - TaskResult::Deactivated { - cell, - generation, - reply: None, - shutdown_drain, - result, - } - }); - transitioning.insert(cell); -} diff --git a/crates/crab-cell-runtime/src/cell/actor/requests.rs b/crates/crab-cell-runtime/src/cell/actor/requests.rs deleted file mode 100644 index a44018bff..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/requests.rs +++ /dev/null @@ -1,730 +0,0 @@ -//! Request execution for one Cell actor. -//! -//! Every request path runs on the actor's own task: it drives one queued -//! command, query, migration, or resolve to a terminal coordination decision, -//! publishes the outcome, and hands the cell back through `continue_cell`. - -use super::admission::{ - fence_active, fence_admission, send_command_reply, send_migration_reply, send_query_reply, - send_resolve_reply, -}; -use super::*; -use tracing::Instrument as _; - -pub(super) async fn execute_migration( - pool: SqlWorkerPool, - mut publisher: Box, - mut migration: Box, - interrupt: Arc, - generation: u64, - effect_id: u64, -) -> TaskResult { - let mut preserve_owner = false; - let mut unpublished_bytes = 0; - let deadline = SqlDeadline::new(std::time::Instant::now() + SQL_WALL_DEADLINE); - let operation = pool.migrate( - migration.cell, - migration.plan, - migration.now_ms, - deadline.clone(), - ); - tokio::pin!(operation); - let pending = match tokio::time::timeout_at(deadline.at().into(), &mut operation).await { - Ok(result) => result, - Err(_) => { - deadline.cancel_queued(); - // Migration has already closed the old capability. Even a queued - // timeout must recover ownership before exposing a usable handle. - interrupt.interrupt(); - fence_admission(&migration.admission); - send_migration_reply(&mut migration, Err(Error::Deadline)); - let _ = operation.await; - let _ = pool.fence(migration.cell).await; - return TaskResult::Migrated { - cell: migration.cell, - generation, - effect_id, - publisher, - migration, - result: Err(Error::Deadline), - fenced: true, - preserve_owner: false, - unpublished_bytes: 0, - }; - } - }; - let result = match pending { - Ok(pending) => { - let durability_started = std::time::Instant::now(); - let durability = publisher.submit_migration_durability(&pending).await; - match durability { - Err(error) => Err(error), - Ok(durability) => { - let node_logged = durability.is_some(); - let retained_bytes = pending.retained_bytes(); - let cell = migration.cell; - let early_outcome = MigrationOutcome { - code: pending.code(), - schema: pending.to_schema(), - commit_sequence: pending.commit_sequence(), - }; - let object = async { - let prepared = publisher.prepare_migration(&pending).await?; - pool.bind_migration_prepared(cell, prepared.clone()).await?; - let root = publisher - .publish_migration( - &prepared, - pending.next_due_ms(), - pending.code(), - pending.to_schema(), - ) - .await?; - if let Some(durability) = durability.as_ref() { - durability.prove_object().await?; - } else { - publisher.record_object_proof(durability_started.elapsed()); - } - pool.confirm_migration_published(cell, root).await - }; - tokio::pin!(object); - let result = match durability.as_ref() { - Some(durability) => { - let fleet = durability.prove_fleet(); - tokio::pin!(fleet); - tokio::select! { - result = &mut object => result, - fleet = &mut fleet => { - if fleet.is_ok() { - let admission = Arc::clone(&migration.successor_admission); - send_migration_reply( - &mut migration, - Ok(MigratedAdmission { - admission, - outcome: early_outcome, - }), - ); - } - object.await - } - } - } - None => object.await, - }; - preserve_owner = node_logged && result.is_err(); - if preserve_owner { - unpublished_bytes = retained_bytes; - } - result - } - } - } - Err(error) => Err(error), - }; - let fenced = result.is_err(); - if fenced { - let _ = pool.fence(migration.cell).await; - } - TaskResult::Migrated { - cell: migration.cell, - generation, - effect_id, - publisher, - migration, - result, - fenced, - preserve_owner, - unpublished_bytes, - } -} - -pub(super) async fn execute_command( - pool: SqlWorkerPool, - durability: CellDurabilitySubmitter, - mut command: Box, - interrupt: Arc, - generation: u64, - effect_id: u64, -) -> TaskResult { - let execution_started = std::time::Instant::now(); - tracing::debug!( - target: "crab_cell_runtime::action", - parent: &command.trace, - event = "cell_execution_started", - actor_queue_us = command.queued_at.elapsed().as_micros(), - ); - let deadline = SqlDeadline::new(std::time::Instant::now() + SQL_WALL_DEADLINE); - let execution = match command.handler.take() { - Some(handler) => { - let cell = command.cell; - let queued_operation = command.operation; - let now_ms = command.now_ms; - let max_result_bytes = command.max_result_bytes; - let worker_pool = pool.clone(); - let worker_deadline = deadline.clone(); - let operation = async move { - match queued_operation { - QueuedOperation::Mutation { - identity, - operation_digest, - } => { - worker_pool - .execute_until( - cell, - identity, - operation_digest, - now_ms, - max_result_bytes, - worker_deadline, - handler, - ) - .await - } - QueuedOperation::Effect { delivery } => { - worker_pool - .deliver_effect( - cell, - delivery, - now_ms, - max_result_bytes, - worker_deadline, - handler, - ) - .await - } - } - }; - let operation = operation.instrument(command.trace.clone()); - tokio::pin!(operation); - match tokio::time::timeout_at(deadline.at().into(), &mut operation).await { - Ok(result) => result, - Err(_) => { - let fenced = !deadline.cancel_queued(); - tracing::warn!(cell = ?command.cell, sql_started = fenced, "Cell SQL command deadline expired"); - if fenced { - interrupt.interrupt(); - fence_admission(&command.admission); - } - let error = if fenced { - command.operation.unknown(Error::Deadline) - } else { - Error::Deadline - }; - send_command_reply(&mut command, Err(error)); - let _ = operation.await; - if fenced { - let _ = pool.fence(command.cell).await; - } - return TaskResult::Executed { - cell: command.cell, - generation, - effect_id, - command, - result: Err(Error::Deadline), - fenced, - }; - } - } - } - None => Err(Error::Fenced), - }; - tracing::debug!( - target: "crab_cell_runtime::action", - parent: &command.trace, - event = "cell_execution_completed", - worker_round_trip_us = execution_started.elapsed().as_micros(), - succeeded = execution.is_ok(), - ); - let (result, must_fence) = match execution { - Ok(WorkerExecution::Recorded(outcome)) => (Ok(CommandTaskResult::Recorded(outcome)), false), - Ok(WorkerExecution::Pending(pending)) => { - let retained_bytes = usize::try_from(pending.retained_bytes()) - .map_err(|_| Error::Capacity("pending publication bytes")); - let result = retained_bytes.and_then(|retained_bytes| { - pool.resource_ledger() - .try_reserve(ResourceCost::zero().with_retained_bytes(retained_bytes)) - .map_err(|error| match error { - Error::Capacity(_) => Error::Capacity("pending publication bytes"), - error => error, - }) - }); - let result = match result { - Ok(retained_reservation) => durability - .submit(pending.outcome().commit_sequence(), pending.cuts()) - .await - .map(|durability| CommandTaskResult::Pending { - pending, - durability, - retained_reservation, - }), - Err(error) => Err(error), - }; - (result, true) - } - Err(error) => ( - Err(error), - !deadline.cancelled() - && !matches!(pool.state(command.cell).await, Ok(WorkerState::Ready)), - ), - }; - let fenced = must_fence && result.is_err(); - if fenced { - tracing::warn!(cell = ?command.cell, error = ?result.as_ref().err(), "Cell command execution fenced its owner"); - let _ = pool.fence(command.cell).await; - } - let result = if fenced { - result.map_err(|source| command.operation.unknown(source)) - } else { - result - }; - TaskResult::Executed { - cell: command.cell, - generation, - effect_id, - command, - result, - fenced, - } -} - -pub(super) async fn prove_command( - pool: SqlWorkerPool, - mut command: Box, - outcome: StoredOutcome, - commit_sequence: u64, - durability: Option, - mut object: oneshot::Receiver>, - generation: u64, - effect_id: u64, -) -> TaskResult { - use crate::node::log::DurabilitySource; - - let proof_started = std::time::Instant::now(); - let proof = match durability { - Some(durability) => { - let fleet_or_object = durability.prove(); - tokio::pin!(fleet_or_object); - tokio::select! { - object = &mut object => match receive_publication_proof(object) { - Ok(()) => Ok(DurabilitySource::Object), - // Object publication failure does not invalidate an - // independently fsynced follower proof for this cut. - Err(_) => fleet_or_object.await, - }, - result = &mut fleet_or_object => match result { - Ok(source) => Ok(source), - // Losing the follower path does not invalidate the same - // cut's object publication, which remains the fallback. - Err(_) => receive_publication_proof(object.await).map(|()| DurabilitySource::Object), - } - } - } - None => receive_publication_proof(object.await).map(|()| DurabilitySource::Object), - }; - tracing::debug!( - target: "crab_cell_runtime::action", - parent: &command.trace, - event = "cell_proof_completed", - proof_wait_us = proof_started.elapsed().as_micros(), - commit_sequence, - succeeded = proof.is_ok(), - ); - let result = match proof { - Ok(source) => { - let confirmation = std::time::Instant::now(); - pool.confirm_durable(command.cell, commit_sequence) - .await - .map(|()| { - command.response_proof = Some((source, confirmation.elapsed())); - outcome - }) - } - Err(error) => Err(error), - }; - let fenced = result.is_err(); - if fenced { - let _ = pool.fence(command.cell).await; - } - let result = if fenced { - result.map_err(|source| command.operation.unknown(source)) - } else { - result - }; - TaskResult::Proven { - cell: command.cell, - generation, - effect_id, - command, - result, - fenced, - } -} - -pub(super) fn receive_publication_proof( - result: std::result::Result, oneshot::error::RecvError>, -) -> crate::Result<()> { - result.map_err(|_| Error::RuntimeClosed)? -} - -pub(super) fn start_publication( - cell: CellId, - active: &mut ActiveCell, - pool: &SqlWorkerPool, - tasks: &mut JoinSet, -) { - let Some(mut publisher) = active.publisher.take() else { - return; - }; - let Some(publication) = active.publications.pop_front() else { - active.publisher = Some(publisher); - return; - }; - let node_logged = publication.durability.is_some(); - let root_sequence_lag = i128::from(publication.pending.outcome().commit_sequence()) - - i128::from(active.published_sequence); - let incarnation = active.incarnation; - let commit_sequence = publication.pending.outcome().commit_sequence(); - tracing::debug!( - target: "crab_cell_runtime::action", - event = "cell_publication_started", - cell = ?cell, - incarnation = ?incarnation, - commit_sequence, - queue_wait_ms = publication.submitted_at.elapsed().as_millis(), - pending_publications = active.coordination.publication_count(), - publication_bytes = active.publication_bytes, - root_sequence_lag = %root_sequence_lag, - "Cell LTX publication started" - ); - let generation = active.generation; - let effect_id = active.begin_task(CoordinationEffect::Publication); - // Moving the publisher out of ActiveCell is the serialization token for - // root preparation and CAS; no second object publisher can overtake it. - let pool = pool.clone(); - let retained_reservation = publication.retained_reservation; - let published_next_due_ms = publication.pending.next_due_ms(); - let published_commit_sequence = publication.pending.outcome().commit_sequence(); - tasks.spawn(async move { - let _retained_reservation = retained_reservation; - let retained_bytes = publication.pending.retained_bytes(); - let mut publication_proof = Some(publication.proof); - let fleet_deadline = std::time::Instant::now() + FLEET_PUBLICATION_GRACE; - let mut retry_delay = std::time::Duration::from_millis(100); - let result = async { - let expected = publication.pending.outcome().clone(); - let prepared = loop { - match publisher.prepare(&publication.pending).await { - Ok(prepared) => break prepared, - Err(error) if node_logged && is_storage_publication_error(&error) => { - if let Some(durability) = publication.durability.as_ref() { - wait_for_fleet_proof(durability, fleet_deadline).await?; - } - if let Some(proof) = publication_proof.take() { - let _ = proof.send(Err(error)); - } - if std::time::Instant::now() >= fleet_deadline { - return Err(Error::Fenced); - } - tokio::time::sleep(retry_delay).await; - retry_delay = retry_delay - .saturating_mul(2) - .min(std::time::Duration::from_secs(2)); - } - Err(error) => return Err(error), - } - }; - pool.bind_prepared(cell, prepared.clone()).await?; - let root = loop { - match publisher - .publish_prepared(&prepared, publication.pending.next_due_ms()) - .await - { - Ok(root) => break root, - Err(error) if node_logged && is_storage_publication_error(&error) => { - if let Some(durability) = publication.durability.as_ref() { - wait_for_fleet_proof(durability, fleet_deadline).await?; - } - if let Some(proof) = publication_proof.take() { - let _ = proof.send(Err(error)); - } - if std::time::Instant::now() >= fleet_deadline { - return Err(Error::Fenced); - } - tokio::time::sleep(retry_delay).await; - retry_delay = retry_delay - .saturating_mul(2) - .min(std::time::Duration::from_secs(2)); - } - Err(error) => return Err(error), - } - }; - if let Some(durability) = publication.durability.as_ref() { - durability.prove_object().await?; - } else { - publisher.record_object_proof(publication.submitted_at.elapsed()); - } - let published = pool.confirm_published(cell, root).await?; - if published != expected { - return Err(Error::Control( - "published result does not match queued commit", - )); - } - Ok(()) - } - .await; - tracing::debug!( - target: "crab_cell_runtime::action", - event = "cell_publication_completed", - cell = ?cell, - incarnation = ?incarnation, - commit_sequence, - publication_lag_ms = publication.submitted_at.elapsed().as_millis(), - succeeded = result.is_ok(), - "Cell LTX publication completed" - ); - let fenced = result.is_err(); - if let Err(error) = &result { - // The proof waiter receives an unknown outcome. Retain the cause - // here so an operator can distinguish storage failure from fencing. - tracing::warn!(cell = ?cell, commit_sequence, error = ?error, "Cell publication fenced its owner"); - } - if let Some(proof) = publication_proof { - let _ = proof.send(if fenced { Err(Error::Fenced) } else { Ok(()) }); - } - if fenced { - let _ = pool.fence(cell).await; - } - TaskResult::Published { - cell, - generation, - effect_id, - publisher: Box::new(publisher), - retained_bytes, - node_logged, - next_due_ms: published_next_due_ms, - commit_sequence: published_commit_sequence, - result, - fenced, - } - }); -} - -pub(super) fn is_storage_publication_error(error: &Error) -> bool { - matches!( - error, - Error::Storage(_) | Error::Ltx(crab_ltx::CrabError::Storage(_)) - ) -} - -pub(super) async fn wait_for_fleet_proof( - durability: &PendingDurability, - deadline: std::time::Instant, -) -> crate::Result<()> { - tokio::time::timeout_at( - tokio::time::Instant::from_std(deadline), - durability.prove_fleet(), - ) - .await - .map_err(|_| Error::Fenced)? -} - -pub(super) async fn execute_query( - pool: SqlWorkerPool, - mut query: Box, - interrupt: Arc, - generation: u64, - effect_id: u64, -) -> TaskResult { - let deadline = SqlDeadline::new(std::time::Instant::now() + SQL_WALL_DEADLINE); - let result = match query.handler.take() { - Some(handler) => { - let operation = pool.query( - query.cell, - query.max_result_bytes, - deadline.clone(), - handler, - ); - tokio::pin!(operation); - match tokio::time::timeout_at(deadline.at().into(), &mut operation).await { - Ok(result) => result, - Err(_) => { - let fenced = !deadline.cancel_queued(); - tracing::warn!(cell = ?query.cell, sql_started = fenced, "Cell SQL query deadline expired"); - if fenced { - interrupt.interrupt(); - fence_admission(&query.admission); - } - send_query_reply(&mut query, Err(Error::Deadline)); - let _ = operation.await; - if fenced { - let _ = pool.fence(query.cell).await; - } - return TaskResult::Queried { - cell: query.cell, - generation, - effect_id, - query, - result: Err(Error::Deadline), - fenced, - }; - } - } - } - None => Err(Error::Fenced), - }; - let fenced = result.is_err() - && !deadline.cancelled() - && !matches!(pool.state(query.cell).await, Ok(WorkerState::Ready)); - if fenced { - let _ = pool.fence(query.cell).await; - } - TaskResult::Queried { - cell: query.cell, - generation, - effect_id, - query, - result, - fenced, - } -} - -pub(super) async fn execute_resolve( - pool: SqlWorkerPool, - mut resolve: Box, - interrupt: Arc, - generation: u64, - effect_id: u64, -) -> TaskResult { - let deadline = SqlDeadline::new(std::time::Instant::now() + SQL_WALL_DEADLINE); - let cell = resolve.cell; - let resolve_operation = resolve.operation; - let now_ms = resolve.now_ms; - let max_result_bytes = resolve.max_result_bytes; - let worker_pool = pool.clone(); - let worker_deadline = deadline.clone(); - let operation = async move { - match resolve_operation { - ResolveOperation::Mutation { - identity, - operation_digest, - } => { - worker_pool - .resolve( - cell, - identity, - operation_digest, - now_ms, - max_result_bytes, - worker_deadline, - ) - .await - } - ResolveOperation::Effect { delivery } => { - worker_pool - .resolve_effect(cell, delivery, now_ms, max_result_bytes, worker_deadline) - .await - } - } - }; - tokio::pin!(operation); - let result = match tokio::time::timeout_at(deadline.at().into(), &mut operation).await { - Ok(result) => result, - Err(_) => { - let fenced = !deadline.cancel_queued(); - tracing::warn!(cell = ?resolve.cell, sql_started = fenced, "Cell SQL resolution deadline expired"); - if fenced { - interrupt.interrupt(); - fence_admission(&resolve.admission); - } - send_resolve_reply(&mut resolve, Ok(Resolution::Unknown)); - let _ = operation.await; - if fenced { - let _ = pool.fence(resolve.cell).await; - } - return TaskResult::Resolved { - cell: resolve.cell, - generation, - effect_id, - resolve, - result: Ok(Resolution::Unknown), - fenced, - }; - } - }; - let timed_out = matches!(result, Err(Error::Deadline)); - let result = if timed_out { - Ok(Resolution::Unknown) - } else { - result - }; - let fenced = timed_out && !deadline.cancelled() - || result.is_err() && !matches!(pool.state(resolve.cell).await, Ok(WorkerState::Ready)); - if fenced { - let _ = pool.fence(resolve.cell).await; - } - TaskResult::Resolved { - cell: resolve.cell, - generation, - effect_id, - resolve, - result, - fenced, - } -} - -pub(super) fn continue_cell( - cell: CellId, - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - node_lease: &RuntimeNodeLease, -) { - if cells - .get(&cell) - .is_some_and(|active| active.transfer.is_some()) - { - let ready = cells.get(&cell).is_some_and(|active| { - !active.inventory_refreshing - && active.queue.is_empty() - && active.coordination.can_deactivate() - }); - if ready { - start_transfer_inspection(cell, pool, cells, tasks, node_lease); - return; - } - if let Some(active) = cells.get_mut(&cell) - && !active.inventory_refreshing - && !active.queue.is_empty() - { - start_next(active, pool, tasks, node_lease); - } - return; - } - let Some(active) = cells.get_mut(&cell) else { - return; - }; - let decision = schedule(active, node_lease.check().is_ok()); - match decision { - CoordinationDecision::ReadyToDeactivateFenced => { - let preserve_owner = active.unpublished_node_logs != 0; - start_fenced_deactivate(cell, pool, cells, transitioning, tasks, preserve_owner); - } - CoordinationDecision::ReadyToDeactivate => { - start_deactivate(cell, pool, cells, transitioning, tasks); - } - CoordinationDecision::StartQueuedWork => { - start_next(active, pool, tasks, node_lease); - } - CoordinationDecision::Ignored - | CoordinationDecision::Admit - | CoordinationDecision::ResolveUnknown - | CoordinationDecision::LocalHandle - | CoordinationDecision::Reject(_) - | CoordinationDecision::Started - | CoordinationDecision::EffectCompleted - | CoordinationDecision::StaleEffect => {} - CoordinationDecision::Fence => { - fence_active(active); - } - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/runtime.rs b/crates/crab-cell-runtime/src/cell/actor/runtime.rs deleted file mode 100644 index 2efada9dc..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/runtime.rs +++ /dev/null @@ -1,707 +0,0 @@ -//! Cell-runtime construction, configuration, and node administration. -//! -//! Everything here configures one runtime or reports and reserves against -//! the node ledger it owns; the activation and takeover paths stay in the -//! actor root. - -use super::*; - -/// Converts one node ledger snapshot into the pressure observation the actor -/// classifies. -/// -/// Memory is the resident and retained reservations against their own ceilings, -/// disk is the replica reservation against the replica budget, and jobs are the -/// same worker/primitive/hydration aggregate the placement block advertises, so -/// a node cannot look calm locally and pressed to the fleet. -pub(super) fn pressure_sample( - snapshot: ResourceSnapshot, - at_ms: i64, -) -> crate::Result { - let used = snapshot.used; - let limit = snapshot.limit; - let memory_used = used - .resident_bytes() - .checked_add(used.retained_bytes()) - .ok_or(Error::Capacity("pressure sample"))?; - let memory_limit = limit - .resident_bytes() - .checked_add(limit.retained_bytes()) - .ok_or(Error::Capacity("pressure sample"))?; - let memory_used = u64::try_from(memory_used).map_err(|_| Error::Capacity("pressure sample"))?; - let memory_limit = - u64::try_from(memory_limit).map_err(|_| Error::Capacity("pressure sample"))?; - let jobs_used = used - .worker_jobs() - .checked_add(used.primitive_jobs()) - .and_then(|jobs| jobs.checked_add(used.hydration_jobs())) - .ok_or(Error::Capacity("pressure sample"))?; - let jobs_limit = limit - .worker_jobs() - .checked_add(limit.primitive_jobs()) - .and_then(|jobs| jobs.checked_add(limit.hydration_jobs())) - .ok_or(Error::Capacity("pressure sample"))?; - let jobs_used = u64::try_from(jobs_used).map_err(|_| Error::Capacity("pressure sample"))?; - let jobs_limit = u64::try_from(jobs_limit).map_err(|_| Error::Capacity("pressure sample"))?; - Ok(PressureSample { - at_ms, - memory_used_permille: permille(memory_used, memory_limit)?, - disk_used_permille: permille(used.disk_bytes(), limit.disk_bytes())?, - jobs_used_permille: permille(jobs_used, jobs_limit)?, - // The ledger is read synchronously, so no sample can be late. - stale: false, - }) -} - -/// Scales `used / limit` to permille, saturating at full utilization. -fn permille(used: u64, limit: u64) -> crate::Result { - if limit == 0 { - return Err(Error::Capacity("pressure sample limit")); - } - let scaled = u128::from(used).saturating_mul(1_000) / u128::from(limit); - Ok(u16::try_from(scaled.min(1_000)).unwrap_or(1_000)) -} - -impl CellRuntime { - /// Starts one dispatcher on the current Tokio runtime. - pub fn new( - pool: SqlWorkerPool, - node_retained_bytes: usize, - session: SessionId, - ) -> crate::Result { - Self::new_with_replica_host( - pool, - node_retained_bytes, - session, - crab_ltx::Host::default(), - ) - } - - /// Starts one dispatcher with caller-sized shared replica job admission. - pub fn new_with_replica_host( - pool: SqlWorkerPool, - node_retained_bytes: usize, - session: SessionId, - replica_host: crab_ltx::Host, - ) -> crate::Result { - Self::new_inner( - pool, - node_retained_bytes, - session, - replica_host, - RuntimeNodeLease::ObjectOnly, - ) - } - - /// Starts one dispatcher that remains fenced until its node lease is installed. - pub fn new_with_replica_host_requiring_node_lease( - pool: SqlWorkerPool, - node_retained_bytes: usize, - session: SessionId, - replica_host: crab_ltx::Host, - ) -> crate::Result { - Self::new_inner( - pool, - node_retained_bytes, - session, - replica_host, - RuntimeNodeLease::Required(OnceLock::new()), - ) - } - - /// Returns the budget shared by this runtime's local replica artifacts. - /// - /// Recovery admission must use this same ledger so streamed bundle bytes - /// cannot bypass WAL, cache, or sparse-page reservations. - #[must_use] - pub fn local_disk_budget(&self) -> crab_ltx::DiskBudget { - self.inner.replica_host.local_disk_budget() - } - - fn new_inner( - pool: SqlWorkerPool, - node_retained_bytes: usize, - session: SessionId, - replica_host: crab_ltx::Host, - node_lease: RuntimeNodeLease, - ) -> crate::Result { - if node_retained_bytes == 0 || node_retained_bytes > Semaphore::MAX_PERMITS { - return Err(Error::Capacity("node retained bytes")); - } - pool.configure_retained_capacity(node_retained_bytes)?; - let resources = pool.resource_ledger(); - let primitive_jobs = Arc::new(Semaphore::new(resources.snapshot()?.limit.primitive_jobs())); - resources.set_disk_limit(replica_host.local_disk_capacity())?; - resources.set_host_limits( - replica_host.io_capacity(), - replica_host.job_capacity(), - replica_host.recovery_capacity(), - replica_host.dirty_capacity(), - replica_host.scratch_capacity() as usize, - )?; - let telemetry = crate::fleet::telemetry::CellTelemetryHandle::default(); - let mut replica_host = replica_host.with_ltx_telemetry(Arc::new(telemetry.clone())); - replica_host.install_resource_admission(Arc::new(LedgerHostResourceAdmission::new( - session, &resources, - ))); - replica_host - .install_disk_admission(Arc::new(LedgerDiskAdmission::new(session, &resources)))?; - let runtime = tokio::runtime::Handle::try_current().map_err(Error::RuntimeStart)?; - let (sender, receiver) = mpsc::channel(INGRESS_REQUESTS); - let publications = broadcast::Sender::new(PUBLICATION_NOTIFICATIONS); - let node_lease = Arc::new(node_lease); - let unpublished_node_log_bytes = Arc::new(AtomicU64::new(0)); - runtime.spawn(run( - receiver, - pool.clone(), - Arc::clone(&node_lease), - Arc::clone(&unpublished_node_log_bytes), - telemetry.clone(), - publications.clone(), - )); - Ok(Self { - inner: Arc::new(RuntimeInner { - sender, - publications, - resources, - primitive_jobs, - shutting_down: AtomicBool::new(false), - accepting_cells: AtomicBool::new(true), - session, - pool, - replica_host, - application_limits: OnceLock::new(), - node_lease, - node_durability: Arc::new(std::sync::RwLock::new(None)), - telemetry, - unpublished_node_log_bytes, - }), - }) - } - - /// Subscribes to bounded advisory hints after Cell activation or object publication. - /// - /// Hints grant no read or ownership capability. Receivers must reload authority - /// and periodically reconcile missed hints, including broadcast lag. Command - /// acknowledgement never waits for receivers; fleet-only proofs send no hint. - #[must_use] - pub fn subscribe_publications(&self) -> broadcast::Receiver { - self.inner.publications.subscribe() - } - - /// Binds declared per-namespace LTX limits before a compiled application starts Cell work. - /// - /// Every later activation must use matching database and capture bounds. - pub fn install_application_limits( - &self, - limits: impl IntoIterator, - ) -> crate::Result<()> { - let mut by_namespace = HashMap::new(); - for (namespace, database, capture) in limits { - if database < 512 - || capture < 128 - || by_namespace - .insert(namespace, (database, capture)) - .is_some() - { - return Err(Error::Control("invalid application Cell limits")); - } - } - if by_namespace.is_empty() { - return Err(Error::Control("application has no Cell types")); - } - self.inner - .application_limits - .set(by_namespace) - .map_err(|_| Error::Control("application Cell limits already installed")) - } - - /// Installs the process telemetry sink before Cell work begins. - pub fn install_telemetry( - &self, - telemetry: Arc, - ) -> crate::Result<()> { - self.inner.telemetry.install(telemetry) - } - - /// Returns the shared sink used by node-log components for this runtime. - #[must_use] - pub fn telemetry_handle(&self) -> crate::fleet::telemetry::CellTelemetryHandle { - self.inner.telemetry.clone() - } - - /// Lists resident Cells whose published due time has passed. - /// - /// The actor answers from its own map, so a scheduler can tick due work it - /// already owns without reading the Cell's catalog entry or control - /// record. Callers must treat the list as a hint: the Tick itself fences - /// against the publish sequence and re-derives what is due inside the - /// Cell's transaction. - pub async fn due_resident( - &self, - now_ms: i64, - limit: usize, - ) -> crate::Result> { - self.ensure_running()?; - if limit == 0 || now_ms < 0 { - return Ok(Vec::new()); - } - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::DueResident { - now_ms, - limit, - reply, - }) - .await - .map_err(|_| Error::RuntimeClosed)?; - let due = response.await.map_err(|_| Error::RuntimeClosed)?; - Ok(due - .into_iter() - .map(|cell| { - super::DueResident::new( - CellHandle { - cell: cell.cell, - incarnation: cell.incarnation, - code: cell.code, - schema: cell.schema, - catalog: cell.catalog, - inner: Arc::clone(&self.inner), - admission: cell.admission, - }, - cell.expected_commit_sequence, - cell.next_due_ms, - ) - }) - .collect()) - } - - /// Installs the successfully published process lease before Cell admission opens. - pub fn install_node_lease(&self, guard: NodeLeaseGuard) -> crate::Result<()> { - if self.inner.shutting_down.load(Ordering::Acquire) { - return Err(Error::RuntimeClosed); - } - self.inner.node_lease.install(guard) - } - - /// Installs the one recruited node-log epoch used by newly activated Cells. - pub fn install_node_durability( - &self, - application: ApplicationId, - durability: Arc, - ) -> crate::Result<()> { - self.ensure_running()?; - let mut slot = self - .inner - .node_durability - .write() - .map_err(|_| Error::Control("Cell runtime node durability lock poisoned"))?; - if slot.is_some() { - return Err(Error::Control( - "Cell runtime node durability was initialized twice", - )); - } - *slot = Some((application, durability)); - Ok(()) - } - - /// Returns the currently installed node-log durability binding. - #[must_use] - pub fn node_durability(&self) -> Option<(ApplicationId, Arc)> { - self.inner - .node_durability - .read() - .ok() - .and_then(|slot| slot.clone()) - } - - /// Replaces the active node-log durability binding after an epoch close. - pub fn replace_node_durability( - &self, - application: ApplicationId, - durability: Arc, - ) -> crate::Result> { - self.ensure_running()?; - let mut slot = self - .inner - .node_durability - .write() - .map_err(|_| Error::Control("Cell runtime node durability lock poisoned"))?; - let Some((installed_application, _)) = slot.as_ref() else { - return Err(Error::Control( - "Cell runtime node durability is not installed", - )); - }; - if *installed_application != application { - return Err(Error::Control( - "Cell runtime node durability application changed", - )); - } - let (_, previous) = slot - .replace((application, durability)) - .ok_or(Error::Control("Cell runtime node durability disappeared"))?; - Ok(previous) - } - - /// Stops admission, drains accepted work, closes every Cell, and releases ownership. - pub async fn shutdown(&self) -> crate::Result<()> { - if self - .inner - .shutting_down - .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) - .is_err() - { - return Err(Error::RuntimeClosed); - } - self.inner.primitive_jobs.close(); - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::Shutdown { reply }) - .await - .map_err(|_| Error::RuntimeClosed)?; - let drain = response.await.map_err(|_| Error::RuntimeClosed)?; - let workers = self.inner.pool.shutdown().await; - let durability = match self.node_durability() { - Some((_, durability)) => durability.shutdown().await, - None => Ok(()), - }; - // Admission and replica work are stopped. Optional fills outlive their - // readers, so keep artifacts/executors until accepted fills complete. - self.inner.replica_host.drain_cache_fills().await; - drain.and(workers).and(durability) - } - - /// Stops new Cell acquisition while existing owners continue serving. - pub fn stop_acquiring(&self) -> crate::Result<()> { - self.ensure_running()?; - self.inner.accepting_cells.store(false, Ordering::Release); - Ok(()) - } - - /// Reports whether a new owner may be acquired on this node. - #[must_use] - pub fn is_acquiring(&self) -> bool { - !self.inner.shutting_down.load(Ordering::Acquire) - && self.inner.accepting_cells.load(Ordering::Acquire) - } - - /// Starts bounded, actor-owned eviction of safe idle Cells. - /// - /// The returned count is the number of drains started. Resource - /// reservations are released only after the worker closes and ownership - /// release completes; callers must not treat this as immediate capacity. - pub async fn evict_idle(&self, limit: usize) -> crate::Result { - self.ensure_running()?; - if limit == 0 { - return Ok(0); - } - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::EvictIdle { limit, reply }) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::RuntimeClosed)? - } - - /// Lists currently settled local Cells as advisory transfer candidates. - /// The exact generation and eligibility are rechecked by `release_idle_cell`. - pub async fn idle_transfer_candidates( - &self, - ) -> crate::Result> { - self.ensure_running()?; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::IdleTransferCandidates { reply }) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::RuntimeClosed)? - } - - /// Lists active, non-draining catalog entries for bounded owner maintenance. - pub async fn active_catalog_entries( - &self, - ) -> crate::Result> { - self.ensure_running()?; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::ActiveCatalogEntries { reply }) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::RuntimeClosed)? - } - - /// Lists tenant-scoped targets of active, non-draining owners for maintenance. - /// - /// Targets come from verified activation proofs, including owners whose - /// callers cancelled after activation. This advisory snapshot cannot authorize release. - pub async fn active_cell_targets(&self) -> crate::Result> { - self.ensure_running()?; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::ActiveCellTargets { reply }) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::RuntimeClosed)? - } - - /// Counts live and transitioning Cells until their release has completed. - pub async fn unreleased_cell_count(&self) -> crate::Result { - self.ensure_running()?; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::UnreleasedCellCount { reply }) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::RuntimeClosed)? - } - - /// Releases one exact local generation only after a fresh settled-work - /// preflight, actor gate, worker close, and authoritative release complete. - pub async fn release_idle_cell( - &self, - cell: CellId, - source: SessionId, - generation: u64, - ) -> crate::Result<()> { - self.ensure_running()?; - if source != self.inner.session || generation == 0 { - return Err(Error::Fenced); - } - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::ReleaseIdleCell { - cell, - generation, - reply, - }) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::RuntimeClosed)? - } - - /// Feeds one measured node sample into the actor-owned hysteretic pressure - /// controller. Sustained shedding starts the same bounded idle-eviction - /// path exposed by [`Self::evict_idle`]. - /// - /// Samples are ordered by `at_ms`, and the actor samples its own ledger on - /// the wall clock, so a caller must not observe an older node time than the - /// node itself already has. - pub async fn observe_pressure(&self, sample: PressureSample) -> crate::Result { - self.ensure_running()?; - let (reply, response) = oneshot::channel(); - self.inner - .sender - .send(Message::ObservePressure { sample, reply }) - .await - .map_err(|_| Error::RuntimeClosed)?; - response.await.map_err(|_| Error::RuntimeClosed)? - } - - /// Reports whether node-wide admission has entered its terminal drain. - #[must_use] - pub fn is_shutting_down(&self) -> bool { - self.inner.shutting_down.load(Ordering::Acquire) - } - - /// Samples node-wide admission usage without waiting for actor work. - #[must_use] - pub fn stats(&self) -> CellRuntimeStats { - let ( - retained, - retained_capacity, - resident, - resident_capacity, - file_descriptors, - file_descriptor_capacity, - worker_jobs, - worker_job_capacity, - primitive_jobs, - primitive_job_capacity, - hydration_jobs, - hydration_job_capacity, - io_slots, - io_slot_capacity, - blocking_jobs, - blocking_job_capacity, - recovery_jobs, - recovery_job_capacity, - dirty_jobs, - dirty_job_capacity, - scratch_units, - scratch_unit_capacity, - disk_bytes, - disk_capacity_bytes, - ) = self - .inner - .resources - .snapshot() - .map(|snapshot| { - ( - snapshot.used.retained_bytes(), - snapshot.limit.retained_bytes(), - snapshot.used.resident_bytes(), - snapshot.limit.resident_bytes(), - snapshot.used.file_descriptors(), - snapshot.limit.file_descriptors(), - snapshot.used.worker_jobs(), - snapshot.limit.worker_jobs(), - snapshot.used.primitive_jobs(), - snapshot.limit.primitive_jobs(), - snapshot.used.hydration_jobs(), - snapshot.limit.hydration_jobs(), - snapshot.used.io_slots(), - snapshot.limit.io_slots(), - snapshot.used.blocking_jobs(), - snapshot.limit.blocking_jobs(), - snapshot.used.recovery_jobs(), - snapshot.limit.recovery_jobs(), - snapshot.used.dirty_jobs(), - snapshot.limit.dirty_jobs(), - snapshot.used.scratch_units(), - snapshot.limit.scratch_units(), - snapshot.used.disk_bytes(), - snapshot.limit.disk_bytes(), - ) - }) - .unwrap_or(( - 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, - )); - CellRuntimeStats { - active_cells: self.inner.pool.active_cells(), - active_cell_capacity: self.inner.pool.active_cell_capacity(), - resident_bytes: resident, - resident_capacity_bytes: resident_capacity, - file_descriptors, - file_descriptor_capacity, - retained_bytes: retained, - retained_capacity_bytes: retained_capacity, - worker_jobs, - worker_job_capacity, - primitive_jobs, - primitive_job_capacity, - hydration_jobs, - hydration_job_capacity, - io_slots, - io_slot_capacity, - blocking_jobs, - blocking_job_capacity, - recovery_jobs, - recovery_job_capacity, - dirty_jobs, - dirty_job_capacity, - scratch_units, - scratch_unit_capacity, - local_disk_reserved_bytes: disk_bytes, - local_disk_capacity_bytes: disk_capacity_bytes, - unpublished_node_log_bytes: self - .inner - .unpublished_node_log_bytes - .load(Ordering::Acquire), - } - } - - /// Reserves node-wide bytes for native work retained outside a Cell mailbox. - /// - /// Returns a capacity error without waiting when the shared budget is full, - /// and returns `RuntimeClosed` once terminal drain begins. - pub fn try_reserve_node_bytes(&self, bytes: usize) -> crate::Result { - self.ensure_running()?; - if bytes == 0 { - return Err(Error::Capacity("node retained bytes")); - } - let reservation = self - .inner - .resources - .try_reserve(ResourceCost::zero().with_retained_bytes(bytes)) - .map_err(|error| match error { - Error::Capacity(_) => Error::Capacity("node retained bytes"), - error => error, - })?; - Ok(NodeByteReservation { - _reservation: reservation, - }) - } - - pub(crate) fn reserve_read_view(&self) -> crate::Result { - self.ensure_running()?; - self.inner.resources.try_reserve( - ResourceCost::zero() - .with_resident_bytes(crate::fleet::resource::READ_REPLICA_NATIVE_BYTES) - .with_file_descriptors(crate::fleet::resource::READ_REPLICA_FILE_DESCRIPTORS), - ) - } - - pub(crate) fn replica_for_read(&self, replica: crab_ltx::CellReplica) -> crab_ltx::CellReplica { - replica.with_host(self.inner.replica_host.clone()) - } - - pub(crate) async fn reserve_sql_job( - &self, - ) -> crate::Result { - self.ensure_running()?; - self.inner.pool.reserve_snapshot_job().await - } - - /// Tries to reserve one worker-job slot from the same ledger as SQL work. - /// - /// A full ledger returns `Ok(None)` so schedulers can leave durable work - /// unclaimed and retry on the next scan. Other failures preserve their - /// original runtime error. - pub fn try_reserve_worker_job(&self) -> crate::Result> { - self.ensure_running()?; - match Arc::clone(&self.inner.primitive_jobs).try_acquire_owned() { - Ok(permit) => self.worker_job_reservation(permit).map(Some), - Err(tokio::sync::TryAcquireError::NoPermits) => Ok(None), - Err(tokio::sync::TryAcquireError::Closed) => Err(Error::RuntimeClosed), - } - } - - /// Waits for one primitive-job slot until the deadline or runtime shutdown. - /// - /// Callers must bound memory retained by waiters. Cancellation before - /// admission reserves nothing; dispatched work must retain the returned - /// reservation until it completes. - pub async fn reserve_worker_job( - &self, - deadline: std::time::Instant, - ) -> crate::Result { - self.ensure_running()?; - if std::time::Instant::now() >= deadline { - return Err(Error::Deadline); - } - let permit = tokio::time::timeout_at( - tokio::time::Instant::from_std(deadline), - Arc::clone(&self.inner.primitive_jobs).acquire_owned(), - ) - .await - .map_err(|_| Error::Deadline)? - .map_err(|_| Error::RuntimeClosed)?; - // Tokio polls a ready permit before its timer. Recheck after waking so - // a released slot cannot revive work whose admission budget expired. - if std::time::Instant::now() >= deadline { - return Err(Error::Deadline); - } - self.worker_job_reservation(permit) - } - - fn worker_job_reservation( - &self, - permit: tokio::sync::OwnedSemaphorePermit, - ) -> crate::Result { - self.ensure_running()?; - Ok(NodeJobReservation { - _reservation: self - .inner - .resources - .try_reserve(ResourceCost::zero().with_primitive_jobs(1))?, - _permit: permit, - }) - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/state.rs b/crates/crab-cell-runtime/src/cell/actor/state.rs deleted file mode 100644 index 2653b71c8..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/state.rs +++ /dev/null @@ -1,458 +0,0 @@ -//! Actor state: queues, admissions, activations, and task results. - -use super::*; - -pub(super) struct RuntimeInner { - pub(super) sender: mpsc::Sender, - pub(super) publications: broadcast::Sender, - pub(super) resources: ResourceLedger, - pub(super) primitive_jobs: Arc, - pub(super) shutting_down: AtomicBool, - pub(super) accepting_cells: AtomicBool, - pub(super) session: SessionId, - pub(super) pool: SqlWorkerPool, - pub(super) replica_host: crab_ltx::Host, - pub(super) application_limits: OnceLock>, - pub(super) node_lease: Arc, - pub(super) node_durability: NodeDurabilitySlot, - pub(super) telemetry: crate::fleet::telemetry::CellTelemetryHandle, - pub(super) unpublished_node_log_bytes: Arc, -} - -pub(super) enum RuntimeNodeLease { - ObjectOnly, - Required(OnceLock), -} - -impl RuntimeNodeLease { - pub(super) fn check(&self) -> crate::Result<()> { - match self { - Self::ObjectOnly => Ok(()), - Self::Required(guard) => guard.get().ok_or(Error::Fenced)?.check(), - } - } - - pub(super) fn guard(&self) -> crate::Result> { - match self { - Self::ObjectOnly => Ok(None), - Self::Required(guard) => guard.get().cloned().map(Some).ok_or(Error::Fenced), - } - } - - pub(super) fn install(&self, guard: NodeLeaseGuard) -> crate::Result<()> { - match self { - Self::ObjectOnly => Err(Error::Control( - "object-only Cell runtime does not accept a node lease", - )), - Self::Required(slot) => slot - .set(guard) - .map_err(|_| Error::Control("Cell runtime node lease was initialized twice")), - } - } -} - -pub(super) enum Activation { - Restored(Box), - Bootstrap(Box), -} - -pub(super) struct RestoredActivation { - pub(super) database: crate::cell::worker::RestoredDatabase, - pub(super) destination: PathBuf, - pub(super) incarnation: crate::identity::IncarnationId, - pub(super) schema: u32, - pub(super) root: crab_ltx::RootRef, - pub(super) reservation: CellReservation, -} - -pub(super) struct BootstrapActivation { - pub(super) replica: crab_ltx::CellReplica, - pub(super) destination: PathBuf, - pub(super) incarnation: crate::identity::IncarnationId, - pub(super) schema: u32, - pub(super) initialize: Initializer, - pub(super) reservation: CellReservation, -} - -pub(super) type IdleTransferCandidates = Vec<(CellId, u64, i64, CatalogRole)>; - -pub(super) enum Message { - Activate { - cell: CellId, - role: CatalogRole, - catalog: CatalogProof, - activation: Activation, - publisher: Box, - reply: oneshot::Sender>>, - }, - Execute(Box), - Query(Box), - Resolve(Box), - Migrate(Box), - Lookup { - cell: CellId, - require_resident: bool, - reply: oneshot::Sender>, - }, - /// Lists resident Cells whose published due time has passed. - /// - /// The scheduler uses this to tick a Cell it already owns without reading - /// its catalog entry or control record first. - DueResident { - now_ms: i64, - limit: usize, - reply: oneshot::Sender>, - }, - Drain { - cell: CellId, - admission: Arc, - reply: oneshot::Sender>, - }, - EvictIdle { - limit: usize, - reply: oneshot::Sender>, - }, - IdleTransferCandidates { - reply: oneshot::Sender>, - }, - ActiveCatalogEntries { - reply: oneshot::Sender>>, - }, - ActiveCellTargets { - reply: oneshot::Sender>>, - }, - UnreleasedCellCount { - reply: oneshot::Sender>, - }, - ReleaseIdleCell { - cell: CellId, - generation: u64, - reply: oneshot::Sender>, - }, - ObservePressure { - sample: PressureSample, - reply: oneshot::Sender>, - }, - Shutdown { - reply: oneshot::Sender>, - }, -} - -pub(super) struct QueuedCommand { - pub(super) trace: tracing::Span, - pub(super) telemetry: crate::fleet::telemetry::CellTelemetryHandle, - pub(super) queued_at: std::time::Instant, - pub(super) response_proof: Option<(crate::node::log::DurabilitySource, std::time::Duration)>, - pub(super) cell: CellId, - pub(super) admission: Arc, - pub(super) operation: QueuedOperation, - pub(super) now_ms: i64, - pub(super) max_result_bytes: usize, - pub(super) handler: Option, - pub(super) reply: Option>>, - pub(super) _work: WorkAdmission, -} - -#[derive(Clone, Copy)] -pub(super) enum QueuedOperation { - Mutation { - identity: MutationIdentity, - operation_digest: Digest, - }, - Effect { - delivery: InboxDelivery, - }, -} - -impl QueuedOperation { - pub(super) fn unknown(self, source: Error) -> Error { - match self { - Self::Mutation { - identity, - operation_digest, - } => Error::OutcomeUnknown { - request_id: identity.request_id, - operation_digest, - source: Box::new(source), - }, - Self::Effect { delivery } => Error::EffectOutcomeUnknown { - effect_id: delivery.effect_id, - operation_digest: delivery.operation_digest, - source: Box::new(source), - }, - } - } -} - -pub(super) struct QueuedQuery { - pub(super) cell: CellId, - pub(super) admission: Arc, - pub(super) max_result_bytes: usize, - pub(super) handler: Option, - pub(super) reply: Option>>>, - pub(super) _work: WorkAdmission, -} - -pub(super) struct QueuedResolve { - pub(super) cell: CellId, - pub(super) admission: Arc, - pub(super) operation: ResolveOperation, - pub(super) now_ms: i64, - pub(super) max_result_bytes: usize, - pub(super) reply: Option>>, - pub(super) _work: WorkAdmission, -} - -pub(super) struct QueuedMigration { - pub(super) cell: CellId, - pub(super) admission: Arc, - pub(super) successor_admission: Arc, - pub(super) plan: MigrationPlan, - pub(super) now_ms: i64, - pub(super) reply: Option>>, - pub(super) _work: WorkAdmission, -} - -pub(super) struct MigratedAdmission { - pub(super) admission: Arc, - pub(super) outcome: MigrationOutcome, -} - -#[derive(Clone, Copy)] -pub(super) enum ResolveOperation { - Mutation { - identity: MutationIdentity, - operation_digest: Digest, - }, - Effect { - delivery: InboxDelivery, - }, -} - -pub(super) enum QueuedWork { - Command(Box), - Query(Box), - Resolve(Box), - Migration(Box), -} - -pub(super) struct ActiveCell { - pub(super) generation: u64, - pub(super) admission: Arc, - pub(super) incarnation: crate::identity::IncarnationId, - pub(super) code: Digest, - pub(super) schema: u32, - pub(super) role: CatalogRole, - pub(super) catalog: CatalogProof, - pub(super) interrupt: Arc, - pub(super) publisher: Option, - pub(super) durability_submitter: CellDurabilitySubmitter, - pub(super) publications: VecDeque, - pub(super) publication_bytes: u64, - pub(super) unpublished_node_logs: usize, - pub(super) queue: VecDeque, - pub(super) coordination: CoordinationState, - pub(super) persisted_work: crate::primitives::maintenance::PersistedWorkInventory, - pub(super) inventory_refreshing: bool, - pub(super) drain: Option>>, - // Transfer closes the old capability and installs a fresh one; failed fresh - // inventory must leave the current owner serving through that capability. - pub(super) transfer: Option, - pub(super) last_used_ms: i64, - pub(super) last_work_at: std::time::Instant, - pub(super) compaction_retry_at: std::time::Instant, - pub(super) hydration_retry_at: std::time::Instant, - // The published head's due time and commit sequence, mirrored from the - // authoritative control so a resident Cell can be ticked without a - // metadata read. Both advance through the same publication that writes - // control, so a stale reader only produces a `Stale` Tick. - pub(super) next_due_ms: Option, - pub(super) published_sequence: u64, -} - -pub(super) struct TransferPreflight { - pub(super) reply: oneshot::Sender>, -} - -pub(super) struct QueuedPublication { - pub(super) pending: PendingCommit, - pub(super) durability: Option, - pub(super) retained_reservation: ResourceReservation, - pub(super) submitted_at: std::time::Instant, - pub(super) proof: oneshot::Sender>, -} - -impl ActiveCell { - pub(super) fn draining(&self) -> bool { - self.drain.is_some() - || self.transfer.is_some() - || self.coordination.is_draining() - || self.coordination.is_transfer_preparing() - } - - pub(super) fn busy(&self) -> bool { - self.coordination.is_busy() - } - - pub(super) fn renewing(&self) -> bool { - self.coordination.is_renewing() - } - - pub(super) fn begin_task(&mut self, effect: CoordinationEffect) -> u64 { - self.coordination.begin_effect(effect) - } - - pub(super) fn finish_task(&mut self, effect_id: u64, effect: CoordinationEffect) -> bool { - matches!( - self.coordination - .step(CoordinationInput::CompleteEffect { effect_id, effect }), - CoordinationDecision::EffectCompleted - ) - } -} - -pub(super) struct ShutdownState { - pub(super) reply: oneshot::Sender>, - pub(super) draining: bool, - pub(super) error: Option, -} - -pub(super) struct LocalCell { - pub(super) admission: Arc, - pub(super) incarnation: crate::identity::IncarnationId, - pub(super) code: Digest, - pub(super) schema: u32, -} - -/// One resident Cell whose published due time has passed. -pub(super) struct DueResidentCell { - pub(super) cell: CellId, - pub(super) catalog: CatalogProof, - pub(super) incarnation: crate::identity::IncarnationId, - pub(super) code: Digest, - pub(super) schema: u32, - pub(super) admission: Arc, - /// Commit sequence the last authoritative publication named. - pub(super) expected_commit_sequence: u64, - pub(super) next_due_ms: i64, -} - -pub(super) enum TaskResult { - Activated { - cell: CellId, - generation: u64, - role: CatalogRole, - catalog: CatalogProof, - publisher: Box, - admission: Arc, - reply: oneshot::Sender>>, - result: crate::Result<( - Arc, - Option, - )>, - persisted_work: crate::Result, - }, - Hydrated { - cell: CellId, - generation: u64, - effect_id: u64, - result: crate::Result, - }, - InventoryRefreshed { - cell: CellId, - generation: u64, - effect_id: u64, - result: crate::Result, - }, - TransferPreflight { - cell: CellId, - generation: u64, - effect_id: u64, - result: crate::Result, - }, - Executed { - cell: CellId, - generation: u64, - effect_id: u64, - command: Box, - result: crate::Result, - fenced: bool, - }, - Proven { - cell: CellId, - generation: u64, - effect_id: u64, - command: Box, - result: crate::Result, - fenced: bool, - }, - Published { - cell: CellId, - generation: u64, - effect_id: u64, - publisher: Box, - retained_bytes: u64, - node_logged: bool, - next_due_ms: Option, - commit_sequence: u64, - result: crate::Result<()>, - fenced: bool, - }, - Compacted { - cell: CellId, - generation: u64, - effect_id: u64, - publisher: Box, - result: crate::Result>, - }, - Queried { - cell: CellId, - generation: u64, - effect_id: u64, - query: Box, - result: crate::Result>, - fenced: bool, - }, - Resolved { - cell: CellId, - generation: u64, - effect_id: u64, - resolve: Box, - result: crate::Result, - fenced: bool, - }, - Migrated { - cell: CellId, - generation: u64, - effect_id: u64, - publisher: Box, - migration: Box, - result: crate::Result, - fenced: bool, - preserve_owner: bool, - unpublished_bytes: u64, - }, - Renewed { - cell: CellId, - generation: u64, - effect_id: u64, - publisher: Box, - result: crate::Result<()>, - }, - Deactivated { - cell: CellId, - generation: u64, - reply: Option>>, - shutdown_drain: bool, - result: crate::Result<()>, - }, -} - -pub(super) enum CommandTaskResult { - Recorded(StoredOutcome), - Pending { - pending: Box, - durability: Option, - retained_reservation: ResourceReservation, - }, -} diff --git a/crates/crab-cell-runtime/src/cell/actor/task.rs b/crates/crab-cell-runtime/src/cell/actor/task.rs deleted file mode 100644 index 7de122922..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/task.rs +++ /dev/null @@ -1,833 +0,0 @@ -//! The actor loop, message dispatch, request execution, and lifecycle -//! scheduling. -//! -//! Finished background tasks are handled in `tasks.rs`. - -use super::admission::*; -use super::*; - -pub(super) async fn run( - mut receiver: mpsc::Receiver, - pool: SqlWorkerPool, - node_lease: Arc, - unpublished_node_log_bytes: Arc, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, - publications: broadcast::Sender, -) { - let mut cells = HashMap::::new(); - let mut transitioning = HashSet::::new(); - let mut tasks = JoinSet::::new(); - let mut next_generation = 0_u64; - let mut shutdown = None::; - let mut pressure = match PressureClassifier::new(800, 600, 1_000) { - Ok(classifier) => classifier, - Err(_) => return, - }; - let mut movement = match MovementBudget::new(2, 1_000) { - Ok(budget) => budget, - Err(_) => return, - }; - let mut movement_permits = HashMap::::new(); - let mut renewal_tick = tokio::time::interval(RENEWAL_SCAN); - renewal_tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - renewal_tick.tick().await; - let mut hydration_tick = tokio::time::interval(HYDRATION_TICK); - hydration_tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - hydration_tick.tick().await; - let mut pressure_tick = tokio::time::interval(PRESSURE_SAMPLE); - pressure_tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - pressure_tick.tick().await; - loop { - if shutdown.as_ref().is_some_and(|state| state.draining) { - if tasks.is_empty() { - if !cells.is_empty() || !transitioning.is_empty() { - fail_shutdown( - &mut shutdown, - Error::Control("Cell shutdown left local state without a task"), - ); - } - finish_shutdown(&mut shutdown); - return; - } - let Some(Ok(result)) = tasks.join_next().await else { - fail_shutdown(&mut shutdown, Error::RuntimeClosed); - finish_shutdown(&mut shutdown); - return; - }; - super::tasks::handle_task( - result, - &pool, - &mut cells, - &mut transitioning, - &mut tasks, - &mut shutdown, - &node_lease, - &unpublished_node_log_bytes, - &publications, - &mut movement, - &mut movement_permits, - ); - continue; - } - if tasks.is_empty() { - tokio::select! { - message = receiver.recv() => { - let Some(message) = message else { - if shutdown.is_some() { - start_shutdown_drain( - &pool, - &mut cells, - &mut transitioning, - &mut tasks, - &mut shutdown, - &node_lease, - ); - continue; - } - break; - }; - handle_message(message, &mut receiver, &pool, &mut cells, &mut transitioning, &mut tasks, &mut shutdown, &node_lease, &telemetry, &mut pressure, &mut movement, &mut movement_permits, &mut next_generation); - } - _ = renewal_tick.tick() => { - start_due_renewals(&pool, &mut cells, &mut tasks, &node_lease); - } - _ = hydration_tick.tick() => { - start_background_hydration( - &pool, - &mut cells, - &mut tasks, - &node_lease, - ); - start_background_inventory(&pool, &mut cells, &mut tasks, &node_lease); - start_background_compaction(&pool, &mut cells, &mut tasks, &node_lease); - } - _ = pressure_tick.tick() => { - sample_node_pressure( - &pool, - &telemetry, - &mut pressure, - &mut cells, - &mut transitioning, - &mut tasks, - &mut movement, - &mut movement_permits, - ); - } - } - continue; - } - tokio::select! { - message = receiver.recv() => { - let Some(message) = message else { - if shutdown.is_some() { - start_shutdown_drain( - &pool, - &mut cells, - &mut transitioning, - &mut tasks, - &mut shutdown, - &node_lease, - ); - continue; - } - while let Some(result) = tasks.join_next().await { - let Ok(result) = result else { return; }; - super::tasks::handle_task(result, &pool, &mut cells, &mut transitioning, &mut tasks, &mut shutdown, &node_lease, &unpublished_node_log_bytes, &publications, &mut movement, &mut movement_permits); - } - break; - }; - handle_message(message, &mut receiver, &pool, &mut cells, &mut transitioning, &mut tasks, &mut shutdown, &node_lease, &telemetry, &mut pressure, &mut movement, &mut movement_permits, &mut next_generation); - } - result = tasks.join_next() => { - let Some(Ok(result)) = result else { - return; - }; - super::tasks::handle_task(result, &pool, &mut cells, &mut transitioning, &mut tasks, &mut shutdown, &node_lease, &unpublished_node_log_bytes, &publications, &mut movement, &mut movement_permits); - } - _ = renewal_tick.tick() => { - start_due_renewals(&pool, &mut cells, &mut tasks, &node_lease); - } - _ = hydration_tick.tick() => { - start_background_hydration( - &pool, - &mut cells, - &mut tasks, - &node_lease, - ); - start_background_inventory(&pool, &mut cells, &mut tasks, &node_lease); - start_background_compaction(&pool, &mut cells, &mut tasks, &node_lease); - } - _ = pressure_tick.tick() => { - sample_node_pressure( - &pool, - &telemetry, - &mut pressure, - &mut cells, - &mut transitioning, - &mut tasks, - &mut movement, - &mut movement_permits, - ); - } - } - } -} - -/// Feeds this node's own reservation ledger into the pressure classifier. -/// -/// A ledger that cannot be read, or a node without configured limits, yields no -/// sample: pressure policy must never stop the actor loop. -fn sample_node_pressure( - pool: &SqlWorkerPool, - telemetry: &crate::fleet::telemetry::CellTelemetryHandle, - pressure: &mut PressureClassifier, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - movement: &mut MovementBudget, - movement_permits: &mut HashMap, -) { - let Ok(snapshot) = pool.resource_ledger().snapshot() else { - return; - }; - let Ok(sample) = super::runtime::pressure_sample(snapshot, unix_millis()) else { - return; - }; - let _ = classify_pressure_sample( - sample, - telemetry, - pressure, - pool, - cells, - transitioning, - tasks, - movement, - movement_permits, - ); -} - -/// Classifies one sample and starts bounded shedding when it demands it. -/// -/// The external observation message and the actor's own ledger sample share -/// this path, so both keep one hysteresis and one movement budget. -#[expect( - clippy::too_many_arguments, - reason = "the shedding path owns actor state, tasks, the movement budget, and its sink separately" -)] -fn classify_pressure_sample( - sample: PressureSample, - telemetry: &crate::fleet::telemetry::CellTelemetryHandle, - pressure: &mut PressureClassifier, - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - movement: &mut MovementBudget, - movement_permits: &mut HashMap, -) -> crate::Result { - let state = pressure.observe(sample)?; - telemetry.pressure_state(state); - if matches!(state, PressureState::Shedding | PressureState::Critical) - && movement_permits.len() < 2 - { - let _ = start_bounded_evictions( - 1, - sample.at_ms, - pool, - cells, - transitioning, - tasks, - movement, - movement_permits, - ); - } - Ok(state) -} - -pub(super) fn start_shutdown_drain( - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - shutdown: &mut Option, - node_lease: &RuntimeNodeLease, -) { - let Some(state) = shutdown.as_mut() else { - return; - }; - if state.draining { - return; - } - state.draining = true; - let mut ready = Vec::new(); - for (cell, active) in cells.iter_mut() { - if let Some(transfer) = active.transfer.take() { - active.coordination.step(CoordinationInput::AbortTransfer); - let _ = transfer.reply.send(Err(Error::CellDraining)); - } - active.inventory_refreshing = false; - active.coordination.step(CoordinationInput::BeginShutdown); - active.admission.draining.store(true, Ordering::Release); - active.admission.requests.close(); - active.admission.bytes.close(); - if !active.queue.is_empty() { - start_next(active, pool, tasks, node_lease); - } - match schedule(active, node_lease.check().is_ok()) { - CoordinationDecision::ReadyToDeactivate => { - ready.push((*cell, false, active.unpublished_node_logs != 0)); - } - CoordinationDecision::ReadyToDeactivateFenced => { - ready.push((*cell, true, active.unpublished_node_logs != 0)); - } - CoordinationDecision::Fence => fence_active(active), - _ => {} - } - } - for (cell, fenced, preserve_owner) in ready { - if fenced { - start_fenced_deactivate(cell, pool, cells, transitioning, tasks, preserve_owner); - } else { - start_deactivate(cell, pool, cells, transitioning, tasks); - } - } -} - -#[expect( - clippy::too_many_arguments, - reason = "the actor adapter passes each independently owned protocol facility explicitly" -)] -pub(super) fn handle_message( - message: Message, - receiver: &mut mpsc::Receiver, - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - shutdown: &mut Option, - node_lease: &RuntimeNodeLease, - telemetry: &crate::fleet::telemetry::CellTelemetryHandle, - pressure: &mut PressureClassifier, - movement: &mut MovementBudget, - movement_permits: &mut HashMap, - next_generation: &mut u64, -) { - if !matches!(message, Message::Shutdown { .. }) && node_lease.check().is_err() { - for active in cells.values_mut() { - active.coordination.step(CoordinationInput::Fence); - fence_active(active); - } - reject_fenced_message(message); - return; - } - match message { - Message::Activate { - cell, - role, - catalog, - activation, - publisher, - reply, - } => { - if cells.contains_key(&cell) || !transitioning.insert(cell) { - let _ = reply.send(Err(Error::CellAlreadyActive)); - return; - } - *next_generation = next_generation.wrapping_add(1).max(1); - let generation = *next_generation; - let admission = new_cell_admission(); - let pool = pool.clone(); - tasks.spawn(async move { - let mut publisher = publisher; - let result = match activation { - Activation::Restored(activation) => { - activate_restored_and_publish(cell, &pool, &mut publisher, *activation) - .await - } - Activation::Bootstrap(activation) => { - bootstrap_and_publish(cell, &pool, &mut publisher, *activation).await - } - }; - let (result, persisted_work) = match result { - Ok(hydration) => { - match pool.interrupt_handle(cell).await { - Ok(interrupt) => { - let persisted_work = - pool.persisted_work_inventory(cell, role).await; - (Ok((Arc::new(interrupt), hydration)), persisted_work) - } - Err(error) => { - let result = - match cleanup_failed_activation(cell, &pool, &mut publisher) - .await - { - Ok(()) => error, - Err(cleanup) => cleanup, - }; - (Err(result), Err(Error::CellNotActive)) - } - } - } - Err(error) => { - let result = - match cleanup_failed_activation(cell, &pool, &mut publisher).await { - Ok(()) => error, - Err(cleanup) => cleanup, - }; - (Err(result), Err(Error::CellNotActive)) - } - }; - TaskResult::Activated { - cell, - generation, - role, - catalog, - publisher, - admission, - reply, - result, - persisted_work, - } - }); - } - Message::Execute(mut command) => { - let Some(active) = cells.get_mut(&command.cell) else { - send_command_reply(&mut command, Err(Error::CellNotActive)); - return; - }; - if active.transfer.is_some() { - send_command_reply(&mut command, Err(Error::CellDraining)); - return; - } - match active.coordination.step(CoordinationInput::Admit { - kind: AdmissionKind::Command, - admission_matches: Arc::ptr_eq(&active.admission, &command.admission), - }) { - CoordinationDecision::Admit => { - active.queue.push_back(QueuedWork::Command(command)); - start_next(active, pool, tasks, node_lease); - } - CoordinationDecision::Reject(reason) => { - send_command_reply(&mut command, Err(rejection_error(reason))); - } - _ => send_command_reply(&mut command, Err(Error::CellNotActive)), - } - } - Message::Query(mut query) => { - let Some(active) = cells.get_mut(&query.cell) else { - send_query_reply(&mut query, Err(Error::CellNotActive)); - return; - }; - if active.transfer.is_some() { - send_query_reply(&mut query, Err(Error::CellDraining)); - return; - } - match active.coordination.step(CoordinationInput::Admit { - kind: AdmissionKind::Query, - admission_matches: Arc::ptr_eq(&active.admission, &query.admission), - }) { - CoordinationDecision::Admit => { - active.queue.push_back(QueuedWork::Query(query)); - start_next(active, pool, tasks, node_lease); - } - CoordinationDecision::Reject(reason) => { - send_query_reply(&mut query, Err(rejection_error(reason))); - } - _ => send_query_reply(&mut query, Err(Error::CellNotActive)), - } - } - Message::Resolve(mut resolve) => { - let Some(active) = cells.get_mut(&resolve.cell) else { - send_resolve_reply(&mut resolve, Err(Error::CellNotActive)); - return; - }; - if active.transfer.is_some() { - send_resolve_reply(&mut resolve, Ok(Resolution::Unknown)); - return; - } - match active.coordination.step(CoordinationInput::Admit { - kind: AdmissionKind::Resolve, - admission_matches: Arc::ptr_eq(&active.admission, &resolve.admission), - }) { - CoordinationDecision::Admit => { - active.queue.push_back(QueuedWork::Resolve(resolve)); - start_next(active, pool, tasks, node_lease); - } - CoordinationDecision::ResolveUnknown => { - send_resolve_reply(&mut resolve, Ok(Resolution::Unknown)); - } - CoordinationDecision::Reject(reason) => { - send_resolve_reply(&mut resolve, Err(rejection_error(reason))); - } - _ => send_resolve_reply(&mut resolve, Err(Error::CellNotActive)), - } - } - Message::Migrate(mut migration) => { - let Some(active) = cells.get_mut(&migration.cell) else { - send_migration_reply(&mut migration, Err(Error::CellNotActive)); - return; - }; - let decision = active.coordination.step(CoordinationInput::Admit { - kind: AdmissionKind::Migration, - admission_matches: Arc::ptr_eq(&active.admission, &migration.admission), - }); - if let CoordinationDecision::Reject(reason) = decision { - send_migration_reply(&mut migration, Err(rejection_error(reason))); - return; - } - if !matches!(decision, CoordinationDecision::Admit) { - send_migration_reply(&mut migration, Err(Error::CellNotActive)); - return; - } - if active.code != migration.plan.from_code() - || active.schema != migration.plan.from_schema() - { - send_migration_reply( - &mut migration, - Err(Error::Registry("migration plan does not match active Cell")), - ); - return; - } - match active.coordination.step(CoordinationInput::BeginMigration) { - CoordinationDecision::Started => {} - CoordinationDecision::Reject(reason) => { - send_migration_reply(&mut migration, Err(rejection_error(reason))); - return; - } - _ => { - send_migration_reply(&mut migration, Err(Error::CellDraining)); - return; - } - } - active.admission = Arc::clone(&migration.successor_admission); - active.queue.push_back(QueuedWork::Migration(migration)); - start_next(active, pool, tasks, node_lease); - } - Message::Lookup { - cell, - require_resident, - reply, - } => { - let local = cells.get(&cell).and_then(|active| { - matches!( - active.coordination.lookup(), - CoordinationDecision::LocalHandle - ) - .then_some(active) - .filter(|active| { - active.transfer.is_none() - && (!require_resident - || active.coordination.residency() == Residency::Resident) - }) - .map(|active| LocalCell { - admission: active.admission.clone(), - incarnation: active.incarnation, - code: active.code, - schema: active.schema, - }) - }); - let _ = reply.send(local); - } - Message::DueResident { - now_ms, - limit, - reply, - } => { - // Scanning the actor's own map is what makes this path free of - // metadata reads. Due times arrive in any order, so order by - // (due, cell) and let the caller bound the batch. - let mut due = cells - .values() - .filter(|active| { - active.transfer.is_none() - && active.drain.is_none() - && active.next_due_ms.is_some_and(|due| due <= now_ms) - && matches!( - active.coordination.lookup(), - CoordinationDecision::LocalHandle - ) - }) - .filter_map(|active| { - let next_due_ms = active.next_due_ms?; - Some(DueResidentCell { - cell: active.catalog.entry().cell(), - catalog: active.catalog.clone(), - incarnation: active.incarnation, - code: active.code, - schema: active.schema, - admission: Arc::clone(&active.admission), - expected_commit_sequence: active.published_sequence, - next_due_ms, - }) - }) - .collect::>(); - due.sort_by_key(|cell| (cell.next_due_ms, *cell.cell.as_bytes())); - due.truncate(limit); - let _ = reply.send(due); - } - Message::Drain { - cell, - admission, - reply, - } => { - let Some(active) = cells.get_mut(&cell) else { - let _ = reply.send(Err(Error::CellNotActive)); - return; - }; - if !Arc::ptr_eq(&active.admission, &admission) { - let _ = reply.send(Err(if active.transfer.is_some() { - Error::CellDraining - } else { - Error::CellNotActive - })); - return; - } - if active.transfer.is_some() { - if let Some(transfer) = active.transfer.take() { - active.coordination.step(CoordinationInput::AbortTransfer); - let _ = transfer.reply.send(Err(Error::CellDraining)); - } - let decision = active.coordination.step(CoordinationInput::BeginDrain); - if let CoordinationDecision::Reject(reason) = decision { - let _ = reply.send(Err(rejection_error(reason))); - return; - } - active.drain = Some(reply); - if matches!( - schedule(active, node_lease.check().is_ok()), - CoordinationDecision::ReadyToDeactivate - ) { - start_deactivate(cell, pool, cells, transitioning, tasks); - } - return; - } - let decision = active.coordination.step(CoordinationInput::BeginDrain); - if let CoordinationDecision::Reject(reason) = decision { - let _ = reply.send(Err(rejection_error(reason))); - return; - } - active.drain = Some(reply); - match schedule(active, node_lease.check().is_ok()) { - CoordinationDecision::ReadyToDeactivate => { - start_deactivate(cell, pool, cells, transitioning, tasks); - } - CoordinationDecision::Fence => fence_active(active), - _ => {} - } - } - Message::EvictIdle { limit, reply } => { - let count = start_bounded_evictions( - limit, - unix_millis(), - pool, - cells, - transitioning, - tasks, - movement, - movement_permits, - ); - let _ = reply.send(Ok(count)); - } - Message::IdleTransferCandidates { reply } => { - if node_lease.check().is_err() { - let _ = reply.send(Err(Error::Fenced)); - return; - } - let candidates = cells - .iter() - .filter_map(|(cell, active)| { - transfer_candidate_observation(*cell, active) - .eligible() - .then_some(active) - .filter(|active| !active.draining()) - .map(|active| (*cell, active.generation, active.last_used_ms, active.role)) - }) - .collect(); - let _ = reply.send(Ok(candidates)); - } - Message::ActiveCatalogEntries { reply } => { - if node_lease.check().is_err() { - let _ = reply.send(Err(Error::Fenced)); - return; - } - let entries = cells - .values() - .filter(|active| !active.draining()) - .map(|active| active.catalog.entry().clone()) - .collect(); - let _ = reply.send(Ok(entries)); - } - Message::ActiveCellTargets { reply } => { - let result = node_lease.check().and_then(|()| { - cells - .values() - .filter(|active| !active.draining()) - .map(|active| active.catalog.target()) - .collect() - }); - let _ = reply.send(result); - } - Message::UnreleasedCellCount { reply } => { - let _ = reply.send(Ok(cells.len().saturating_add(transitioning.len()))); - } - Message::ReleaseIdleCell { - cell, - generation, - reply, - } => { - if node_lease.check().is_err() { - let _ = reply.send(Err(Error::Fenced)); - return; - } - let Some(active) = cells.get(&cell) else { - let _ = reply.send(Err(Error::CellNotActive)); - return; - }; - if active.generation != generation - || active.draining() - || active.transfer.is_some() - || active.inventory_refreshing - { - let _ = reply.send(Err(Error::CellDraining)); - return; - } - let Ok(mut permit) = movement.try_start(unix_millis()) else { - let _ = reply.send(Err(Error::Capacity("movement budget"))); - return; - }; - { - let Some(active) = cells.get_mut(&cell) else { - movement.complete(&mut permit); - let _ = reply.send(Err(Error::CellNotActive)); - return; - }; - if active.transfer.is_some() - || active.drain.is_some() - || active.inventory_refreshing - { - movement.complete(&mut permit); - let _ = reply.send(Err(Error::CellDraining)); - return; - } - let decision = - active - .coordination - .step(CoordinationInput::BeginTransferPreflight { - queue_empty: active.queue.is_empty(), - publication_idle: active.coordination.publication_count() == 0, - lease_live: node_lease.check().is_ok(), - }); - match decision { - CoordinationDecision::Started => {} - CoordinationDecision::Fence => { - fence_active(active); - movement.complete(&mut permit); - let _ = reply.send(Err(Error::Fenced)); - return; - } - CoordinationDecision::Reject(reason) => { - movement.complete(&mut permit); - let _ = reply.send(Err(rejection_error(reason))); - return; - } - CoordinationDecision::Ignored => { - movement.complete(&mut permit); - let _ = reply.send(Err(Error::CellDraining)); - return; - } - _ => { - movement.complete(&mut permit); - let _ = reply.send(Err(Error::CellDraining)); - return; - } - } - active.admission.draining.store(true, Ordering::Release); - active.admission.requests.close(); - active.admission.bytes.close(); - active.admission = new_cell_admission(); - active.transfer = Some(TransferPreflight { reply }); - }; - movement_permits.insert(cell, permit); - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); - } - Message::ObservePressure { sample, reply } => { - let result = classify_pressure_sample( - sample, - telemetry, - pressure, - pool, - cells, - transitioning, - tasks, - movement, - movement_permits, - ); - let _ = reply.send(result); - } - Message::Shutdown { reply } => { - if shutdown.is_some() { - let _ = reply.send(Err(Error::RuntimeClosed)); - return; - } - receiver.close(); - *shutdown = Some(ShutdownState { - reply, - draining: false, - error: None, - }); - } - } -} - -pub(super) fn reject_fenced_message(message: Message) { - match message { - Message::Activate { reply, .. } => { - let _ = reply.send(Err(Error::Fenced)); - } - Message::Execute(mut command) => { - send_command_reply(&mut command, Err(Error::Fenced)); - } - Message::Query(mut query) => { - send_query_reply(&mut query, Err(Error::Fenced)); - } - Message::Resolve(mut resolve) => { - send_resolve_reply(&mut resolve, Ok(Resolution::Unknown)); - } - Message::Migrate(mut migration) => { - send_migration_reply(&mut migration, Err(Error::Fenced)); - } - Message::Lookup { reply, .. } => { - let _ = reply.send(None); - } - Message::DueResident { reply, .. } => { - let _ = reply.send(Vec::new()); - } - Message::Drain { reply, .. } => { - let _ = reply.send(Err(Error::Fenced)); - } - Message::EvictIdle { reply, .. } => { - let _ = reply.send(Err(Error::Fenced)); - } - Message::IdleTransferCandidates { reply } => { - let _ = reply.send(Err(Error::Fenced)); - } - Message::ActiveCatalogEntries { reply } => { - let _ = reply.send(Err(Error::Fenced)); - } - Message::ActiveCellTargets { reply } => { - let _ = reply.send(Err(Error::Fenced)); - } - Message::UnreleasedCellCount { reply } => { - let _ = reply.send(Err(Error::Fenced)); - } - Message::ReleaseIdleCell { reply, .. } => { - let _ = reply.send(Err(Error::Fenced)); - } - Message::ObservePressure { reply, .. } => { - let _ = reply.send(Err(Error::Fenced)); - } - Message::Shutdown { reply } => { - let _ = reply.send(Err(Error::RuntimeClosed)); - } - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/tasks.rs b/crates/crab-cell-runtime/src/cell/actor/tasks.rs deleted file mode 100644 index 702062c0e..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/tasks.rs +++ /dev/null @@ -1,213 +0,0 @@ -//! Task-completion handling for one Cell actor. -//! -//! Everything here reacts to a finished background task: activation, -//! publication, migration, transfer, hydration, eviction, and drain. The loop -//! itself and the request paths stay in `task.rs`; each completion kind has -//! its own handler in `tasks/`, and this root keeps the dispatch. - -use super::admission::{ - fail_shutdown, fence_active, fence_admission, finish_migration, finish_work, rejection_error, - send_command_reply, send_command_task_reply, send_migration_reply, send_query_reply, - send_resolve_reply, subtract_unpublished_bytes, -}; -use super::*; - -mod activation; -mod movement; -mod publication; -mod residency; -mod work; - -/// Actor-loop facilities a finished task borrows while it is handled. -struct TaskContext<'a> { - pool: &'a SqlWorkerPool, - cells: &'a mut HashMap, - transitioning: &'a mut HashSet, - tasks: &'a mut JoinSet, - shutdown: &'a mut Option, - node_lease: &'a RuntimeNodeLease, - unpublished_node_log_bytes: &'a AtomicU64, - publications: &'a broadcast::Sender, - movement: &'a mut MovementBudget, - movement_permits: &'a mut HashMap, -} - -#[expect( - clippy::too_many_arguments, - reason = "the actor adapter passes each independently owned protocol facility explicitly" -)] -pub(super) fn handle_task( - result: TaskResult, - pool: &SqlWorkerPool, - cells: &mut HashMap, - transitioning: &mut HashSet, - tasks: &mut JoinSet, - shutdown: &mut Option, - node_lease: &RuntimeNodeLease, - unpublished_node_log_bytes: &AtomicU64, - publications: &broadcast::Sender, - movement: &mut MovementBudget, - movement_permits: &mut HashMap, -) { - let context = TaskContext { - pool, - cells, - transitioning, - tasks, - shutdown, - node_lease, - unpublished_node_log_bytes, - publications, - movement, - movement_permits, - }; - match result { - TaskResult::Activated { - cell, - generation, - role, - catalog, - publisher, - admission, - reply, - result, - persisted_work, - } => activation::handle_activated( - context, - cell, - generation, - role, - catalog, - publisher, - admission, - reply, - result, - persisted_work, - ), - TaskResult::Hydrated { - cell, - generation, - effect_id, - result, - } => activation::handle_hydrated(context, cell, generation, effect_id, result), - TaskResult::Executed { - cell, - generation, - effect_id, - command, - result, - fenced, - } => work::handle_executed( - context, cell, generation, effect_id, command, result, fenced, - ), - TaskResult::Queried { - cell, - generation, - effect_id, - query, - result, - fenced, - } => work::handle_queried(context, cell, generation, effect_id, query, result, fenced), - TaskResult::Resolved { - cell, - generation, - effect_id, - resolve, - result, - fenced, - } => work::handle_resolved( - context, cell, generation, effect_id, resolve, result, fenced, - ), - TaskResult::Proven { - cell, - generation, - effect_id, - command, - result, - fenced, - } => publication::handle_proven( - context, cell, generation, effect_id, command, result, fenced, - ), - TaskResult::Published { - cell, - generation, - effect_id, - publisher, - retained_bytes, - node_logged, - next_due_ms, - commit_sequence, - result, - fenced, - } => publication::handle_published( - context, - cell, - generation, - effect_id, - publisher, - retained_bytes, - node_logged, - next_due_ms, - commit_sequence, - result, - fenced, - ), - TaskResult::Compacted { - cell, - generation, - effect_id, - publisher, - result, - } => publication::handle_compacted(context, cell, generation, effect_id, publisher, result), - TaskResult::TransferPreflight { - cell, - generation, - effect_id, - result, - } => movement::handle_transfer_preflight(context, cell, generation, effect_id, result), - TaskResult::Migrated { - cell, - generation, - effect_id, - publisher, - migration, - result, - fenced, - preserve_owner, - unpublished_bytes, - } => movement::handle_migrated( - context, - cell, - generation, - effect_id, - publisher, - migration, - result, - fenced, - preserve_owner, - unpublished_bytes, - ), - TaskResult::InventoryRefreshed { - cell, - generation, - effect_id, - result, - } => residency::handle_inventory_refreshed(context, cell, generation, effect_id, result), - TaskResult::Renewed { - cell, - generation, - effect_id, - publisher, - result, - } => residency::handle_renewed(context, cell, generation, effect_id, publisher, result), - TaskResult::Deactivated { - cell, - generation, - reply, - shutdown_drain, - result, - } => { - residency::handle_deactivated(context, cell, generation, reply, shutdown_drain, result) - } - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/tasks/activation.rs b/crates/crab-cell-runtime/src/cell/actor/tasks/activation.rs deleted file mode 100644 index f3698979c..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/tasks/activation.rs +++ /dev/null @@ -1,219 +0,0 @@ -//! Activation and hydration completions for one Cell actor. - -use super::*; - -/// Applies an activation result: admit the Cell, or fence and clean up. -#[expect( - clippy::too_many_arguments, - reason = "the actor loop hands each protocol facility and finished-task field to the handler explicitly" -)] -pub(super) fn handle_activated( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - role: CatalogRole, - catalog: CatalogProof, - publisher: Box, - admission: Arc, - reply: oneshot::Sender>>, - result: crate::Result<( - Arc, - Option, - )>, - persisted_work: crate::Result, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - shutdown, - node_lease, - publications, - .. - } = context; - match result { - Ok((interrupt, hydration)) => { - if node_lease.check().is_err() { - fence_admission(&admission); - let _ = reply.send(Err(Error::Fenced)); - start_orphan_deactivate( - cell, - pool, - *publisher, - transitioning, - tasks, - false, - generation, - ); - return; - } - if shutdown.as_ref().is_some_and(|state| state.draining) { - admission.draining.store(true, Ordering::Release); - admission.requests.close(); - admission.bytes.close(); - let _ = reply.send(Err(Error::RuntimeClosed)); - start_orphan_deactivate( - cell, - pool, - *publisher, - transitioning, - tasks, - true, - generation, - ); - return; - } - if reply.send(Ok(admission.clone())).is_err() { - start_orphan_deactivate( - cell, - pool, - *publisher, - transitioning, - tasks, - false, - generation, - ); - return; - } - transitioning.remove(&cell); - let control = publisher.control().value(); - let incarnation = control.incarnation; - let code = control.code; - let schema = control.schema; - let next_due_ms = control.next_due_ms; - let published_sequence = control.root.as_ref().map_or(0, |root| root.commit_sequence); - let durability_submitter = publisher.durability_submitter(); - let residency = hydration.map_or(Residency::Resident, |progress| { - if progress.complete() { - Residency::Resident - } else { - Residency::Sparse - } - }); - let persisted_work = match persisted_work { - Ok(inventory) => inventory, - Err(_) => { - // An inventory read is a safety precondition for - // eviction. Unknown accounting must remain ineligible. - crate::primitives::maintenance::PersistedWorkInventory::unknown() - } - }; - // A restored or bootstrapped owner can already have a reader policy. - // Its hint is advisory; receivers still reload the new authority. - let _ = publications.send(catalog.entry().clone()); - cells.insert( - cell, - ActiveCell { - generation, - admission, - incarnation, - code, - schema, - role, - catalog, - interrupt, - publisher: Some(*publisher), - durability_submitter, - publications: VecDeque::new(), - publication_bytes: 0, - unpublished_node_logs: 0, - queue: VecDeque::new(), - coordination: CoordinationState::serving_with_residency(true, residency), - persisted_work, - inventory_refreshing: false, - drain: None, - transfer: None, - last_used_ms: unix_millis(), - last_work_at: std::time::Instant::now(), - compaction_retry_at: std::time::Instant::now(), - hydration_retry_at: std::time::Instant::now(), - next_due_ms, - published_sequence, - }, - ); - } - Err(error) => { - transitioning.remove(&cell); - let error = if node_lease.check().is_err() { - Error::Fenced - } else { - error - }; - let _ = reply.send(Err(error)); - } - } -} - -/// Applies a hydration result and continues the Cell's schedule. -pub(super) fn handle_hydrated( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - result: crate::Result, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Hydration) - { - return; - } - active.finish_task(effect_id, CoordinationEffect::Hydration); - match result { - Ok(HydrationStep::Progress(Some(progress))) => { - active - .coordination - .step(CoordinationInput::FinishHydration { - complete: progress.complete(), - stale: false, - }); - } - Ok(HydrationStep::Progress(None)) => { - active - .coordination - .step(CoordinationInput::FinishHydration { - complete: true, - stale: false, - }); - } - Ok(HydrationStep::Deferred(after)) => { - active - .coordination - .step(CoordinationInput::FinishHydration { - complete: false, - stale: false, - }); - // Respect provider backoff and avoid retrying resource pressure on - // every tick. An unrepresentable delay cannot be retried safely. - match std::time::Instant::now().checked_add(after.max(HYDRATION_RETRY)) { - Some(at) => active.hydration_retry_at = at, - None => fence_active(active), - } - } - Err(_) => { - let decision = active - .coordination - .step(CoordinationInput::FinishHydration { - complete: false, - stale: true, - }); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - } - } - } - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} diff --git a/crates/crab-cell-runtime/src/cell/actor/tasks/movement.rs b/crates/crab-cell-runtime/src/cell/actor/tasks/movement.rs deleted file mode 100644 index 9f02d24e8..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/tasks/movement.rs +++ /dev/null @@ -1,185 +0,0 @@ -//! Transfer preflight and migration completions. - -use super::*; - -/// Applies a transfer preflight result: deactivate, continue, or abort. -pub(super) fn handle_transfer_preflight( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - result: crate::Result, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - movement, - movement_permits, - .. - } = context; - let preflight = { - let Some(active) = cells.get_mut(&cell) else { - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Inventory) - { - return; - } - active.finish_task(effect_id, CoordinationEffect::Inventory); - active.inventory_refreshing = false; - match active.transfer.take() { - Some(transfer) => { - let mut ready_to_deactivate = false; - let mut fenced = false; - let transfer_result = if node_lease.check().is_err() { - fenced = true; - active.coordination.step(CoordinationInput::Fence); - fence_active(active); - Err(Error::Fenced) - } else { - match result { - Ok(inventory) => { - if inventory.is_settled() - && active.queue.is_empty() - && active.coordination.can_deactivate() - && transfer_observation(cell, active, inventory).eligible() - { - match active.coordination.step(CoordinationInput::ConfirmTransfer) { - CoordinationDecision::ReadyToDeactivate => { - ready_to_deactivate = true; - Ok(()) - } - CoordinationDecision::Started => Ok(()), - CoordinationDecision::Reject(RejectReason::Fenced) => { - fenced = true; - fence_active(active); - Err(Error::Fenced) - } - CoordinationDecision::Reject(reason) => { - Err(rejection_error(reason)) - } - _ => Err(Error::CellDraining), - } - } else { - Err(Error::CellDraining) - } - } - Err(error) => Err(error), - } - }; - if transfer_result.is_err() && !fenced { - active.coordination.step(CoordinationInput::AbortTransfer); - } - Some((transfer, ready_to_deactivate, fenced, transfer_result)) - } - None => None, - } - }; - let Some((transfer, ready_to_deactivate, _fenced, transfer_result)) = preflight else { - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); - return; - }; - if transfer_result.is_ok() { - if ready_to_deactivate { - if let Some(active) = cells.get_mut(&cell) { - active.drain = Some(transfer.reply); - } - start_deactivate(cell, pool, cells, transitioning, tasks); - } else if let Some(active) = cells.get_mut(&cell) { - active.drain = Some(transfer.reply); - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); - } else { - if let Some(mut permit) = movement_permits.remove(&cell) { - movement.complete(&mut permit); - } - } - } else { - let _ = transfer.reply.send(transfer_result); - if let Some(mut permit) = movement_permits.remove(&cell) { - movement.complete(&mut permit); - } - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); - } -} - -/// Applies a migration result and reconciles the source Cell. -#[expect( - clippy::too_many_arguments, - reason = "the actor loop hands each protocol facility and finished-task field to the handler explicitly" -)] -pub(super) fn handle_migrated( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - publisher: Box, - mut migration: Box, - mut result: crate::Result, - mut fenced: bool, - preserve_owner: bool, - unpublished_bytes: u64, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - unpublished_node_log_bytes, - publications, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - let result = match result { - Ok(_) => Err(Error::Fenced), - Err(error) => Err(error), - }; - send_migration_reply(&mut migration, result); - return; - }; - if active.generation != generation - || !active.coordination.effect_matches( - effect_id, - CoordinationEffect::Work(AdmissionKind::Migration), - ) - { - let result = match result { - Ok(_) => Err(Error::Fenced), - Err(error) => Err(error), - }; - send_migration_reply(&mut migration, result); - return; - } - active.finish_task( - effect_id, - CoordinationEffect::Work(AdmissionKind::Migration), - ); - if node_lease.check().is_err() { - result = Err(Error::Fenced); - fenced = true; - } - active.publisher = Some(*publisher); - if preserve_owner { - active.unpublished_node_logs = active.unpublished_node_logs.saturating_add(1); - unpublished_node_log_bytes.fetch_add(unpublished_bytes, Ordering::AcqRel); - } - let completion = finish_migration(active, fenced); - match result { - Ok(outcome) if !matches!(completion, CoordinationDecision::Fence) => { - active.code = outcome.code; - active.schema = outcome.schema; - let _ = publications.send(active.catalog.entry().clone()); - let admission = Arc::clone(&migration.successor_admission); - send_migration_reply(&mut migration, Ok(MigratedAdmission { admission, outcome })); - } - Ok(_) => send_migration_reply(&mut migration, Err(Error::Fenced)), - Err(error) => send_migration_reply(&mut migration, Err(error)), - } - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} diff --git a/crates/crab-cell-runtime/src/cell/actor/tasks/publication.rs b/crates/crab-cell-runtime/src/cell/actor/tasks/publication.rs deleted file mode 100644 index 09eb26666..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/tasks/publication.rs +++ /dev/null @@ -1,166 +0,0 @@ -//! Publication proof, publish, and compaction completions. - -use super::*; - -/// Applies a publication proof result and answers its waiter. -pub(super) fn handle_proven( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - mut command: Box, - mut result: crate::Result, - mut fenced: bool, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - // The publication task may fence and remove the actor first; proof owns the - // caller's final result and must not be rewritten as CellNotActive. - send_command_reply(&mut command, result); - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Proof) - { - send_command_reply(&mut command, result); - return; - } - active.finish_task(effect_id, CoordinationEffect::Proof); - if node_lease.check().is_err() { - result = Err(command.operation.unknown(Error::Fenced)); - fenced = true; - } - finish_work(active, fenced); - send_command_reply(&mut command, result); - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} - -/// Applies a publish result, its byte accounting, and its failure cleanup. -#[expect( - clippy::too_many_arguments, - reason = "the actor loop hands each protocol facility and finished-task field to the handler explicitly" -)] -pub(super) fn handle_published( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - publisher: Box, - retained_bytes: u64, - node_logged: bool, - next_due_ms: Option, - commit_sequence: u64, - mut result: crate::Result<()>, - mut fenced: bool, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - unpublished_node_log_bytes, - publications, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Publication) - { - return; - } - active.finish_task(effect_id, CoordinationEffect::Publication); - active.last_work_at = std::time::Instant::now(); - let object_published = result.is_ok(); - if object_published { - // Control now names this commit, so the local mirror can answer a due - // scan without reading the record back. - active.next_due_ms = next_due_ms; - active.published_sequence = commit_sequence; - } - if node_lease.check().is_err() { - result = Err(Error::Fenced); - fenced = true; - } - active.publisher = Some(*publisher); - active.publication_bytes = active.publication_bytes.saturating_sub(retained_bytes); - if node_logged && object_published { - active.unpublished_node_logs = active.unpublished_node_logs.saturating_sub(1); - subtract_unpublished_bytes(unpublished_node_log_bytes, retained_bytes); - } - let decision = active - .coordination - .step(CoordinationInput::FinishPublication { - fenced, - succeeded: result.is_ok(), - }); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - } else { - // Only the completed object path can wake snapshot readers. Fleet proof - // may acknowledge earlier; hints never substitute for a published root. - if result.is_ok() { - let _ = publications.send(active.catalog.entry().clone()); - } - start_publication(cell, active, pool, tasks); - } - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} - -/// Applies a compaction result and returns the publisher to the Cell. -pub(super) fn handle_compacted( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - publisher: Box, - result: crate::Result>, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Compaction) - { - return; - } - active.finish_task(effect_id, CoordinationEffect::Compaction); - if let Err(error) = &result { - tracing::warn!(cell = ?cell, error = ?error, "Cell compaction fenced its owner"); - } - let fenced = result.is_err() || node_lease.check().is_err(); - active.publisher = Some(*publisher); - if matches!(result, Ok(None)) { - active.compaction_retry_at = std::time::Instant::now() + COMPACTION_RETRY; - } - let decision = active - .coordination - .step(CoordinationInput::FinishCompaction { fenced }); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - } - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} diff --git a/crates/crab-cell-runtime/src/cell/actor/tasks/residency.rs b/crates/crab-cell-runtime/src/cell/actor/tasks/residency.rs deleted file mode 100644 index 4e2e1c4d4..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/tasks/residency.rs +++ /dev/null @@ -1,131 +0,0 @@ -//! Inventory refresh, lease renewal, and deactivation completions. - -use super::*; - -/// Applies an inventory refresh result to the Cell's persisted work. -pub(super) fn handle_inventory_refreshed( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - result: crate::Result, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Inventory) - { - return; - } - active.finish_task(effect_id, CoordinationEffect::Inventory); - active.inventory_refreshing = false; - if let Ok(inventory) = result { - active.persisted_work = inventory; - } - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} - -/// Applies a lease renewal result and continues the Cell's schedule. -pub(super) fn handle_renewed( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - publisher: Box, - mut result: crate::Result<()>, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Renewal) - { - return; - } - active.finish_task(effect_id, CoordinationEffect::Renewal); - if let Err(error) = &result { - tracing::warn!(cell = ?cell, error = ?error, "Cell renewal fenced its owner"); - } - if node_lease.check().is_err() { - result = Err(Error::Fenced); - } - active.publisher = Some(*publisher); - let decision = active.coordination.step(CoordinationInput::FinishRenewal { - fenced: result.is_err(), - }); - if matches!(decision, CoordinationDecision::Fence) { - fence_active(active); - } - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} - -/// Applies a deactivation result, releasing movement permits and shutdown waiters. -pub(super) fn handle_deactivated( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - reply: Option>>, - shutdown_drain: bool, - result: crate::Result<()>, -) { - let TaskContext { - cells, - transitioning, - shutdown, - movement, - movement_permits, - .. - } = context; - if cells - .get(&cell) - .is_some_and(|active| active.generation != generation) - { - return; - } - if let Some(mut permit) = movement_permits.remove(&cell) { - movement.complete(&mut permit); - } - transitioning.remove(&cell); - let runtime_waiting = shutdown_drain || shutdown.as_ref().is_some_and(|state| state.draining); - match (reply, runtime_waiting) { - (Some(reply), true) => { - if result.is_err() { - fail_shutdown( - shutdown, - Error::Control("one or more Cells failed to drain"), - ); - } - let _ = reply.send(result); - } - (Some(reply), false) => { - let _ = reply.send(result); - } - (None, true) => { - if let Err(error) = result { - fail_shutdown(shutdown, error); - } - } - (None, false) => {} - } -} diff --git a/crates/crab-cell-runtime/src/cell/actor/tasks/work.rs b/crates/crab-cell-runtime/src/cell/actor/tasks/work.rs deleted file mode 100644 index 372bab420..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/tasks/work.rs +++ /dev/null @@ -1,208 +0,0 @@ -//! Command, query, and resolve completions for one Cell actor. - -use super::*; - -/// Applies a command execution result and answers its waiters. -pub(super) fn handle_executed( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - mut command: Box, - mut result: crate::Result, - mut fenced: bool, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - unpublished_node_log_bytes, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - // Publication failure may fence and remove the Cell before its proof waiter - // completes; the accepted command still owns exactly one terminal outcome. - send_command_task_reply(&mut command, result); - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Work(AdmissionKind::Command)) - { - send_command_task_reply(&mut command, result); - return; - } - active.finish_task(effect_id, CoordinationEffect::Work(AdmissionKind::Command)); - if node_lease.check().is_err() { - result = Err(command.operation.unknown(Error::Fenced)); - fenced = true; - } - if fenced { - finish_work(active, true); - send_command_task_reply(&mut command, result); - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); - return; - } - match result { - Ok(CommandTaskResult::Recorded(outcome)) => { - finish_work(active, false); - send_command_reply(&mut command, Ok(outcome)); - } - Ok(CommandTaskResult::Pending { - pending, - durability, - retained_reservation, - }) => { - let retained_bytes = pending.retained_bytes(); - let publication = active - .coordination - .step(CoordinationInput::BeginPublication); - let CoordinationDecision::Started = publication else { - drop(retained_reservation); - finish_work(active, false); - let error = command.operation.unknown(match publication { - CoordinationDecision::Reject(reason) => rejection_error(reason), - _ => Error::Fenced, - }); - send_command_reply(&mut command, Err(error)); - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); - return; - }; - active.publication_bytes = match active.publication_bytes.checked_add(retained_bytes) { - Some(bytes) => bytes, - None => { - finish_work(active, true); - let error = command - .operation - .unknown(Error::Capacity("pending publication bytes")); - send_command_reply(&mut command, Err(error)); - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); - return; - } - }; - let outcome = pending.outcome().clone(); - let commit_sequence = outcome.commit_sequence(); - let (proof, object) = oneshot::channel(); - active.publications.push_back(QueuedPublication { - pending: *pending, - durability: durability.clone(), - retained_reservation, - submitted_at: std::time::Instant::now(), - proof, - }); - if durability.is_some() { - active.unpublished_node_logs += 1; - unpublished_node_log_bytes.fetch_add(retained_bytes, Ordering::AcqRel); - } - start_publication(cell, active, pool, tasks); - let pool = pool.clone(); - let generation = active.generation; - let effect_id = active.begin_task(CoordinationEffect::Proof); - tasks.spawn(async move { - prove_command( - pool, - command, - outcome, - commit_sequence, - durability, - object, - generation, - effect_id, - ) - .await - }); - } - Err(error) => { - finish_work(active, false); - send_command_reply(&mut command, Err(error)); - } - } - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} - -/// Applies a query result and answers its waiter. -pub(super) fn handle_queried( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - mut query: Box, - mut result: crate::Result>, - mut fenced: bool, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - // A query accepted before a fence keeps its result even when deactivation wins - // the actor turn before this completion is delivered. - send_query_reply(&mut query, result); - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Work(AdmissionKind::Query)) - { - send_query_reply(&mut query, result); - return; - } - active.finish_task(effect_id, CoordinationEffect::Work(AdmissionKind::Query)); - if node_lease.check().is_err() { - result = Err(Error::Fenced); - fenced = true; - } - finish_work(active, fenced); - send_query_reply(&mut query, result); - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} - -/// Applies a resolve result and answers its waiter. -pub(super) fn handle_resolved( - context: TaskContext<'_>, - cell: CellId, - generation: u64, - effect_id: u64, - mut resolve: Box, - mut result: crate::Result, - mut fenced: bool, -) { - let TaskContext { - pool, - cells, - transitioning, - tasks, - node_lease, - .. - } = context; - let Some(active) = cells.get_mut(&cell) else { - // Resolution is an accepted observation, not a new admission; preserve its - // unknown/committed result across a concurrent fenced deactivation. - send_resolve_reply(&mut resolve, result); - return; - }; - if active.generation != generation - || !active - .coordination - .effect_matches(effect_id, CoordinationEffect::Work(AdmissionKind::Resolve)) - { - send_resolve_reply(&mut resolve, result); - return; - } - active.finish_task(effect_id, CoordinationEffect::Work(AdmissionKind::Resolve)); - if node_lease.check().is_err() { - result = Ok(Resolution::Unknown); - fenced = true; - } - finish_work(active, fenced); - send_resolve_reply(&mut resolve, result); - continue_cell(cell, pool, cells, transitioning, tasks, node_lease); -} diff --git a/crates/crab-cell-runtime/src/cell/actor/tests.rs b/crates/crab-cell-runtime/src/cell/actor/tests.rs deleted file mode 100644 index 0fda5188a..000000000 --- a/crates/crab-cell-runtime/src/cell/actor/tests.rs +++ /dev/null @@ -1,40 +0,0 @@ -use super::{CellRuntimeStats, bounded_u32}; - -#[test] -fn placement_projection_saturates_large_node_counters() { - let stats = CellRuntimeStats { - active_cells: usize::MAX, - active_cell_capacity: usize::MAX, - resident_bytes: 0, - resident_capacity_bytes: 0, - file_descriptors: 0, - file_descriptor_capacity: 0, - retained_bytes: 0, - retained_capacity_bytes: 0, - worker_jobs: usize::MAX, - worker_job_capacity: usize::MAX, - primitive_jobs: usize::MAX, - primitive_job_capacity: usize::MAX, - hydration_jobs: usize::MAX, - hydration_job_capacity: usize::MAX, - io_slots: 0, - io_slot_capacity: 0, - blocking_jobs: 0, - blocking_job_capacity: 0, - recovery_jobs: 0, - recovery_job_capacity: 0, - dirty_jobs: 0, - dirty_job_capacity: 0, - scratch_units: 0, - scratch_unit_capacity: 0, - local_disk_reserved_bytes: 0, - local_disk_capacity_bytes: 0, - unpublished_node_log_bytes: 0, - }; - - assert_eq!(bounded_u32(usize::MAX), u32::MAX); - assert_eq!(stats.placement_active_cells(), u32::MAX); - assert_eq!(stats.placement_active_cell_capacity(), u32::MAX); - assert_eq!(stats.placement_running_jobs(), u32::MAX); - assert_eq!(stats.placement_job_capacity(), u32::MAX); -} diff --git a/crates/crab-cell-runtime/src/cell/application.rs b/crates/crab-cell-runtime/src/cell/application.rs deleted file mode 100644 index aaed769e2..000000000 --- a/crates/crab-cell-runtime/src/cell/application.rs +++ /dev/null @@ -1,155 +0,0 @@ -//! Application identities and the store that binds a repository application to its Cell. -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::{StorageError, Store}; -use object_store::path::Path; -use serde::{Deserialize, Serialize}; - -use crate::identity::encode_hex; -use crate::identity::{ApplicationId, TenantId}; -use crate::{Error, Result}; - -const MAX_IDENTITY_BYTES: u64 = 1_024; - -/// Stable tenant and application IDs persisted once for an authoritative root. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct ApplicationIdentity { - tenant: TenantId, - application: ApplicationId, -} - -impl ApplicationIdentity { - /// Binds one tenant to one application identity. - #[must_use] - pub const fn new(tenant: TenantId, application: ApplicationId) -> Self { - Self { - tenant, - application, - } - } - - /// Returns the tenant the identity belongs to. - #[must_use] - pub const fn tenant(&self) -> TenantId { - self.tenant - } - - /// Returns the application the identity names. - #[must_use] - pub const fn application(&self) -> ApplicationId { - self.application - } - - fn encode(self) -> Result> { - self.validate()?; - Ok(serde_json::to_vec(&RawIdentity { - application: encode_hex(self.application.as_bytes()), - tenant: encode_hex(self.tenant.as_bytes()), - version: 1, - })?) - } - - fn decode(bytes: &[u8]) -> Result { - let raw: RawIdentity = serde_json::from_slice(bytes)?; - if raw.version != 1 { - return Err(Error::Identity("unsupported application identity version")); - } - let identity = Self::new( - TenantId::from_bytes(parse_id(&raw.tenant, "tenant")?), - ApplicationId::from_bytes(parse_id(&raw.application, "application")?), - ); - identity.validate()?; - if identity.encode()?.as_slice() != bytes { - return Err(Error::Identity("application identity is not canonical")); - } - Ok(identity) - } - - fn validate(self) -> Result<()> { - if self.tenant.as_bytes().iter().all(|byte| *byte == 0) - || self.application.as_bytes().iter().all(|byte| *byte == 0) - { - return Err(Error::Identity( - "tenant and application IDs must be nonzero", - )); - } - Ok(()) - } -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct RawIdentity { - application: String, - tenant: String, - version: u8, -} - -/// Owns strict creation and loading of one root's immutable application identity. -#[derive(Clone)] -pub struct ApplicationIdentityStore { - store: Store, - root: Path, - path: Path, -} - -impl ApplicationIdentityStore { - /// Binds the identity store to one storage prefix. - #[must_use] - pub fn new(store: Store, root: Path) -> Self { - let path = CellStorageLayout::root_identity_path(&root); - Self { store, root, path } - } - - /// Loads the persisted identity, returning absence only for a missing object. - pub async fn load(&self) -> Result> { - let body = match self - .store - .get_with_etag_bounded(&self.path, MAX_IDENTITY_BYTES) - .await - { - Ok((body, _)) => body, - Err(StorageError::NotFound { .. }) => return Ok(None), - Err(error) => return Err(error.into()), - }; - Ok(Some(ApplicationIdentity::decode(&body)?)) - } - - /// Strict-creates the proposed identity or adopts the exact concurrent winner. - pub async fn initialize(&self, proposed: ApplicationIdentity) -> Result { - proposed.validate()?; - let encoded = proposed.encode()?; - match self - .store - .create_strict(&self.path, Bytes::from(encoded)) - .await - { - Ok(()) => Ok(proposed), - Err(create_error) => match self.load().await? { - Some(current) if current == proposed => Ok(current), - Some(_) => Err(Error::Identity( - "authoritative root already belongs to another application", - )), - None => Err(create_error.into()), - }, - } - } - - /// Binds application-scoped paths only after reloading the persisted identity. - pub async fn layout(&self, identity: ApplicationIdentity) -> Result { - if self.load().await? != Some(identity) { - return Err(Error::Identity( - "application identity is not authoritative for this root", - )); - } - Ok(CellStorageLayout::new( - self.store.clone(), - self.root.clone(), - *identity.application().as_bytes(), - )) - } -} - -fn parse_id(value: &str, field: &'static str) -> Result<[u8; 16]> { - crate::identity::decode_hex(value).map_err(|_| Error::Identity(field)) -} diff --git a/crates/crab-cell-runtime/src/cell/catalog.rs b/crates/crab-cell-runtime/src/cell/catalog.rs deleted file mode 100644 index a71052400..000000000 --- a/crates/crab-cell-runtime/src/cell/catalog.rs +++ /dev/null @@ -1,738 +0,0 @@ -//! Sharded catalog of Cell entries, roles, and proofs. -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::{ETag, StorageError}; -use serde::{Deserialize, Serialize}; - -use crate::fleet::telemetry::{CatalogReadKind, CellTelemetryHandle}; -use crate::identity::encode_hex; -use crate::identity::{ApplicationId, CellId, CellTarget, Digest, NamespaceId, TenantId}; -use crate::retry::{Backoff, retry_hint, retryable_storage_error}; -use crate::{Error, Result}; - -// A version-two head carries one page locator per immutable page: a 64-hex -// digest and a 64-hex first Cell id. The full 256-page shard needs about -// 44 KiB, so the bound is one 64 KiB object read rather than the 32 KiB that -// held only digests. -const MAX_HEAD_BYTES: u64 = 64 * 1024; -const MAX_PAGE_BYTES: u64 = 1024 * 1024; -const ENTRIES_PER_PAGE: usize = 256; -const MAX_PAGES: usize = 256; -const MAX_ENTRIES: usize = ENTRIES_PER_PAGE * MAX_PAGES; - -/// Namespace behavior pinned when a Cell is first provisioned. -#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum CatalogRole { - /// The namespace serves repository content. - Repository, - /// The namespace serves SQL. - Sql, - /// The namespace serves key-value data. - Kv, - /// The namespace serves queue messages. - Queue, - /// The namespace serves workflow runs. - Workflow, - /// The namespace serves blob objects. - Blob, - /// The namespace serves cron schedules. - Cron, -} - -/// Immutable identity and bootstrap contract for one cataloged Cell. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct CatalogEntry { - cell: CellId, - namespace: NamespaceId, - partition: Vec, - role: CatalogRole, - initial_code: Digest, - initial_schema: u32, -} - -impl CatalogEntry { - /// Creates a bootstrap entry from a resolved target. - pub fn new( - target: &CellTarget, - role: CatalogRole, - initial_code: Digest, - initial_schema: u32, - ) -> Result { - if initial_schema == 0 { - return Err(Error::Catalog("initial schema is zero")); - } - Ok(Self { - cell: target.cell_id(), - namespace: target.namespace(), - partition: target.partition().to_vec(), - role, - initial_code, - initial_schema, - }) - } - - /// Returns the Cell this entry pins. - #[must_use] - pub const fn cell(&self) -> CellId { - self.cell - } - - /// Returns the namespace the Cell serves. - #[must_use] - pub const fn namespace(&self) -> NamespaceId { - self.namespace - } - - /// Returns the partition key that selected the Cell. - #[must_use] - pub fn partition(&self) -> &[u8] { - &self.partition - } - - /// Returns the namespace role provisioned for the Cell. - #[must_use] - pub const fn role(&self) -> CatalogRole { - self.role - } - - /// Returns the code digest provisioned for the Cell. - #[must_use] - pub const fn initial_code(&self) -> Digest { - self.initial_code - } - - /// Returns the schema version provisioned for the Cell. - #[must_use] - pub const fn initial_schema(&self) -> u32 { - self.initial_schema - } - - fn validate(&self, tenant: TenantId, application: ApplicationId) -> Result<()> { - if self.initial_schema == 0 || self.partition.len() > 1_024 { - return Err(Error::Catalog("invalid entry bounds")); - } - let target = CellTarget::new(tenant, application, self.namespace, &self.partition)?; - if target.cell_id() != self.cell { - return Err(Error::Catalog("entry Cell digest mismatch")); - } - Ok(()) - } -} - -/// Verified proof that an immutable catalog page is reachable from a shard head. -#[derive(Clone)] -pub struct CatalogProof { - tenant: TenantId, - application: ApplicationId, - entry: CatalogEntry, - revision: u64, -} - -/// One verified immutable catalog page from a revision-pinned shard scan. -pub struct CatalogScanPage { - revision: u64, - entries: Vec, -} - -impl CatalogScanPage { - /// Returns the catalog revision this page belongs to. - #[must_use] - pub const fn revision(&self) -> u64 { - self.revision - } - - /// Returns the proofs this page carries in key order. - #[must_use] - pub fn entries(&self) -> &[CatalogProof] { - &self.entries - } -} - -/// Stateful bounded scan over one immutable catalog-head snapshot. -pub struct CatalogShardScan { - catalog: CellCatalog, - shard: u8, - revision: u64, - pages: Vec, - next_page: usize, - previous: Option, -} - -impl CatalogShardScan { - /// Returns the pinned catalog revision. - #[must_use] - pub const fn revision(&self) -> u64 { - self.revision - } - - /// Returns the immutable pages pinned by this shard-head observation. - #[must_use] - pub fn page_digests(&self) -> Vec { - self.pages.iter().map(|page| page.digest).collect() - } - - /// Loads and verifies at most one 256-entry immutable page. - pub async fn next_page(&mut self) -> Result> { - if self.next_page >= self.pages.len() { - return Ok(None); - } - let entries = self - .catalog - .load_located_page(self.shard, &self.pages, self.next_page) - .await?; - let mut proofs = Vec::with_capacity(entries.len()); - for entry in entries { - if self - .previous - .is_some_and(|previous| previous.as_bytes() >= entry.cell.as_bytes()) - { - return Err(Error::Catalog("invalid scanned catalog ordering")); - } - self.previous = Some(entry.cell); - proofs.push(CatalogProof { - tenant: self.catalog.tenant, - application: self.catalog.application, - entry, - revision: self.revision, - }); - } - self.next_page += 1; - Ok(Some(CatalogScanPage { - revision: self.revision, - entries: proofs, - })) - } -} - -impl CatalogProof { - /// Creates a process-local identity proof for an actor-owned resident. - /// - /// Revision zero is intentional: this value is never written as catalog - /// authority or used for retention; the actor lifecycle is the freshness - /// boundary and the slow route still performs a verified catalog scan. - pub(crate) fn local(entry: CatalogEntry, target: &CellTarget) -> Self { - Self { - entry, - revision: 0, - tenant: target.tenant(), - application: target.application(), - } - } - - pub(crate) fn target(&self) -> Result { - CellTarget::new( - self.tenant, - self.application, - self.entry.namespace, - &self.entry.partition, - ) - } - - /// Returns the catalog entry this proof carries. - #[must_use] - pub const fn entry(&self) -> &CatalogEntry { - &self.entry - } - - /// Returns the revision the proof was read at. - #[must_use] - pub const fn revision(&self) -> u64 { - self.revision - } -} - -/// CAS-backed catalog whose immutable pages precede mutable head publication. -#[derive(Clone)] -pub struct CellCatalog { - layout: CellStorageLayout, - tenant: TenantId, - application: ApplicationId, - telemetry: CellTelemetryHandle, -} - -impl CellCatalog { - /// Binds the catalog to one Cell layout and tenant. - #[must_use] - pub fn new(layout: CellStorageLayout, tenant: TenantId) -> Self { - Self::with_telemetry(layout, tenant, CellTelemetryHandle::default()) - } - - /// Binds the catalog to one Cell layout, tenant, and telemetry sink. - /// - /// Routing and due-scan callers pass the node's handle so the metadata - /// plane's object-store reads appear beside the LTX origin counters. - #[must_use] - pub fn with_telemetry( - layout: CellStorageLayout, - tenant: TenantId, - telemetry: CellTelemetryHandle, - ) -> Self { - let application = ApplicationId::from_bytes(*layout.application_id()); - Self { - layout, - tenant, - application, - telemetry, - } - } - - /// Returns the application every entry is scoped to. - #[must_use] - pub const fn application(&self) -> ApplicationId { - self.application - } - - pub(crate) fn matches_identity( - &self, - identity: crate::cell::application::ApplicationIdentity, - ) -> bool { - self.tenant == identity.tenant() && self.application == identity.application() - } - - /// Adds one entry with immutable-page-before-head publication ordering. - pub async fn provision(&self, entry: CatalogEntry) -> Result { - entry.validate(self.tenant, self.application)?; - let shard = entry.cell.as_bytes()[0]; - let mut backoff = Backoff::default(); - loop { - let observed = self.load_head(shard).await?; - let Some(head) = self.insert_entry(shard, observed.as_ref(), &entry).await? else { - return Ok(CatalogProof { - tenant: self.tenant, - application: self.application, - entry, - revision: observed.as_ref().map_or(0, |head| head.head.revision), - }); - }; - let revision = head.revision; - let encoded = head.encode()?; - let path = self.layout.catalog_head_path(self.tenant.as_bytes(), shard); - let published = match observed { - Some(observed) => { - self.layout - .store() - .update(&path, Bytes::from(encoded), observed.token) - .await - } - None => { - self.layout - .store() - .create_strict_with_etag(&path, Bytes::from(encoded)) - .await - } - }; - match published { - Ok(_) => { - return Ok(CatalogProof { - entry, - revision, - tenant: self.tenant, - application: self.application, - }); - } - Err(error) => { - loop { - match self.lookup_after_failed_publish(&entry).await { - Ok(Some(proof)) => return Ok(proof), - Ok(None) => break, - Err(Error::Storage(load_error)) - if retryable_storage_error(&load_error) => - { - backoff.wait(retry_hint(&load_error)).await; - } - Err(load_error) => return Err(load_error), - } - } - if !retryable_storage_error(&error) { - return Err(error.into()); - } - backoff.wait(retry_hint(&error)).await; - } - } - } - } - - /// Loads one entry after checking the head locator and exactly one page. - /// - /// The head is a binary-search key list over immutable pages, so a lookup - /// reads the head and the one page that can hold the Cell. A locator that - /// disagrees with its page fails closed instead of reporting absence. - pub async fn lookup(&self, cell: CellId) -> Result> { - let shard = cell.as_bytes()[0]; - let Some(observed) = self.load_head(shard).await? else { - return Ok(None); - }; - let Some(index) = observed.head.page_index(cell) else { - // The Cell id sorts below every provisioned entry in this shard. - return Ok(None); - }; - let entries = self - .load_located_page(shard, &observed.head.pages, index) - .await?; - Ok(entries - .binary_search_by(|entry| entry.cell.as_bytes().cmp(cell.as_bytes())) - .ok() - .map(|index| CatalogProof { - tenant: self.tenant, - application: self.application, - entry: entries[index].clone(), - revision: observed.head.revision, - })) - } - - /// Pins one shard head for bounded immutable-page iteration. - pub async fn scan_shard(&self, shard: u8) -> Result { - let observed = self.load_head(shard).await?; - let (revision, pages) = observed - .map(|observed| (observed.head.revision, observed.head.pages)) - .unwrap_or_default(); - Ok(CatalogShardScan { - catalog: self.clone(), - shard, - revision, - pages, - next_page: 0, - previous: None, - }) - } - - pub(crate) async fn pinned_cells( - &self, - shard: u8, - revision: u64, - pages: &[Digest], - ) -> Result> { - Ok(self.load_pinned_pages(shard, revision, pages).await?.1) - } - - pub(crate) async fn install_pinned_shard( - &self, - shard: u8, - revision: u64, - pages: &[Digest], - ) -> Result<()> { - let (page_refs, _) = self.load_pinned_pages(shard, revision, pages).await?; - let observed = self.load_head(shard).await?; - if revision == 0 { - return if observed.is_none() { - Ok(()) - } else { - Err(Error::Catalog( - "empty restored shard conflicts with an existing head", - )) - }; - } - let head = CatalogHead { - revision, - pages: page_refs, - }; - let encoded = head.encode()?; - let path = self.layout.catalog_head_path(self.tenant.as_bytes(), shard); - match self - .layout - .store() - .create_strict_with_etag(&path, Bytes::from(encoded)) - .await - { - Ok(_) => Ok(()), - Err(create_error) => match self.load_head(shard).await? { - Some(current) - if current.head.revision == revision - && current - .head - .pages - .iter() - .map(|page| page.digest) - .eq(pages.iter().copied()) => - { - Ok(()) - } - Some(_) => Err(Error::Catalog( - "restored shard conflicts with an existing head", - )), - None => Err(create_error.into()), - }, - } - } - - async fn lookup_after_failed_publish( - &self, - expected: &CatalogEntry, - ) -> Result> { - match self.lookup(expected.cell).await? { - Some(proof) if proof.entry == *expected => Ok(Some(proof)), - Some(_) => Err(Error::CatalogCollision), - None => Ok(None), - } - } - - async fn load_head(&self, shard: u8) -> Result> { - let path = self.layout.catalog_head_path(self.tenant.as_bytes(), shard); - let started = std::time::Instant::now(); - let observed = self - .layout - .store() - .get_with_etag_bounded(&path, MAX_HEAD_BYTES) - .await; - let (body, token) = match observed { - Ok(value) => value, - Err(StorageError::NotFound { .. }) => { - self.telemetry - .catalog_read(CatalogReadKind::Head, started.elapsed(), true); - return Ok(None); - } - Err(error) => { - self.telemetry - .catalog_read(CatalogReadKind::Head, started.elapsed(), false); - return Err(error.into()); - } - }; - self.telemetry - .catalog_read(CatalogReadKind::Head, started.elapsed(), true); - Ok(Some(ObservedHead { - head: CatalogHead::decode(&body)?, - token, - })) - } - - /// Loads every entry of one shard head in key order. - /// - /// Provisioning reads the complete shard because it republishes the page - /// set. Routing uses `lookup`, which reads one page. - async fn load_entries(&self, shard: u8, head: &CatalogHead) -> Result> { - let mut entries = Vec::new(); - for index in 0..head.pages.len() { - for entry in self.load_located_page(shard, &head.pages, index).await? { - if entries.last().is_some_and(|previous: &CatalogEntry| { - previous.cell.as_bytes() >= entry.cell.as_bytes() - }) { - return Err(Error::Catalog("entries are not globally ordered")); - } - entries.push(entry); - } - } - if entries.len() > MAX_ENTRIES { - return Err(Error::Catalog("head exceeds entry limit")); - } - Ok(entries) - } - - /// Replace only the page that can contain this Cell. The head CAS remains - /// the serialization point, so a losing writer retries against fresh pages. - async fn insert_entry( - &self, - shard: u8, - observed: Option<&ObservedHead>, - entry: &CatalogEntry, - ) -> Result> { - let Some(observed) = observed else { - let reference = self.upload_page(std::slice::from_ref(entry)).await?; - return Ok(Some(CatalogHead { - revision: 1, - pages: vec![reference], - })); - }; - let mut pages = observed.head.pages.clone(); - let index = observed.head.page_index(entry.cell).unwrap_or(0); - let mut entries = self.load_located_page(shard, &pages, index).await?; - match entries.binary_search_by(|value| value.cell.as_bytes().cmp(entry.cell.as_bytes())) { - Ok(index) if entries[index] == *entry => return Ok(None), - Ok(_) => return Err(Error::CatalogCollision), - Err(index) => entries.insert(index, entry.clone()), - } - let revision = observed - .head - .revision - .checked_add(1) - .ok_or(Error::Catalog("head revision overflow"))?; - if entries.len() <= ENTRIES_PER_PAGE { - pages[index] = self.upload_page(&entries).await?; - return Ok(Some(CatalogHead { revision, pages })); - } - if pages.len() == MAX_PAGES { - // Full pages are not guaranteed after earlier splits. Repack once - // at the head limit before reporting that the shard is full. - let mut all = self.load_entries(shard, &observed.head).await?; - let position = match all - .binary_search_by(|value| value.cell.as_bytes().cmp(entry.cell.as_bytes())) - { - Ok(_) => return Err(Error::Catalog("catalog page insertion changed")), - Err(position) => position, - }; - all.insert(position, entry.clone()); - if all.len() > MAX_ENTRIES { - return Err(Error::CatalogFull); - } - return self.upload_pages(revision, &all).await.map(Some); - } - let right = entries.split_off(entries.len() / 2); - let left = self.upload_page(&entries).await?; - let right = self.upload_page(&right).await?; - pages.splice(index..=index, [left, right]); - Ok(Some(CatalogHead { revision, pages })) - } - - /// Loads one immutable page and proves it belongs where the locator says. - /// - /// The checks make a head that disagrees with its pages fail closed: the - /// page must open at the located first Cell id, stay inside the shard, - /// remain ordered, and end below the next locator key. - async fn load_located_page( - &self, - shard: u8, - pages: &[CatalogPageRef], - index: usize, - ) -> Result> { - let reference = pages - .get(index) - .ok_or(Error::Catalog("invalid head page count"))?; - if reference.first.as_bytes()[0] != shard { - return Err(Error::Catalog("catalog page locator names another shard")); - } - let entries = self.load_page(reference.digest).await?; - let mut previous: Option<&[u8]> = None; - for entry in &entries { - entry.validate(self.tenant, self.application)?; - if entry.cell.as_bytes()[0] != shard - || previous.is_some_and(|value| value >= entry.cell.as_bytes().as_slice()) - { - return Err(Error::Catalog("invalid located catalog ordering or shard")); - } - previous = Some(entry.cell.as_bytes()); - } - if entries - .first() - .is_none_or(|entry| entry.cell != reference.first) - { - return Err(Error::Catalog( - "catalog page locator disagrees with its page", - )); - } - if let Some(next) = pages.get(index + 1) - && entries - .last() - .is_some_and(|entry| entry.cell.as_bytes() >= next.first.as_bytes()) - { - return Err(Error::Catalog( - "catalog page locator disagrees with its page", - )); - } - Ok(entries) - } - - /// Loads a pinned page list and returns its locators and Cells. - /// - /// Backup restore replays a manifest that names page digests only, so the - /// locators are rebuilt from the verified pages here. - async fn load_pinned_pages( - &self, - shard: u8, - revision: u64, - pages: &[Digest], - ) -> Result<(Vec, Vec)> { - if pages.len() > MAX_PAGES || (revision == 0) != pages.is_empty() { - return Err(Error::Catalog("invalid pinned catalog head")); - } - let mut unique = std::collections::HashSet::with_capacity(pages.len()); - if pages - .iter() - .any(|digest| !unique.insert(*digest.as_bytes())) - { - return Err(Error::Catalog("duplicate pinned catalog page")); - } - let mut page_refs = Vec::with_capacity(pages.len()); - let mut cells = Vec::new(); - for digest in pages { - let entries = self.load_page(*digest).await?; - let Some(first) = entries.first().map(|entry| entry.cell) else { - return Err(Error::Catalog("invalid page entry count")); - }; - for entry in &entries { - entry.validate(self.tenant, self.application)?; - if entry.cell.as_bytes()[0] != shard - || cells.last().is_some_and(|previous: &CellId| { - previous.as_bytes() >= entry.cell.as_bytes() - }) - { - return Err(Error::Catalog("invalid pinned catalog ordering or shard")); - } - cells.push(entry.cell); - } - page_refs.push(CatalogPageRef { - digest: *digest, - first, - }); - } - if cells.len() > MAX_ENTRIES { - return Err(Error::Catalog("pinned catalog exceeds entry limit")); - } - Ok((page_refs, cells)) - } - - async fn load_page(&self, digest: Digest) -> Result> { - let path = self.layout.catalog_object_path(digest.as_bytes()); - let started = std::time::Instant::now(); - let observed = self - .layout - .store() - .get_with_etag_bounded(&path, MAX_PAGE_BYTES) - .await; - let (body, _) = match observed { - Ok(value) => value, - Err(error) => { - self.telemetry - .catalog_read(CatalogReadKind::Page, started.elapsed(), false); - return Err(error.into()); - } - }; - self.telemetry - .catalog_read(CatalogReadKind::Page, started.elapsed(), true); - if blake3::hash(&body).as_bytes() != digest.as_bytes() { - return Err(Error::Catalog("page digest mismatch")); - } - let page = CatalogPage::decode(&body)?; - if page.entries.is_empty() || page.entries.len() > ENTRIES_PER_PAGE { - return Err(Error::Catalog("invalid page entry count")); - } - Ok(page.entries) - } - - async fn upload_pages(&self, revision: u64, entries: &[CatalogEntry]) -> Result { - let mut pages = Vec::new(); - for entries in entries.chunks(ENTRIES_PER_PAGE) { - pages.push(self.upload_page(entries).await?); - } - if pages.is_empty() || pages.len() > MAX_PAGES { - return Err(Error::Catalog("invalid head page count")); - } - Ok(CatalogHead { revision, pages }) - } - - async fn upload_page(&self, entries: &[CatalogEntry]) -> Result { - let first = entries - .first() - .ok_or(Error::Catalog("empty catalog page"))? - .cell; - let encoded = CatalogPage { - entries: entries.to_vec(), - } - .encode()?; - if encoded.len() as u64 > MAX_PAGE_BYTES { - return Err(Error::Catalog("encoded page exceeds 1 MiB")); - } - let digest = Digest::from_bytes(*blake3::hash(&encoded).as_bytes()); - self.layout - .store() - .put( - &self.layout.catalog_object_path(digest.as_bytes()), - Bytes::from(encoded), - ) - .await?; - Ok(CatalogPageRef { digest, first }) - } -} - -mod codec; - -use codec::*; diff --git a/crates/crab-cell-runtime/src/cell/catalog/codec.rs b/crates/crab-cell-runtime/src/cell/catalog/codec.rs deleted file mode 100644 index b610d40fc..000000000 --- a/crates/crab-cell-runtime/src/cell/catalog/codec.rs +++ /dev/null @@ -1,242 +0,0 @@ -//! Catalog head and page wire shapes with their hex codecs. - -use super::*; -use crate::identity::nibble; - -pub(super) struct ObservedHead { - pub(super) head: CatalogHead, - pub(super) token: ETag, -} - -/// One immutable catalog page and the first Cell id it can contain. -/// -/// The head carries these locators so a reader finds an entry with one page -/// read. Without them a lookup must download every page in the shard, which -/// makes cold routing cost grow with the Cell population. -#[derive(Clone)] -pub(super) struct CatalogPageRef { - pub(super) digest: Digest, - pub(super) first: CellId, -} - -pub(super) struct CatalogHead { - pub(super) revision: u64, - pub(super) pages: Vec, -} - -impl CatalogHead { - pub(super) fn encode(&self) -> Result> { - self.validate()?; - let encoded = serde_json::to_vec(&RawHead { - version: HEAD_VERSION, - revision: self.revision.to_string(), - pages: self - .pages - .iter() - .map(|page| RawPageRef { - digest: encode_hex(page.digest.as_bytes()), - first: encode_hex(page.first.as_bytes()), - }) - .collect(), - })?; - if encoded.len() as u64 > MAX_HEAD_BYTES { - return Err(Error::Catalog("encoded head exceeds 64 KiB")); - } - Ok(encoded) - } - - pub(super) fn decode(body: &[u8]) -> Result { - let raw: RawHead = serde_json::from_slice(body)?; - if raw.version != HEAD_VERSION { - return Err(Error::Catalog("unsupported head version")); - } - if raw.pages.is_empty() || raw.pages.len() > MAX_PAGES { - return Err(Error::Catalog("invalid head page count")); - } - let revision = canonical_u64(&raw.revision)?; - let pages = raw - .pages - .iter() - .map(|page| { - Ok(CatalogPageRef { - digest: Digest::from_bytes(decode_fixed(&page.digest)?), - first: CellId::from_bytes(decode_fixed(&page.first)?), - }) - }) - .collect::>>()?; - let head = Self { revision, pages }; - if head.encode()?.as_slice() != body { - return Err(Error::Catalog("head JSON is not canonical")); - } - Ok(head) - } - - /// Validates the version-two head invariants that both codecs share. - fn validate(&self) -> Result<()> { - if self.revision == 0 || self.pages.is_empty() || self.pages.len() > MAX_PAGES { - return Err(Error::Catalog("invalid head bounds")); - } - let mut unique = std::collections::HashSet::with_capacity(self.pages.len()); - let mut previous: Option<&[u8]> = None; - for page in &self.pages { - if !unique.insert(*page.digest.as_bytes()) { - return Err(Error::Catalog("duplicate page digest")); - } - // The locator is a binary search key list: strictly increasing - // `first` values are what make one page answer exact. - if previous.is_some_and(|value| value >= page.first.as_bytes().as_slice()) { - return Err(Error::Catalog("catalog page locator is not ordered")); - } - previous = Some(page.first.as_bytes()); - } - Ok(()) - } - - /// Returns the index of the one page that can hold `cell`. - /// - /// `None` means the Cell id sorts below every provisioned entry in the - /// shard, so no page can name it. - pub(super) fn page_index(&self, cell: CellId) -> Option { - self.pages - .partition_point(|page| page.first.as_bytes() <= cell.as_bytes()) - .checked_sub(1) - } -} - -pub(super) struct CatalogPage { - pub(super) entries: Vec, -} - -impl CatalogPage { - pub(super) fn encode(&self) -> Result> { - let entries = self.entries.iter().map(RawEntry::from).collect(); - Ok(serde_json::to_vec(&RawPage { - version: 1, - entries, - })?) - } - - pub(super) fn decode(body: &[u8]) -> Result { - let raw: RawPage = serde_json::from_slice(body)?; - if raw.version != 1 { - return Err(Error::Catalog("unsupported page version")); - } - let entries = raw - .entries - .into_iter() - .map(CatalogEntry::try_from) - .collect::>>()?; - let page = Self { entries }; - if page.encode()?.as_slice() != body { - return Err(Error::Catalog("page JSON is not canonical")); - } - Ok(page) - } -} - -const HEAD_VERSION: u32 = 2; - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(super) struct RawHead { - version: u32, - revision: String, - pages: Vec, -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(super) struct RawPageRef { - digest: String, - first: String, -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(super) struct RawPage { - version: u32, - entries: Vec, -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(super) struct RawEntry { - cell: String, - namespace: String, - partition: String, - role: CatalogRole, - initial_code: String, - initial_schema: u32, -} - -impl From<&CatalogEntry> for RawEntry { - fn from(entry: &CatalogEntry) -> Self { - Self { - cell: encode_hex(entry.cell.as_bytes()), - namespace: encode_hex(entry.namespace.as_bytes()), - partition: encode_hex(&entry.partition), - role: entry.role, - initial_code: encode_hex(entry.initial_code.as_bytes()), - initial_schema: entry.initial_schema, - } - } -} - -impl TryFrom for CatalogEntry { - type Error = Error; - - fn try_from(raw: RawEntry) -> Result { - if raw.initial_schema == 0 { - return Err(Error::Catalog("initial schema is zero")); - } - Ok(Self { - cell: CellId::from_bytes(decode_fixed(&raw.cell)?), - namespace: NamespaceId::from_bytes(decode_fixed(&raw.namespace)?), - partition: decode_partition(&raw.partition)?, - role: raw.role, - initial_code: Digest::from_bytes(decode_fixed(&raw.initial_code)?), - initial_schema: raw.initial_schema, - }) - } -} - -pub(super) fn canonical_u64(value: &str) -> Result { - let parsed = value - .parse::() - .map_err(|_| Error::Catalog("invalid decimal u64"))?; - if parsed == 0 || parsed.to_string() != value { - return Err(Error::Catalog("noncanonical decimal u64")); - } - Ok(parsed) -} - -pub(super) fn decode_fixed(value: &str) -> Result<[u8; N]> { - let decoded = decode_hex(value)?; - decoded - .try_into() - .map_err(|_| Error::Catalog("fixed hex length")) -} - -pub(super) fn decode_partition(value: &str) -> Result> { - if value.len() > 2_048 { - return Err(Error::Catalog("partition exceeds 1024 bytes")); - } - decode_hex(value) -} - -pub(super) fn decode_hex(value: &str) -> Result> { - if !value.len().is_multiple_of(2) { - return Err(Error::Catalog("hex length is odd")); - } - value - .as_bytes() - .as_chunks::<2>() - .0 - .iter() - .map(|pair| { - let high = nibble(pair[0]).ok_or(Error::Catalog("hex is not lowercase"))?; - let low = nibble(pair[1]).ok_or(Error::Catalog("hex is not lowercase"))?; - Ok((high << 4) | low) - }) - .collect() -} diff --git a/crates/crab-cell-runtime/src/cell/due.rs b/crates/crab-cell-runtime/src/cell/due.rs deleted file mode 100644 index 65f49c56a..000000000 --- a/crates/crab-cell-runtime/src/cell/due.rs +++ /dev/null @@ -1,108 +0,0 @@ -//! Bounded due hints: one small key per released Cell deadline. -//! -//! A hint is an accelerator, never authority. The owner writes one key when it -//! releases a Cell that still has a due time, and the scheduler lists the -//! buckets up to now and confirms each candidate against its own control before -//! ticking it. A missing, stale, or unreadable hint costs a slower discovery, -//! never a missed or duplicated deadline: the full catalog scan remains the -//! backstop, and the Cell's own SQLite state decides what is due. - -use crab_ltx::CellStorageLayout; -use object_store::path::Path; - -use crate::identity::{CellId, decode_hex}; -use crate::{Error, Result}; - -/// Milliseconds covered by one hint bucket. -pub const HINT_BUCKET_MS: i64 = 60_000; -/// Buckets a listing walks back from the current bucket. -pub const HINT_LOOKBACK_BUCKETS: u32 = 5; -/// Hints one listing returns. -const MAX_HINT_BATCH: usize = 128; - -/// Records one released Cell's next due time. -/// -/// The write is best effort by contract: the caller may ignore an error and -/// still be correct, because the backstop scan covers a missing hint. -/// -/// A deadline already older than the listing window is not written at all: no -/// listing would ever see that key, so writing it would only leave an object -/// behind that nothing consumes. The backstop covers that deadline instead. -pub async fn publish( - layout: &CellStorageLayout, - cell: CellId, - due_ms: i64, - now_ms: i64, -) -> Result<()> { - let bucket = bucket_for(due_ms)?; - let current = bucket_for(now_ms)?; - if current.saturating_sub(bucket) > u64::from(HINT_LOOKBACK_BUCKETS) { - return Ok(()); - } - layout - .store() - .put( - &layout.due_hint_path(bucket, cell.as_bytes()), - bytes::Bytes::new(), - ) - .await?; - Ok(()) -} - -/// Lists and consumes up to `limit` hints whose bucket has arrived. -/// -/// A returned hint is deleted: the scheduler holds the candidate list, the -/// backstop covers anything the caller loses, and a stale hint would otherwise -/// be listed on every cycle. -pub async fn take(layout: &CellStorageLayout, now_ms: i64, limit: usize) -> Result> { - let bucket = bucket_for(now_ms)?; - let mut cells = Vec::new(); - for offset in 0..=HINT_LOOKBACK_BUCKETS { - if cells.len() >= limit || bucket < u64::from(offset) { - break; - } - let bucket = bucket - u64::from(offset); - list_bucket(layout, bucket, limit - cells.len(), &mut cells).await?; - } - Ok(cells) -} - -/// Returns the bucket one due time belongs to. -pub fn bucket_for(due_ms: i64) -> Result { - if due_ms < 0 { - return Err(Error::Control("due hint time is negative")); - } - Ok(due_ms as u64 / HINT_BUCKET_MS as u64) -} - -async fn list_bucket( - layout: &CellStorageLayout, - bucket: u64, - limit: usize, - cells: &mut Vec, -) -> Result<()> { - let prefix = layout.due_hint_prefix(bucket); - let mut stream = layout.store().list_stream(&prefix); - let max_cells = cells.len().saturating_add(limit.min(MAX_HINT_BATCH)); - while cells.len() < max_cells { - let Some(item) = futures_util::StreamExt::next(&mut stream).await else { - break; - }; - let meta = item?; - let Some(cell) = hint_cell(layout, bucket, &meta.location) else { - // A key this module did not write must not be listed forever. - let _ = layout.store().delete(&meta.location).await; - continue; - }; - let _ = layout.store().delete(&meta.location).await; - cells.push(cell); - } - Ok(()) -} - -/// Accepts only the canonical key for a Cell in this bucket. -fn hint_cell(layout: &CellStorageLayout, bucket: u64, location: &Path) -> Option { - let stem = location.filename()?.strip_suffix(".json")?; - let bytes = decode_hex::<32>(stem).ok()?; - (layout.due_hint_path(bucket, &bytes) == *location).then_some(CellId::from_bytes(bytes)) -} diff --git a/crates/crab-cell-runtime/src/cell/executor.rs b/crates/crab-cell-runtime/src/cell/executor.rs deleted file mode 100644 index cb5ff96d8..000000000 --- a/crates/crab-cell-runtime/src/cell/executor.rs +++ /dev/null @@ -1,1309 +0,0 @@ -//! Single-Cell executor: commands, queries, migrations, and pending publication. -use std::collections::VecDeque; - -use crab_ltx::{CaptureBatch, Db, TransactionError, rusqlite::OptionalExtension}; - -use crate::cell::catalog::CatalogRole; -use crate::identity::{CellId, Digest}; -use crate::identity::{IncarnationId, RequestId}; -use crate::primitives::maintenance::PersistedWorkInventory; -use crate::primitives::maintenance::TransferWorkInventory; -use crate::{Error, Result}; - -const MAX_RESULT_BYTES: usize = crate::codec::MAX_WIRE_BYTES; -const MAX_REQUEST_LIFETIME_MS: i64 = 24 * 60 * 60 * 1000; -const MAX_ISSUED_FUTURE_MS: i64 = 5 * 60 * 1000; -const REQUEST_RETENTION_MS: i64 = 24 * 60 * 60 * 1000; -pub(crate) const MAX_PENDING_PUBLICATIONS: usize = 64; -pub(crate) const PENDING_PUBLICATION_HIGH_WATER_BYTES: u64 = 64 << 20; - -/// Stable caller identity retained across retries and outcome resolution. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct MutationIdentity { - /// Caller-supplied request identity. - pub request_id: RequestId, - /// Logical time the caller created the mutation. - pub issued_at_ms: i64, - /// Logical time after which the identity is refused. - pub expires_at_ms: i64, -} - -impl MutationIdentity { - pub(crate) fn validate(self, now_ms: i64) -> Result<()> { - self.validate_bounds(now_ms)?; - if self.expires_at_ms <= now_ms { - return Err(Error::Command("invalid mutation identity lifetime")); - } - Ok(()) - } - - fn validate_bounds(self, now_ms: i64) -> Result<()> { - if now_ms < 0 - || self.issued_at_ms < 0 - || self.expires_at_ms <= self.issued_at_ms - || self.expires_at_ms - self.issued_at_ms > MAX_REQUEST_LIFETIME_MS - || self.issued_at_ms > now_ms.saturating_add(MAX_ISSUED_FUTURE_MS) - { - return Err(Error::Command("invalid mutation identity lifetime")); - } - Ok(()) - } - - pub(crate) fn expired(self, now_ms: i64) -> Result { - self.validate_bounds(now_ms)?; - Ok(self.expires_at_ms <= now_ms) - } -} - -/// Bounded handler decision made inside the application savepoint. -pub enum HandlerOutcome { - /// The handler committed a successful result. - Success(Vec), - /// The handler committed a rejection. - Rejected(Vec), -} - -/// A durable result stored in `sys_requests`. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum StoredOutcome { - /// Committed success with the sequence it was stored at. - Success { - /// Encoded result bytes. - result: Vec, - /// Commit sequence the outcome was stored at. - commit_sequence: u64, - }, - /// Committed rejection with the sequence it was stored at. - Rejected { - /// Encoded result bytes. - result: Vec, - /// Commit sequence the outcome was stored at. - commit_sequence: u64, - }, -} - -/// Authoritative request-ledger observation from the current Cell owner. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum Resolution { - /// The request has a durable outcome. - Committed(StoredOutcome), - /// No outcome is stored for the request. - Absent, - /// The owning Cell could not be reached, so the outcome is unknown. - Unknown, - /// The request identity expired before an outcome was stored. - Expired, -} - -impl StoredOutcome { - /// Returns the commit sequence the outcome was stored at. - #[must_use] - pub fn commit_sequence(&self) -> u64 { - match self { - Self::Success { - commit_sequence, .. - } - | Self::Rejected { - commit_sequence, .. - } => *commit_sequence, - } - } - - /// Returns the encoded result bytes. - #[must_use] - pub fn result(&self) -> &[u8] { - match self { - Self::Success { result, .. } | Self::Rejected { result, .. } => result, - } - } -} - -/// Locally committed command retained until immutable upload and control CAS. -#[derive(Clone)] -pub struct PendingCommit { - outcome: StoredOutcome, - logical_time_ms: i64, - next_due_ms: Option, - cuts: CaptureBatch, - prepared: Option, - durable: bool, -} - -/// Locally committed schema step retained until its root and control pair publish. -#[derive(Clone)] -pub struct PendingMigration { - code: Digest, - from_schema: u32, - to_schema: u32, - digest: Option, - commit_sequence: u64, - next_due_ms: Option, - cuts: CaptureBatch, - prepared: Option, -} - -impl PendingMigration { - /// Returns the application code digest the migration installs. - #[must_use] - pub const fn code(&self) -> Digest { - self.code - } - - /// Returns the schema version the migration starts from. - #[must_use] - pub const fn from_schema(&self) -> u32 { - self.from_schema - } - - /// Returns the schema version the migration installs. - #[must_use] - pub const fn to_schema(&self) -> u32 { - self.to_schema - } - - /// Returns the migration plan digest, when the plan declares one. - #[must_use] - pub const fn digest(&self) -> Option { - self.digest - } - - /// Returns the commit sequence the migration committed at. - #[must_use] - pub const fn commit_sequence(&self) -> u64 { - self.commit_sequence - } - - /// Returns the logical time the owner asked to be renewed by. - #[must_use] - pub const fn next_due_ms(&self) -> Option { - self.next_due_ms - } - - /// Returns the captured cuts awaiting publication. - #[must_use] - pub const fn cuts(&self) -> &CaptureBatch { - &self.cuts - } - - #[must_use] - pub(crate) fn retained_bytes(&self) -> u64 { - retained_bytes(&self.cuts) - } -} - -/// Durably proven identity of one completed schema migration. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct MigrationOutcome { - /// Application code digest the migration installed. - pub code: Digest, - /// Schema version the migration installed. - pub schema: u32, - /// Commit sequence the migration committed at. - pub commit_sequence: u64, -} - -impl PendingCommit { - /// Returns the durable outcome awaiting publication. - #[must_use] - pub fn outcome(&self) -> &StoredOutcome { - &self.outcome - } - - /// Returns the logical time the commit was made at. - #[must_use] - pub fn logical_time_ms(&self) -> i64 { - self.logical_time_ms - } - - /// Returns the logical time the owner asked to be renewed by. - #[must_use] - pub fn next_due_ms(&self) -> Option { - self.next_due_ms - } - - /// Returns the captured cuts awaiting publication. - #[must_use] - pub fn cuts(&self) -> &CaptureBatch { - &self.cuts - } - - /// Returns the prepared root, once one has been uploaded. - #[must_use] - pub fn prepared(&self) -> Option { - self.prepared - } - - #[must_use] - pub(crate) fn retained_bytes(&self) -> u64 { - retained_bytes(&self.cuts) - } -} - -/// Bytes the captured cuts retain until they publish. -fn retained_bytes(cuts: &CaptureBatch) -> u64 { - cuts.segments - .iter() - .map(|segment| segment.info().size_bytes) - .sum() -} - -/// Immediate executor result; pending output cannot be observed before publication. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum CommandExecution { - /// The command has a durable outcome. - Recorded(StoredOutcome), - /// The command committed locally and waits for publication. - Pending, -} - -/// Single-threaded SQL owner for one active Cell. -/// -/// The actor may continue after the preceding commit has a durability proof. -/// Dropping a caller does not remove queued cuts or their result. Only -/// `confirm_published` releases retained files after an authoritative root -/// matches the oldest local commit. -pub struct CellExecutor { - db: Db, - cell: CellId, - incarnation: IncarnationId, - schema: u32, - pending: VecDeque, - pending_bytes: u64, - published_sequence: u64, - pending_migration: Option, - fenced: bool, -} - -impl CellExecutor { - pub(crate) fn interrupt_handle(&self) -> crab_ltx::rusqlite::InterruptHandle { - self.db.interrupt_handle() - } - - /// Creates the executor for one active Cell at the given schema version. - #[must_use] - pub fn new(db: Db, cell: CellId, incarnation: IncarnationId, schema: u32) -> Self { - Self { - db, - cell, - incarnation, - schema, - pending: VecDeque::new(), - pending_bytes: 0, - published_sequence: 0, - pending_migration: None, - fenced: false, - } - } - - pub(crate) fn bootstrap( - mut db: Db, - cell: CellId, - incarnation: IncarnationId, - schema: u32, - initialize: impl FnOnce(&crab_ltx::rusqlite::Transaction<'_>) -> Result<()>, - ) -> Result<(Self, CaptureBatch, Option)> { - let initialized = db.transaction_with(|transaction| { - crate::cell::schema::install_runtime_schema_in(transaction, cell, incarnation, schema)?; - initialize(transaction)?; - crate::primitives::capacity::validate(transaction)?; - crate::fleet::scheduler::scheduler_next_due_ms(transaction, 0) - }); - let next_due_ms = match initialized { - Ok(next_due_ms) => next_due_ms, - Err(error) => { - let error = transaction_error_with_io(&db, error); - let _ = db.close(); - return Err(error); - } - }; - // Bootstrap publishes these exact bytes before activation. Local file - // durability would duplicate the root publication proof. - let cuts = match db.capture_deferred() { - Ok(cuts) if !cuts.segments.is_empty() => cuts, - Ok(_) => { - let _ = db.close(); - return Err(Error::Control("bootstrap produced no LTX cut")); - } - Err(error) => { - let _ = db.close(); - return Err(error.into()); - } - }; - Ok((Self::new(db, cell, incarnation, schema), cuts, next_due_ms)) - } - - pub(crate) fn from_restored( - mut db: Db, - cell: CellId, - incarnation: IncarnationId, - schema: u32, - root: crab_ltx::RootRef, - ) -> Result { - let expected_sequence = match i64::try_from(root.commit_sequence) { - Ok(sequence) => sequence, - Err(_) => { - let _ = db.close(); - return Err(Error::Control("root commit sequence exceeds SQLite range")); - } - }; - if root.cell != *cell.as_bytes() - || root.incarnation != *incarnation.as_bytes() - || db.position() != root.position - { - let _ = db.close(); - return Err(Error::Control( - "restored SQLite position does not match root", - )); - } - let verification = db.query_with(|connection| { - let metadata = connection.query_row( - "SELECT cell_id, incarnation, commit_sequence, logical_time_ms, schema_version FROM sys_meta WHERE singleton = 1", - [], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, i64>(3)?, - row.get::<_, u32>(4)?, - )) - }, - )?; - let latest_ledger = connection.query_row( - "SELECT COALESCE(MAX(commit_sequence), 0) FROM (SELECT commit_sequence FROM sys_requests UNION ALL SELECT commit_sequence FROM sys_inbox)", - [], - |row| row.get::<_, i64>(0), - )?; - if metadata.0.as_slice() != cell.as_bytes() - || metadata.1.as_slice() != incarnation.as_bytes() - || metadata.2 != expected_sequence - || metadata.3 < 0 - || metadata.4 != schema - || latest_ledger > metadata.2 - { - return Err(Error::Control( - "restored SQLite metadata does not match authoritative root", - )); - } - Ok(()) - }); - if let Err(error) = verification { - let error = db.take_io_error().map_or_else( - || match error { - crab_ltx::QueryError::Operation(error) => error, - crab_ltx::QueryError::Sqlite(error) => error.into(), - crab_ltx::QueryError::State(error) => error.into(), - }, - ltx_error, - ); - let _ = db.close(); - return Err(error); - } - let mut executor = Self::new(db, cell, incarnation, schema); - executor.published_sequence = root.commit_sequence; - Ok(executor) - } - - /// Executes one accepted command or returns its already published result. - pub fn execute( - &mut self, - identity: MutationIdentity, - operation_digest: Digest, - now_ms: i64, - max_result_bytes: usize, - handler: impl FnOnce(&crab_ltx::rusqlite::Transaction<'_>) -> Result, - ) -> Result { - if self.fenced { - return Err(Error::Fenced); - } - if !self.accepts_publication() { - return Err(Error::PendingPublication); - } - if max_result_bytes > MAX_RESULT_BYTES { - return Err(Error::Command("result exceeds wire limit")); - } - identity.validate(now_ms)?; - let cell = self.cell; - let incarnation = self.incarnation; - let schema = self.schema; - let transaction = self.db.transaction_with(|transaction| { - let (commit_sequence, prior_logical_time_ms) = - runtime_metadata(transaction, cell, incarnation, schema)?; - - let existing = transaction - .query_row( - "SELECT operation_digest, outcome, result, commit_sequence FROM sys_requests WHERE request_id = ?1", - [identity.request_id.as_bytes().as_slice()], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, Vec>(2)?, - row.get::<_, i64>(3)?, - )) - }, - ) - .optional()?; - if let Some((digest, outcome, result, sequence)) = existing { - if digest.as_slice() != operation_digest.as_bytes() { - return Err(Error::RequestConflict); - } - let outcome = stored_outcome(outcome, result, sequence)?; - if outcome.result().len() > max_result_bytes { - return Err(Error::Command("stored result exceeds command limit")); - } - return Ok(TransactionResult::Recorded(outcome)); - } - - let sequence = commit_sequence - .checked_add(1) - .filter(|value| *value > 0) - .ok_or(Error::Command("commit sequence overflow"))?; - let logical_time_ms = now_ms.max(prior_logical_time_ms); - transaction.execute_batch("SAVEPOINT application")?; - let decision = handler(transaction)?; - let (outcome, result) = match decision { - HandlerOutcome::Success(result) => { - transaction.execute_batch("RELEASE application")?; - (1, result) - } - HandlerOutcome::Rejected(result) => { - transaction.execute_batch( - "ROLLBACK TO application; RELEASE application", - )?; - (2, result) - } - }; - if result.len() > max_result_bytes { - return Err(Error::Command("handler result exceeds command limit")); - } - let retain_until_ms = identity - .expires_at_ms - .checked_add(REQUEST_RETENTION_MS) - .ok_or(Error::Command("request retention overflow"))?; - transaction.execute( - "INSERT INTO sys_requests(request_id, operation_digest, outcome, result, commit_sequence, expires_at_ms, retain_until_ms) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7)", - ( - identity.request_id.as_bytes().as_slice(), - operation_digest.as_bytes().as_slice(), - outcome, - result.as_slice(), - sequence, - identity.expires_at_ms, - retain_until_ms, - ), - )?; - if transaction.execute( - "UPDATE sys_meta SET commit_sequence = ?1, logical_time_ms = ?2 WHERE singleton = 1", - (sequence, logical_time_ms), - )? != 1 - { - return Err(Error::Command("runtime metadata row missing")); - } - crate::primitives::capacity::validate(transaction)?; - let next_due_ms = crate::fleet::scheduler::scheduler_next_due_ms(transaction, logical_time_ms)?; - Ok(TransactionResult::Committed { - outcome: stored_outcome(outcome, result, sequence)?, - logical_time_ms, - next_due_ms, - }) - }); - - self.finish_transaction(transaction) - } - - /// Applies or replays one destination inbox operation through normal publication. - pub fn deliver_effect( - &mut self, - delivery: crate::primitives::effects::InboxDelivery, - now_ms: i64, - max_result_bytes: usize, - handler: impl FnOnce(&crab_ltx::rusqlite::Transaction<'_>) -> Result, - ) -> Result { - if self.fenced { - return Err(Error::Fenced); - } - if !self.accepts_publication() { - return Err(Error::PendingPublication); - } - if max_result_bytes > MAX_RESULT_BYTES { - return Err(Error::Command("effect result exceeds wire limit")); - } - let cell = self.cell; - let incarnation = self.incarnation; - let schema = self.schema; - let transaction = self.db.transaction_with(|transaction| { - let (commit_sequence, prior_logical_time_ms) = - runtime_metadata(transaction, cell, incarnation, schema)?; - let applied = crate::primitives::effects::inbox_apply( - transaction, - now_ms, - delivery, - max_result_bytes, - handler, - )?; - let (outcome, duplicate) = match applied { - crate::primitives::effects::InboxApplyOutcome::Success { - result, - commit_sequence, - duplicate, - } => ( - StoredOutcome::Success { - result, - commit_sequence, - }, - duplicate, - ), - crate::primitives::effects::InboxApplyOutcome::Rejected { - result, - commit_sequence, - duplicate, - } => ( - StoredOutcome::Rejected { - result, - commit_sequence, - }, - duplicate, - ), - crate::primitives::effects::InboxApplyOutcome::Conflict => return Err(Error::RequestConflict), - crate::primitives::effects::InboxApplyOutcome::Expired => return Err(Error::EffectExpired), - }; - if duplicate { - return Ok(TransactionResult::Recorded(outcome)); - } - let sequence = commit_sequence - .checked_add(1) - .filter(|value| *value > 0) - .ok_or(Error::Command("commit sequence overflow"))?; - let published_sequence = u64::try_from(sequence) - .map_err(|_| Error::Command("effect destination sequence overflow"))?; - if outcome.commit_sequence() != published_sequence { - return Err(Error::Command("effect inbox sequence does not follow Cell state")); - } - let logical_time_ms = now_ms.max(prior_logical_time_ms); - if transaction.execute( - "UPDATE sys_meta SET commit_sequence = ?1, logical_time_ms = ?2 WHERE singleton = 1", - (sequence, logical_time_ms), - )? != 1 - { - return Err(Error::Command("runtime metadata row missing")); - } - crate::primitives::capacity::validate(transaction)?; - let next_due_ms = crate::fleet::scheduler::scheduler_next_due_ms(transaction, logical_time_ms)?; - Ok(TransactionResult::Committed { - outcome, - logical_time_ms, - next_due_ms, - }) - }); - self.finish_transaction(transaction) - } - - /// Runs one bounded read from the actor-serialized logical head. - pub fn query( - &mut self, - max_result_bytes: usize, - handler: impl FnOnce(&crab_ltx::rusqlite::Connection) -> Result>, - ) -> Result> { - if self.fenced { - return Err(Error::Fenced); - } - if !self.logical_head_is_durable() { - return Err(Error::PendingPublication); - } - if max_result_bytes > MAX_RESULT_BYTES { - return Err(Error::Command("result exceeds wire limit")); - } - let result = self.db.query_with(handler); - if let Some(error) = self.db.take_io_error() { - self.fenced = true; - return Err(ltx_error(error)); - } - match result { - Ok(result) if result.len() <= max_result_bytes => Ok(result), - Ok(_) => Err(Error::Command("query result exceeds command limit")), - Err(crab_ltx::QueryError::Operation(error)) => Err(error), - Err(crab_ltx::QueryError::Sqlite(error)) => { - self.fenced = true; - Err(error.into()) - } - Err(crab_ltx::QueryError::State(error)) => { - self.fenced = true; - Err(error.into()) - } - } - } - - pub(crate) fn prepare_hydration( - &self, - pages: u32, - ) -> Result> { - if self.fenced { - return Err(Error::Fenced); - } - if !(1..=64).contains(&pages) { - return Err(Error::Capacity("hydration pages")); - } - self.db.prepare_hydration(pages).map_err(Into::into) - } - - pub(crate) fn install_hydration( - &mut self, - batch: crab_ltx::db::HydrationBatch, - ) -> Result { - if self.fenced { - return Err(Error::Fenced); - } - match self.db.install_hydration(batch) { - Ok(hydration) => Ok(hydration), - Err(error) => { - self.fenced = true; - Err(error.into()) - } - } - } - - pub(crate) fn hydration(&self) -> Result> { - self.db.hydration().map_err(Into::into) - } - - /// Reads the durable work classes that can block safe owner release. - pub(crate) fn persisted_work_inventory( - &mut self, - role: CatalogRole, - ) -> Result { - if self.fenced { - return Err(Error::Fenced); - } - let result = self.db.query_with(|connection| { - crate::primitives::maintenance::inspect_persisted_work(connection, role) - }); - if let Some(error) = self.db.take_io_error() { - self.fenced = true; - return Err(ltx_error(error)); - } - match result { - Ok(inventory) => Ok(inventory), - Err(crab_ltx::QueryError::Operation(error)) => Err(error), - Err(crab_ltx::QueryError::Sqlite(error)) => { - self.fenced = true; - Err(error.into()) - } - Err(crab_ltx::QueryError::State(error)) => { - self.fenced = true; - Err(error.into()) - } - } - } - - pub(crate) fn transfer_work_inventory( - &mut self, - role: CatalogRole, - now_ms: i64, - ) -> Result { - if self.fenced { - return Err(Error::Fenced); - } - let result = self.db.query_with(|connection| { - crate::primitives::maintenance::inspect_transfer_work(connection, role, now_ms) - }); - if let Some(error) = self.db.take_io_error() { - self.fenced = true; - return Err(ltx_error(error)); - } - match result { - Ok(inventory) => Ok(inventory), - Err(crab_ltx::QueryError::Operation(error)) => Err(error), - Err(crab_ltx::QueryError::Sqlite(error)) => { - self.fenced = true; - Err(error.into()) - } - Err(crab_ltx::QueryError::State(error)) => { - self.fenced = true; - Err(error.into()) - } - } - } - - /// Resolves one identity against the actor-serialized logical SQLite state. - pub fn resolve( - &mut self, - identity: MutationIdentity, - operation_digest: Digest, - now_ms: i64, - max_result_bytes: usize, - ) -> Result { - if self.fenced { - return Ok(Resolution::Unknown); - } - if !self.logical_head_is_durable() { - return Ok(Resolution::Unknown); - } - if identity.expired(now_ms)? { - return Ok(Resolution::Expired); - } - if max_result_bytes > MAX_RESULT_BYTES { - return Err(Error::Command("result exceeds wire limit")); - } - let result = self.db.query_with(|connection| { - connection - .query_row( - "SELECT operation_digest, outcome, result, commit_sequence FROM sys_requests WHERE request_id = ?1", - [identity.request_id.as_bytes().as_slice()], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, Vec>(2)?, - row.get::<_, i64>(3)?, - )) - }, - ) - .optional() - }); - if let Some(error) = self.db.take_io_error() { - self.fenced = true; - return Err(ltx_error(error)); - } - let existing = match result { - Ok(existing) => existing, - Err(crab_ltx::QueryError::Operation(error)) => return Err(error.into()), - Err(crab_ltx::QueryError::Sqlite(error)) => { - self.fenced = true; - return Err(error.into()); - } - Err(crab_ltx::QueryError::State(error)) => { - self.fenced = true; - return Err(error.into()); - } - }; - let Some((digest, outcome, result, sequence)) = existing else { - return Ok(Resolution::Absent); - }; - if digest.as_slice() != operation_digest.as_bytes() { - return Err(Error::RequestConflict); - } - let outcome = stored_outcome(outcome, result, sequence)?; - if outcome.result().len() > max_result_bytes { - return Err(Error::Command("stored result exceeds command limit")); - } - Ok(Resolution::Committed(outcome)) - } - - /// Resolves one destination inbox identity from the logical SQLite state. - pub fn resolve_effect( - &mut self, - delivery: crate::primitives::effects::InboxDelivery, - now_ms: i64, - max_result_bytes: usize, - ) -> Result { - if self.fenced || !self.logical_head_is_durable() { - return Ok(Resolution::Unknown); - } - let result = self.db.query_with(|connection| { - crate::primitives::effects::inbox_resolve( - connection, - now_ms, - delivery, - max_result_bytes, - ) - }); - if let Some(error) = self.db.take_io_error() { - self.fenced = true; - return Err(ltx_error(error)); - } - match result { - Ok(resolution) => Ok(resolution), - Err(crab_ltx::QueryError::Operation(error)) => Err(error), - Err(crab_ltx::QueryError::Sqlite(error)) => { - self.fenced = true; - Err(error.into()) - } - Err(crab_ltx::QueryError::State(error)) => { - self.fenced = true; - Err(error.into()) - } - } - } - - /// Returns the oldest commit awaiting publication. - #[must_use] - pub fn pending(&self) -> Option<&PendingCommit> { - self.pending.front() - } - - #[must_use] - pub(crate) fn latest_pending(&self) -> Option<&PendingCommit> { - self.pending.back() - } - - /// Marks one actor-ordered logical commit safe to observe before object publication. - pub(crate) fn confirm_durable(&mut self, commit_sequence: u64) -> Result<()> { - if commit_sequence <= self.published_sequence { - return Ok(()); - } - let index = self - .pending - .iter() - .position(|pending| pending.outcome.commit_sequence() == commit_sequence) - .ok_or(Error::PendingPublication)?; - if self - .pending - .iter() - .take(index) - .any(|pending| !pending.durable) - { - return Err(Error::Control("durability proof skipped an earlier commit")); - } - let pending = self - .pending - .get_mut(index) - .ok_or(Error::PendingPublication)?; - pending.durable = true; - Ok(()) - } - - /// Returns the schema migration awaiting publication. - #[must_use] - pub fn pending_migration(&self) -> Option<&PendingMigration> { - self.pending_migration.as_ref() - } - - /// Applies one trusted registry migration and retains its captured cut. - pub fn migrate(&mut self, plan: crate::registry::MigrationPlan, now_ms: i64) -> Result<()> { - if self.fenced { - return Err(Error::Fenced); - } - if self.has_pending() { - return Err(Error::PendingPublication); - } - let valid_step = match (plan.sql(), plan.digest()) { - (Some(sql), Some(digest)) => { - self.schema.checked_add(1) == Some(plan.to_schema()) - && digest == Digest::from_bytes(*blake3::hash(sql.as_bytes()).as_bytes()) - } - (None, None) => self.schema == plan.to_schema() && plan.from_code() != plan.to_code(), - _ => false, - }; - if now_ms < 0 || plan.from_schema() != self.schema || !valid_step { - return Err(Error::Registry("invalid Cell migration plan")); - } - let cell = self.cell; - let incarnation = self.incarnation; - let from_schema = self.schema; - let to_schema = plan.to_schema(); - let digest = plan.digest(); - let transaction = self.db.transaction_with(|transaction| { - let (commit_sequence, prior_logical_time_ms) = - runtime_metadata(transaction, cell, incarnation, from_schema)?; - let sequence = commit_sequence - .checked_add(1) - .filter(|value| *value > 0) - .ok_or(Error::Command("commit sequence overflow"))?; - let logical_time_ms = now_ms.max(prior_logical_time_ms); - if let (Some(sql), Some(digest)) = (plan.sql(), digest) { - let existing = transaction - .query_row( - "SELECT digest, applied_sequence FROM sys_migrations WHERE version = ?1", - [to_schema], - |row| Ok((row.get::<_, Vec>(0)?, row.get::<_, i64>(1)?)), - ) - .optional()?; - if let Some((existing_digest, _)) = existing { - return Err(if existing_digest.as_slice() == digest.as_bytes() { - Error::Control("migration is recorded ahead of runtime schema") - } else { - Error::Registry("migration digest conflicts with SQLite history") - }); - } - transaction.execute_batch(sql)?; - transaction.execute( - "INSERT INTO sys_migrations(version, digest, applied_sequence) VALUES (?1, ?2, ?3)", - (to_schema, digest.as_bytes().as_slice(), sequence), - )?; - } - if transaction.execute( - "UPDATE sys_meta SET commit_sequence = ?1, logical_time_ms = ?2, schema_version = ?3 WHERE singleton = 1 AND schema_version = ?4", - (sequence, logical_time_ms, to_schema, from_schema), - )? != 1 - { - return Err(Error::Control("migration metadata changed during execution")); - } - let metadata = runtime_metadata(transaction, cell, incarnation, to_schema)?; - if metadata != (sequence, logical_time_ms) { - return Err(Error::Control("migration metadata did not validate")); - } - crate::primitives::capacity::validate(transaction)?; - let next_due_ms = crate::fleet::scheduler::scheduler_next_due_ms(transaction, logical_time_ms)?; - Ok((sequence, next_due_ms)) - }); - if let Some(error) = self.db.take_io_error() { - self.fenced = true; - return Err(ltx_error(error)); - } - let (sequence, next_due_ms) = match transaction { - Ok(value) => value, - Err(TransactionError::Operation(error)) => return Err(error), - Err(TransactionError::Admission(error)) => return Err(admission_error(error)), - Err(error) => { - self.fenced = true; - return Err(transaction_error(error)); - } - }; - // Migration output stays gated on follower or object durability. Keep - // the local cut readable for publication without serially fsyncing it. - let cuts = match self.db.capture_deferred() { - Ok(cuts) if !cuts.segments.is_empty() => cuts, - Ok(_) => { - self.fenced = true; - return Err(Error::Control("migration produced no LTX cut")); - } - Err(error) => { - self.fenced = true; - return Err(error.into()); - } - }; - let commit_sequence = - u64::try_from(sequence).map_err(|_| Error::Command("migration sequence overflow"))?; - self.schema = to_schema; - self.pending_migration = Some(PendingMigration { - code: plan.to_code(), - from_schema, - to_schema, - digest, - commit_sequence, - next_due_ms, - cuts, - prepared: None, - }); - Ok(()) - } - - /// Pins the one immutable proposal that may satisfy the pending commit. - pub fn bind_prepared(&mut self, prepared: &crab_ltx::PreparedRoot) -> Result<()> { - let pending = self.pending.front_mut().ok_or(Error::PendingPublication)?; - let root = prepared.root(); - if root.cell != *self.cell.as_bytes() - || root.incarnation != *self.incarnation.as_bytes() - || root.position != pending.cuts.position - || root.commit_sequence != pending.outcome.commit_sequence() - || prepared.verified().schema() != self.schema - || pending.prepared.is_some_and(|existing| existing != root) - { - return Err(Error::Command( - "prepared root does not match pending commit", - )); - } - pending.prepared = Some(root); - Ok(()) - } - - /// Pins the immutable proposal for the pending schema migration. - pub fn bind_migration_prepared(&mut self, prepared: &crab_ltx::PreparedRoot) -> Result<()> { - let pending = self - .pending_migration - .as_mut() - .ok_or(Error::PendingPublication)?; - let root = prepared.root(); - if root.cell != *self.cell.as_bytes() - || root.incarnation != *self.incarnation.as_bytes() - || root.position != pending.cuts.position - || root.commit_sequence != pending.commit_sequence - || prepared.verified().schema() != pending.to_schema - || pending.prepared.is_some_and(|existing| existing != root) - { - return Err(Error::Command( - "prepared root does not match pending migration", - )); - } - pending.prepared = Some(root); - Ok(()) - } - - /// Releases the stored result only after the published root proves inclusion. - pub fn confirm_published(&mut self, root: &crab_ltx::RootRef) -> Result { - let pending = self.pending.front().ok_or(Error::PendingPublication)?; - if pending.prepared.as_ref() != Some(root) { - return Err(Error::Command( - "published root does not match prepared commit", - )); - } - self.db.prune_captured(&pending.cuts)?; - let pending = self.pending.pop_front().ok_or(Error::PendingPublication)?; - self.pending_bytes = self - .pending_bytes - .checked_sub(pending.retained_bytes()) - .ok_or(Error::Control("pending publication accounting underflow"))?; - self.published_sequence = root.commit_sequence; - Ok(pending.outcome) - } - - /// Finalizes local migration state after its exact schema-bearing root publishes. - pub fn confirm_migration_published( - &mut self, - root: &crab_ltx::RootRef, - ) -> Result { - let pending = self - .pending_migration - .as_ref() - .ok_or(Error::PendingPublication)?; - if pending.prepared.as_ref() != Some(root) { - return Err(Error::Command( - "published root does not match prepared migration", - )); - } - self.db.prune_captured(&pending.cuts)?; - self.pending_migration - .take() - .map(|pending| MigrationOutcome { - code: pending.code, - schema: pending.to_schema, - commit_sequence: pending.commit_sequence, - }) - .ok_or(Error::PendingPublication) - } - - pub(crate) fn confirm_bootstrap_published( - &mut self, - cuts: &crab_ltx::CaptureBatch, - ) -> Result<()> { - if self.has_pending() || self.fenced || cuts.segments.is_empty() { - return Err(Error::PendingPublication); - } - self.db.prune_captured(cuts)?; - Ok(()) - } - - /// Closes a drained executor; pending or fenced state requires recovery. - pub fn close(self) -> Result<()> { - if self.has_pending() || self.fenced { - return Err(Error::Fenced); - } - self.db.close()?; - Ok(()) - } - - /// Closes a drained executor and leaves a local record of what it holds. - /// - /// The record only accelerates the next same-node activation of this exact - /// root: a database that cannot produce one is still closed cleanly, because - /// the object store, not the local file, is the authority. - pub(crate) fn close_resumable(self, root: crate::control::RootRef, code: Digest) -> Result<()> { - if self.has_pending() || self.fenced { - return Err(Error::Fenced); - } - let database = self.db.path().to_owned(); - if let Err(error) = self.db.persist_continuation() { - tracing::debug!(error = %error, "Cell capture continuation was not recorded"); - } else { - match crate::cell::resume::ResumeRecord::new( - self.cell, - self.incarnation, - self.schema, - code, - root, - &database, - ) { - Ok(record) => crate::cell::resume::write(&record, &database), - Err(error) => tracing::debug!(error = %error, "Cell resume record was refused"), - } - } - self.db.close()?; - Ok(()) - } - - /// Closes a fenced executor without treating local pending state as authority. - /// - /// The caller must recover only from the authoritative immutable root. Local - /// database and LTX artifacts remain quarantined for diagnosis. - pub(crate) fn discard(self) -> Result<()> { - self.db.close()?; - Ok(()) - } - - pub(crate) fn fence(&mut self) { - self.fenced = true; - } - - pub(crate) fn drained(&self) -> bool { - !self.has_pending() && !self.fenced - } - - pub(crate) fn worker_state(&self) -> crate::cell::worker::WorkerState { - if self.fenced { - crate::cell::worker::WorkerState::Fenced - } else if self.has_pending() { - crate::cell::worker::WorkerState::Pending - } else { - crate::cell::worker::WorkerState::Ready - } - } - - fn has_pending(&self) -> bool { - !self.pending.is_empty() || self.pending_migration.is_some() - } - - fn accepts_publication(&self) -> bool { - self.pending_migration.is_none() - && self.logical_head_is_durable() - && self.pending.len() < MAX_PENDING_PUBLICATIONS - && self.pending_bytes < PENDING_PUBLICATION_HIGH_WATER_BYTES - } - - fn logical_head_is_durable(&self) -> bool { - self.pending.back().is_none_or(|pending| pending.durable) - } - - fn finish_transaction( - &mut self, - transaction: std::result::Result>, - ) -> Result { - if let Some(error) = self.db.take_io_error() { - self.fenced = true; - return Err(ltx_error(error)); - } - let transaction = match transaction { - Ok(value) => value, - Err(TransactionError::Operation(error)) => return Err(error), - Err(TransactionError::Admission(error)) => return Err(admission_error(error)), - Err(error) => { - self.fenced = true; - return Err(transaction_error(error)); - } - }; - match transaction { - TransactionResult::Recorded(outcome) => Ok(CommandExecution::Recorded(outcome)), - TransactionResult::Committed { - outcome, - logical_time_ms, - next_due_ms, - } => { - // The actor cannot expose this commit until the same cut is - // durable on followers or behind the authoritative root CAS. - let cuts = match self.db.capture_deferred() { - Ok(cuts) => cuts, - Err(error) => { - self.fenced = true; - return Err(error.into()); - } - }; - let pending = PendingCommit { - outcome, - logical_time_ms, - next_due_ms, - cuts, - prepared: None, - durable: false, - }; - self.pending_bytes = self - .pending_bytes - .checked_add(pending.retained_bytes()) - .ok_or(Error::Capacity("pending publication bytes"))?; - self.pending.push_back(pending); - Ok(CommandExecution::Pending) - } - } - } -} - -fn runtime_metadata( - transaction: &crab_ltx::rusqlite::Transaction<'_>, - cell: CellId, - incarnation: IncarnationId, - schema: u32, -) -> Result<(i64, i64)> { - let meta = transaction.query_row( - "SELECT cell_id, incarnation, commit_sequence, logical_time_ms, schema_version FROM sys_meta WHERE singleton = 1", - [], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, i64>(3)?, - row.get::<_, u32>(4)?, - )) - }, - )?; - if meta.0.as_slice() != cell.as_bytes() - || meta.1.as_slice() != incarnation.as_bytes() - || meta.2 < 0 - || meta.3 < 0 - || meta.4 != schema - { - return Err(Error::Command("runtime schema identity mismatch")); - } - Ok((meta.2, meta.3)) -} - -fn transaction_error(error: TransactionError) -> Error { - match error { - TransactionError::Admission(error) => admission_error(error), - TransactionError::Operation(error) => error, - TransactionError::Sqlite(error) => error.into(), - TransactionError::Capture(error) => error.into(), - } -} - -fn admission_error(error: crab_ltx::CrabError) -> Error { - // A declared bound refused the work before any side effect, so the caller - // sees a capacity refusal instead of a fence or an unknown outcome. - match error.classify() { - crab_ltx::FailureClass::Capacity => ltx_capacity_error(&error), - _ => error.into(), - } -} - -fn ltx_capacity_error(error: &crab_ltx::CrabError) -> Error { - match error { - crab_ltx::CrabError::Limit(kind) => Error::Capacity(kind.as_str()), - _ => Error::Capacity("local storage"), - } -} - -fn transaction_error_with_io(db: &Db, error: TransactionError) -> Error { - db.take_io_error() - .map_or_else(|| transaction_error(error), ltx_error) -} - -fn ltx_error(error: crab_ltx::CrabError) -> Error { - match error.classify() { - crab_ltx::FailureClass::Capacity => ltx_capacity_error(&error), - crab_ltx::FailureClass::Fenced => Error::Fenced, - _ => match error { - crab_ltx::CrabError::Deadline => Error::Deadline, - error => Error::Ltx(error), - }, - } -} - -enum TransactionResult { - Recorded(StoredOutcome), - Committed { - outcome: StoredOutcome, - logical_time_ms: i64, - next_due_ms: Option, - }, -} - -fn stored_outcome(outcome: i64, result: Vec, sequence: i64) -> Result { - if result.len() > MAX_RESULT_BYTES || sequence <= 0 { - return Err(Error::Command("invalid stored request outcome")); - } - let commit_sequence = - u64::try_from(sequence).map_err(|_| Error::Command("invalid stored commit sequence"))?; - match outcome { - 1 => Ok(StoredOutcome::Success { - result, - commit_sequence, - }), - 2 => Ok(StoredOutcome::Rejected { - result, - commit_sequence, - }), - _ => Err(Error::Command("invalid stored request outcome")), - } -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/cell/executor/tests.rs b/crates/crab-cell-runtime/src/cell/executor/tests.rs deleted file mode 100644 index bcba4b354..000000000 --- a/crates/crab-cell-runtime/src/cell/executor/tests.rs +++ /dev/null @@ -1,197 +0,0 @@ -use std::sync::OnceLock; - -use super::*; - -const CODE_ONLY_MODULE: &str = "code-only-test"; -const CODE_ONLY_NAMESPACE: crate::NamespaceId = crate::NamespaceId::from_bytes([45; 16]); -const PREDECESSOR_CODE: Digest = Digest::from_bytes([43; 32]); - -struct CodeOnlyModule; - -impl crate::CellModule for CodeOnlyModule { - const NAME: &'static str = CODE_ONLY_MODULE; - - fn descriptor(&self) -> &'static crate::ModuleDescriptor { - static DESCRIPTOR: OnceLock = OnceLock::new(); - DESCRIPTOR.get_or_init(|| { - let sql = "SELECT 1"; - crate::ModuleDescriptor { - name: CODE_ONLY_MODULE, - source_digest: Digest::from_bytes([46; 32]), - retained_codes: &[crate::registry::RetainedCodeDescriptor { - code: PREDECESSOR_CODE, - schema_min: 1, - schema_max: 1, - }], - schema_min: 1, - schema_max: 1, - migrations: Box::leak(Box::new([crate::registry::MigrationDescriptor { - version: 1, - sql, - digest: Digest::from_bytes(*blake3::hash(sql.as_bytes()).as_bytes()), - }])), - commands: &[], - queries: &[], - workflow_definitions: &[], - activity_types: &[], - namespaces: &[crate::NamespaceDescriptor { - id: CODE_ONLY_NAMESPACE, - name: CODE_ONLY_MODULE, - role: crate::CatalogRole::Sql, - shards: 1, - effect_targets: &[], - dead_letter: None, - }], - } - }) - } - - fn register(self, _registry: &mut crate::RegistryBuilder) -> Result<()> { - Ok(()) - } -} - -fn code_only_registry() -> crate::Registry { - let mut registry = crate::RegistryBuilder::new(crate::BuildDescriptor { - source_revision: CODE_ONLY_MODULE.into(), - cargo_lock_digest: Digest::from_bytes([47; 32]), - }); - registry.register(CodeOnlyModule).unwrap(); - registry.finish().unwrap() -} - -#[test] -fn full_pending_publication_budget_refuses_new_commands() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("cell.sqlite"); - let cell = CellId::from_bytes([61; 32]); - let incarnation = IncarnationId::from_bytes([62; 16]); - let mut connection = crab_ltx::rusqlite::Connection::open(&path).unwrap(); - crate::cell::schema::install_runtime_schema(&mut connection, cell, incarnation, 1).unwrap(); - drop(connection); - let db = Db::open(&path, crab_ltx::Limits::default()).unwrap(); - let mut executor = CellExecutor::new(db, cell, incarnation, 1); - - // Each command commits locally, becomes pending, and is released by its - // fleet proof, so object publication may lag behind the actor. The budget - // that stops that lag from growing without bound is the admission under - // test. - for sequence in 1_u64..=MAX_PENDING_PUBLICATIONS as u64 { - let identity = MutationIdentity { - request_id: RequestId::from_bytes([sequence as u8; 16]), - issued_at_ms: 10, - expires_at_ms: 10_000, - }; - let execution = executor - .execute( - identity, - Digest::from_bytes([sequence as u8; 32]), - 20, - 1 << 20, - |transaction| { - transaction.execute( - "UPDATE sys_meta SET logical_time_ms = logical_time_ms + 1", - [], - )?; - Ok(HandlerOutcome::Success(vec![sequence as u8])) - }, - ) - .unwrap(); - assert!( - matches!(execution, CommandExecution::Pending), - "sequence {sequence}" - ); - executor.confirm_durable(sequence).unwrap(); - } - - // The budget is full: the next command is refused instead of queueing more - // unpublished work behind a provider that is already behind. - let identity = MutationIdentity { - request_id: RequestId::from_bytes([201; 16]), - issued_at_ms: 10, - expires_at_ms: 10_000, - }; - assert!(matches!( - executor.execute(identity, Digest::from_bytes([9; 32]), 20, 1 << 20, |_| { - Ok(HandlerOutcome::Success(Vec::new())) - }), - Err(Error::PendingPublication) - )); -} - -#[test] -fn restored_executor_rejects_root_sequence_ahead_of_sqlite_metadata() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("cell.sqlite"); - let cell = CellId::from_bytes([31; 32]); - let incarnation = IncarnationId::from_bytes([32; 16]); - let mut connection = crab_ltx::rusqlite::Connection::open(&path).unwrap(); - crate::cell::schema::install_runtime_schema(&mut connection, cell, incarnation, 1).unwrap(); - drop(connection); - let mut db = Db::open(&path, crab_ltx::Limits::default()).unwrap(); - db.transaction(|transaction| { - transaction.execute("UPDATE sys_meta SET logical_time_ms = 1", [])?; - Ok(()) - }) - .unwrap(); - db.capture().unwrap(); - let root = crab_ltx::RootRef { - cell: *cell.as_bytes(), - incarnation: *incarnation.as_bytes(), - digest: [33; 32], - position: db.position(), - commit_sequence: 1, - }; - assert!(matches!( - CellExecutor::from_restored(db, cell, incarnation, 1, root), - Err(Error::Control( - "restored SQLite metadata does not match authoritative root" - )) - )); -} - -#[test] -fn code_only_migration_commits_a_captured_system_cut() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("cell.sqlite"); - let cell = CellId::from_bytes([41; 32]); - let incarnation = IncarnationId::from_bytes([42; 16]); - let mut connection = crab_ltx::rusqlite::Connection::open(&path).unwrap(); - crate::cell::schema::install_runtime_schema(&mut connection, cell, incarnation, 1).unwrap(); - drop(connection); - let db = Db::open(&path, crab_ltx::Limits::default()).unwrap(); - let mut executor = CellExecutor::new(db, cell, incarnation, 1); - let registry = code_only_registry(); - let target_code = registry.module_code(CODE_ONLY_MODULE).unwrap(); - let plan = registry - .next_migration(CODE_ONLY_NAMESPACE, PREDECESSOR_CODE, 1) - .unwrap() - .unwrap(); - executor.migrate(plan, 10).unwrap(); - let pending = executor.pending_migration().unwrap(); - assert_eq!(pending.code(), target_code); - assert_eq!(pending.from_schema(), 1); - assert_eq!(pending.to_schema(), 1); - assert_eq!(pending.digest(), None); - assert_eq!(pending.commit_sequence(), 1); - assert!(!pending.cuts().segments.is_empty()); -} - -#[test] -fn declared_admission_limits_become_capacity_errors() { - let disk = admission_error(crab_ltx::CrabError::Limit( - crab_ltx::LimitKind::LocalDiskBytes, - )); - assert!(matches!(disk, Error::Capacity("local disk bytes"))); - - let database = crab_ltx::CrabError::Limit(crab_ltx::LimitKind::DatabaseBytes); - assert!(matches!( - admission_error(database), - Error::Capacity("database bytes") - )); - - // A fenced session is not a capacity refusal: the caller must restore - // authoritative state instead of retrying the same request. - let fenced = admission_error(crab_ltx::CrabError::Fenced); - assert!(matches!(fenced, Error::Ltx(crab_ltx::CrabError::Fenced))); -} diff --git a/crates/crab-cell-runtime/src/cell/resume.rs b/crates/crab-cell-runtime/src/cell/resume.rs deleted file mode 100644 index 4894f714e..000000000 --- a/crates/crab-cell-runtime/src/cell/resume.rs +++ /dev/null @@ -1,221 +0,0 @@ -//! Local resume records: the same-node wake that skips a restore. -//! -//! One record is written beside a database that a clean release left behind, -//! and it names exactly the published root that database still holds. It is an -//! accelerator, never authority: an activation still re-verifies the database -//! against the observed control before the Cell serves anything, and any -//! mismatch discards the local image and restores the exact root instead. Every -//! failure below therefore degrades one wake to a restore and never fails it. - -use std::io; -use std::path::{Path, PathBuf}; - -use crate::control::Control; -use crate::identity::{CellId, Digest, IncarnationId}; -use crate::{Error, Result}; - -const RECORD_MAGIC: [u8; 8] = *b"CRABRES1"; -const RECORD_VERSION: u32 = 1; -const RECORD_SUFFIX: &str = ".resume"; -const RECORD_HEADER_BYTES: usize = 154; -const MAX_DATABASE_NAME_BYTES: usize = 255; - -/// Identity of the local database one clean release left resumable. -pub(crate) struct ResumeRecord { - cell: CellId, - incarnation: IncarnationId, - schema: u32, - code: Digest, - root: crate::control::RootRef, - database: String, -} - -impl ResumeRecord { - pub(crate) fn new( - cell: CellId, - incarnation: IncarnationId, - schema: u32, - code: Digest, - root: crate::control::RootRef, - database: &Path, - ) -> Result { - let name = database - .file_name() - .and_then(|name| name.to_str()) - .ok_or(Error::Control( - "resume database path has no UTF-8 file name", - ))?; - if name.len() > MAX_DATABASE_NAME_BYTES { - return Err(Error::Control("resume database file name is too long")); - } - Ok(Self { - cell, - incarnation, - schema, - code, - root, - database: name.to_owned(), - }) - } - - fn encode(&self) -> Vec { - let mut bytes = Vec::with_capacity(RECORD_HEADER_BYTES + self.database.len()); - bytes.extend_from_slice(&RECORD_MAGIC); - bytes.extend_from_slice(&RECORD_VERSION.to_be_bytes()); - bytes.extend_from_slice(&self.schema.to_be_bytes()); - bytes.extend_from_slice(self.code.as_bytes()); - bytes.extend_from_slice(self.cell.as_bytes()); - bytes.extend_from_slice(self.incarnation.as_bytes()); - bytes.extend_from_slice(self.root.digest.as_bytes()); - bytes.extend_from_slice(&self.root.txid.to_be_bytes()); - bytes.extend_from_slice(&self.root.checksum.to_be_bytes()); - bytes.extend_from_slice(&self.root.commit_sequence.to_be_bytes()); - bytes.extend_from_slice(&(self.database.len() as u16).to_be_bytes()); - bytes.extend_from_slice(self.database.as_bytes()); - bytes - } - - fn decode(bytes: &[u8]) -> Result { - if bytes.len() < RECORD_HEADER_BYTES || bytes[..8] != RECORD_MAGIC { - return Err(Error::Control("unrecognized Cell resume record")); - } - let version = u32::from_be_bytes(array(bytes, 8)?); - if version != RECORD_VERSION { - return Err(Error::Control("unsupported Cell resume record version")); - } - let name_len = usize::from(u16::from_be_bytes(array(bytes, 152)?)); - if name_len == 0 - || name_len > MAX_DATABASE_NAME_BYTES - || bytes.len() != RECORD_HEADER_BYTES + name_len - { - return Err(Error::Control("Cell resume record length")); - } - let database = std::str::from_utf8(&bytes[RECORD_HEADER_BYTES..])?.to_owned(); - if database.contains(['/', '\\']) { - return Err(Error::Control("Cell resume record escapes its directory")); - } - Ok(Self { - schema: u32::from_be_bytes(array(bytes, 12)?), - code: Digest::from_bytes(array(bytes, 16)?), - cell: CellId::from_bytes(array(bytes, 48)?), - incarnation: IncarnationId::from_bytes(array(bytes, 80)?), - root: crate::control::RootRef { - digest: Digest::from_bytes(array(bytes, 96)?), - txid: u64::from_be_bytes(array(bytes, 128)?), - checksum: u64::from_be_bytes(array(bytes, 136)?), - commit_sequence: u64::from_be_bytes(array(bytes, 144)?), - }, - database, - }) - } - - /// Reports whether this record names exactly what the observed control names. - /// - /// The ownership epoch is deliberately absent: acquiring an idle Cell raises - /// the epoch without touching the root, and every writer that did commit - /// moves the root digest instead. - pub(crate) fn matches(&self, control: &Control) -> bool { - control.cell == self.cell - && control.incarnation == self.incarnation - && control.schema == self.schema - && control.code == self.code - && control.root.as_ref() == Some(&self.root) - } - - fn database_path(&self, directory: &Path) -> PathBuf { - directory.join(&self.database) - } -} - -fn array(bytes: &[u8], start: usize) -> Result<[u8; N]> { - bytes - .get(start..start + N) - .and_then(|slice| slice.try_into().ok()) - .ok_or(Error::Control("Cell resume record length")) -} - -/// Returns the resume record path that belongs to a database. -pub(crate) fn record_path(database: &Path) -> PathBuf { - let mut path = database.as_os_str().to_owned(); - path.push(RECORD_SUFFIX); - path.into() -} - -/// Writes the record for a database that a clean release just closed. -pub(crate) fn write(record: &ResumeRecord, database: &Path) { - let path = record_path(database); - if let Err(error) = std::fs::write(&path, record.encode()) { - tracing::debug!(error = %error, "Cell resume record was not written"); - } -} - -fn remove_file(path: &Path) { - if let Err(error) = std::fs::remove_file(path) - && error.kind() != io::ErrorKind::NotFound - { - tracing::debug!(error = %error, "Cell resume artifact was not removed"); - } -} - -/// Removes one record and everything it names. -pub(crate) fn discard(record: &Path, database: &Path, replica: &crab_ltx::CellReplica) { - if let Err(error) = replica.discard_resumed(database) { - tracing::debug!(error = %error, "Cell resume database was not discarded"); - } - remove_file(record); -} - -/// Consumes the record that still names the observed control's root. -/// -/// Every other record in the directory is discarded with the database it names, -/// so a rejected candidate cannot accumulate. The returned database keeps its -/// files: the caller either opens them or discards them. -pub(crate) fn take_matching( - destination: &Path, - control: &Control, - replica: &crab_ltx::CellReplica, -) -> Option { - let directory = destination.parent()?; - let entries = match std::fs::read_dir(directory) { - Ok(entries) => entries, - Err(error) if error.kind() == io::ErrorKind::NotFound => return None, - Err(error) => { - tracing::debug!(error = %error, "Cell resume directory was not read"); - return None; - } - }; - let mut matched = None; - for entry in entries.flatten() { - let path = entry.path(); - let is_record = path - .file_name() - .and_then(|name| name.to_str()) - .is_some_and(|name| name.ends_with(RECORD_SUFFIX)); - if !is_record { - continue; - } - let record = std::fs::read(&path) - .ok() - .and_then(|bytes| ResumeRecord::decode(&bytes).ok()); - let Some(record) = record else { - discard(&path, &database_of_record(&path), replica); - continue; - }; - let database = record.database_path(directory); - if record.matches(control) && matched.is_none() { - remove_file(&path); - matched = Some(database); - continue; - } - discard(&path, &database, replica); - } - matched -} - -fn database_of_record(record: &Path) -> PathBuf { - let name = record - .file_name() - .and_then(|name| name.to_str()) - .unwrap_or_default(); - record.with_file_name(name.strip_suffix(RECORD_SUFFIX).unwrap_or_default()) -} diff --git a/crates/crab-cell-runtime/src/cell/schema.rs b/crates/crab-cell-runtime/src/cell/schema.rs deleted file mode 100644 index 041ab89cc..000000000 --- a/crates/crab-cell-runtime/src/cell/schema.rs +++ /dev/null @@ -1,111 +0,0 @@ -//! Runtime schema installation and version. -use rusqlite::{Connection, OptionalExtension}; - -use crate::identity::CellId; -use crate::identity::IncarnationId; -use crate::{Error, Result}; - -const RUNTIME_SCHEMA: &str = include_str!("../migrations/runtime.sql"); - -/// Installs runtime schema version one in a new SQLite Cell. -/// -/// The connection must not contain an existing runtime schema. Installation and -/// initial metadata insertion commit atomically; callers capture and publish the -/// resulting SQLite transaction through `crab-ltx` before exposing the Cell. -pub fn install_runtime_schema( - connection: &mut Connection, - cell: CellId, - incarnation: IncarnationId, - schema_version: u32, -) -> Result<()> { - if schema_version == 0 { - return Err(Error::Control("invalid initial schema version")); - } - connection.execute_batch("PRAGMA foreign_keys = ON")?; - let transaction = - connection.transaction_with_behavior(rusqlite::TransactionBehavior::Immediate)?; - install_runtime_schema_in(&transaction, cell, incarnation, schema_version)?; - transaction.commit()?; - Ok(()) -} - -pub(crate) fn install_runtime_schema_in( - transaction: &rusqlite::Transaction<'_>, - cell: CellId, - incarnation: IncarnationId, - schema_version: u32, -) -> Result<()> { - if schema_version == 0 { - return Err(Error::Control("invalid initial schema version")); - } - transaction.execute_batch(RUNTIME_SCHEMA)?; - transaction.execute( - "INSERT INTO sys_meta(singleton, cell_id, incarnation, commit_sequence, logical_time_ms, schema_version) VALUES (1, ?1, ?2, 0, 0, ?3)", - (cell.as_bytes().as_slice(), incarnation.as_bytes().as_slice(), schema_version), - )?; - Ok(()) -} - -/// Verifies the persisted identity/schema before an existing Cell is admitted. -pub fn verify_runtime_schema( - connection: &Connection, - cell: CellId, - incarnation: IncarnationId, - schema_version: u32, -) -> Result<()> { - let metadata = connection - .query_row( - "SELECT cell_id, incarnation, schema_version FROM sys_meta WHERE singleton = 1", - [], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, u32>(2)?, - )) - }, - ) - .optional()?; - match metadata { - Some((stored_cell, stored_incarnation, stored_schema)) - if stored_cell == cell.as_bytes() - && stored_incarnation == incarnation.as_bytes() - && stored_schema == schema_version => - { - Ok(()) - } - Some(_) => Err(Error::Control("SQLite metadata does not match control")), - None => Err(Error::Control("SQLite runtime metadata is missing")), - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn schema_installation_is_atomic_and_binds_cell_identity() { - let mut connection = Connection::open_in_memory().unwrap(); - let cell = CellId::from_bytes([1; 32]); - let incarnation = IncarnationId::from_bytes([2; 16]); - install_runtime_schema(&mut connection, cell, incarnation, 3).unwrap(); - verify_runtime_schema(&connection, cell, incarnation, 3).unwrap(); - assert!( - verify_runtime_schema(&connection, CellId::from_bytes([9; 32]), incarnation, 3) - .is_err() - ); - assert!(install_runtime_schema(&mut connection, cell, incarnation, 3).is_err()); - let integrity: String = connection - .query_row("PRAGMA integrity_check", [], |row| row.get(0)) - .unwrap(); - assert_eq!(integrity, "ok"); - } - - #[test] - fn embedded_runtime_migration_matches_normative_contract() { - assert_eq!( - RUNTIME_SCHEMA, - include_str!("../../docs/contracts/runtime.sql") - ); - } -} diff --git a/crates/crab-cell-runtime/src/cell/worker.rs b/crates/crab-cell-runtime/src/cell/worker.rs deleted file mode 100644 index 8afa99028..000000000 --- a/crates/crab-cell-runtime/src/cell/worker.rs +++ /dev/null @@ -1,1205 +0,0 @@ -//! Bounded SQL worker pool and its per-shard worker loop. -use std::{ - collections::HashMap, - panic::{AssertUnwindSafe, catch_unwind}, - path::PathBuf, - sync::{ - Arc, Mutex, - atomic::{AtomicU8, Ordering}, - }, - thread::JoinHandle, - time::{Duration, Instant}, -}; - -use tokio::sync::{OwnedSemaphorePermit, Semaphore, mpsc, oneshot}; - -use crate::cell::catalog::CatalogRole; -use crate::cell::executor::{ - CellExecutor, CommandExecution, HandlerOutcome, MigrationOutcome, PendingCommit, - PendingMigration, StoredOutcome, -}; -use crate::cell::executor::{MutationIdentity, Resolution}; -use crate::fleet::resource::ResourceCost; -use crate::fleet::resource::{ - ACTIVE_CELL_NATIVE_BYTES, HYDRATION_JOB_CAPACITY, ResourceLedger, ResourceReservation, -}; -use crate::identity::{CellId, Digest}; -use crate::primitives::effects::InboxDelivery; -use crate::primitives::maintenance::PersistedWorkInventory; -use crate::primitives::maintenance::TransferWorkInventory; -use crate::registry::MigrationPlan; -use crate::{Error, Result}; - -mod run; - -const MAX_WORKERS: usize = 16; -const MAX_ACTIVE_CELLS: usize = 10_000; -const WORKER_QUEUE: usize = 256; -const DEFAULT_PAGE_IO_DEADLINE: Duration = Duration::from_secs(30); - -const SQL_QUEUED: u8 = 0; -const SQL_STARTED: u8 = 1; -const SQL_CANCELLED: u8 = 2; - -#[derive(Clone)] -pub(crate) struct SqlDeadline { - at: Instant, - state: Arc, -} - -impl SqlDeadline { - pub(crate) fn new(at: Instant) -> Self { - Self { - at, - state: Arc::new(AtomicU8::new(SQL_QUEUED)), - } - } - - pub(crate) fn at(&self) -> Instant { - self.at - } - - // Only the winner of queued -> started may touch SQLite. A timer can - // cancel queued work without fencing a Cell or permitting a late mutation. - pub(crate) fn cancel_queued(&self) -> bool { - self.state - .compare_exchange( - SQL_QUEUED, - SQL_CANCELLED, - Ordering::AcqRel, - Ordering::Acquire, - ) - .is_ok() - || self.cancelled() - } - - pub(crate) fn cancelled(&self) -> bool { - self.state.load(Ordering::Acquire) == SQL_CANCELLED - } - - fn start(&self) -> Result<()> { - if Instant::now() >= self.at { - self.cancel_queued(); - } - self.state - .compare_exchange(SQL_QUEUED, SQL_STARTED, Ordering::AcqRel, Ordering::Acquire) - .map(|_| ()) - .map_err(|_| Error::Deadline) - } -} - -/// Minimum SQLite page-cache reservation for one active Cell. -pub const ACTIVE_CELL_PAGE_CACHE_BYTES: u64 = - crab_ltx::MANAGED_SQLITE_CONNECTIONS * crab_ltx::MANAGED_CONNECTION_PAGE_CACHE_BYTES; - -pub use crate::fleet::resource::ACTIVE_CELL_FILE_DESCRIPTORS; - -pub(crate) type Handler = Box< - dyn for<'connection> FnOnce( - &crab_ltx::rusqlite::Transaction<'connection>, - ) -> Result - + Send - + 'static, ->; - -pub(crate) type Initializer = Box< - dyn for<'connection> FnOnce(&crab_ltx::rusqlite::Transaction<'connection>) -> Result<()> - + Send - + 'static, ->; - -pub(crate) type QueryHandler = - Box Result> + Send + 'static>; - -pub(crate) struct BootstrapExecution { - pub(crate) cuts: crab_ltx::CaptureBatch, - pub(crate) next_due_ms: Option, -} - -#[derive(Clone, Copy, PartialEq, Eq)] -pub(crate) enum WorkerState { - Ready, - Pending, - Fenced, -} - -#[derive(Debug)] -pub(crate) enum HydrationStep { - Progress(Option), - Deferred(Duration), -} - -/// Result returned by one SQL worker without releasing pending command output. -#[derive(Clone)] -pub enum WorkerExecution { - /// The command has a durable outcome. - Recorded(StoredOutcome), - /// The command committed locally and awaits publication. - Pending(Box), -} - -/// Fixed shard pool whose threads exclusively own all active SQLite handles. -#[derive(Clone)] -pub struct SqlWorkerPool { - inner: Arc, -} - -impl SqlWorkerPool { - /// Starts a fixed pool with bounded queues and active-Cell admission. - pub fn new(worker_count: usize, max_active_cells: usize) -> Result { - if worker_count == 0 - || worker_count > MAX_WORKERS - || max_active_cells == 0 - || max_active_cells > MAX_ACTIVE_CELLS - { - return Err(Error::Capacity("invalid SQL worker configuration")); - } - let resources = ResourceLedger::new( - ResourceCost::zero() - .with_active_cells(max_active_cells) - .with_resident_bytes(max_active_cells.saturating_mul(ACTIVE_CELL_NATIVE_BYTES)) - .with_file_descriptors( - max_active_cells.saturating_mul(ACTIVE_CELL_FILE_DESCRIPTORS), - ) - .with_worker_jobs(worker_count) - .with_primitive_jobs(worker_count) - .with_hydration_jobs(HYDRATION_JOB_CAPACITY), - ); - let mut workers = Vec::with_capacity(worker_count); - let mut threads: Vec> = Vec::with_capacity(worker_count); - for index in 0..worker_count { - let (sender, receiver) = mpsc::channel(WORKER_QUEUE); - let thread = match std::thread::Builder::new() - .name(format!("crab-cell-sql-{index}")) - .spawn(move || run::run_worker(receiver)) - { - Ok(thread) => thread, - Err(error) => { - drop(workers); - for thread in threads { - let _ = thread.join(); - } - return Err(Error::WorkerStart(Box::new(error))); - } - }; - workers.push(sender); - threads.push(thread); - } - Ok(Self { - inner: Arc::new(PoolInner { - lifecycle: Mutex::new(WorkerLifecycle { - workers, - threads, - closing: false, - }), - resources, - max_active_cells, - worker_count, - worker_permits: (0..worker_count) - .map(|_| Arc::new(Semaphore::new(1))) - .collect(), - }), - }) - } - - /// Sets the shared native-memory admission ceiling for writers and read snapshots. - /// - /// The default covers the configured writer count. Hosts admitting read - /// replicas must also budget their snapshots, including overlapping refreshes. - /// This is reserved native memory, not an RSS limit; retained cuts are separate. - /// Zero or a ceiling below existing reservations returns a capacity error. - pub fn with_native_memory_limit(self, bytes: usize) -> Result { - self.inner.resources.set_resident_limit(bytes)?; - Ok(self) - } - - /// Uses the design's CPU-derived worker count, capped at sixteen. - pub fn for_system(max_active_cells: usize) -> Result { - let workers = std::thread::available_parallelism() - .map(usize::from) - .unwrap_or(1) - .clamp(1, MAX_WORKERS); - Self::new(workers, max_active_cells) - } - - /// Moves a newly restored/opened Cell executor onto its sole worker. - pub async fn activate(&self, cell: CellId, executor: CellExecutor) -> Result<()> { - let reservation = self.reserve_activation()?; - let (reply, response) = oneshot::channel(); - self.send( - cell, - WorkerCommand::Activate { - cell, - executor: Box::new(executor), - reservation, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Creates and captures a new Cell on its assigned SQL worker. - pub(crate) async fn bootstrap( - &self, - cell: CellId, - replica: crab_ltx::CellReplica, - destination: PathBuf, - incarnation: crate::identity::IncarnationId, - schema: u32, - initialize: Initializer, - reservation: CellReservation, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send( - cell, - WorkerCommand::Bootstrap(Box::new(WorkerBootstrap { - cell, - replica, - destination, - incarnation, - schema, - initialize, - reservation, - reply, - })), - ) - .await?; - receive(response).await - } - - pub(crate) async fn confirm_bootstrap_published( - &self, - cell: CellId, - cuts: crab_ltx::CaptureBatch, - ) -> Result<()> { - let (reply, response) = oneshot::channel(); - self.send( - cell, - WorkerCommand::ConfirmBootstrapPublished { - cell, - cuts: Box::new(cuts), - reply, - }, - ) - .await?; - receive(response).await - } - - /// Opens and verifies one exact immutable root on its assigned SQL worker. - pub(crate) async fn activate_restored( - &self, - cell: CellId, - database: RestoredDatabase, - destination: PathBuf, - incarnation: crate::identity::IncarnationId, - schema: u32, - root: crab_ltx::RootRef, - reservation: CellReservation, - ) -> Result<()> { - let (reply, response) = oneshot::channel(); - self.send( - cell, - WorkerCommand::ActivateRestored { - cell, - database: Box::new(database), - destination, - incarnation, - schema, - root, - reservation, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Runs one synchronous handler on the Cell's assigned SQLite worker. - pub async fn execute( - &self, - cell: CellId, - identity: MutationIdentity, - operation_digest: Digest, - now_ms: i64, - max_result_bytes: usize, - handler: F, - ) -> Result - where - F: for<'connection> FnOnce( - &crab_ltx::rusqlite::Transaction<'connection>, - ) -> Result - + Send - + 'static, - { - self.execute_until( - cell, - identity, - operation_digest, - now_ms, - max_result_bytes, - SqlDeadline::new(Instant::now() + DEFAULT_PAGE_IO_DEADLINE), - Box::new(handler), - ) - .await - } - - pub(crate) async fn execute_until( - &self, - cell: CellId, - identity: MutationIdentity, - operation_digest: Digest, - now_ms: i64, - max_result_bytes: usize, - deadline: SqlDeadline, - handler: Handler, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send_worker_job( - cell, - WorkerCommand::Execute { - trace: tracing::Span::current(), - queued_at: Instant::now(), - cell, - identity, - operation_digest, - now_ms, - max_result_bytes, - deadline, - handler, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Runs one registry-verified schema step on the Cell's assigned worker. - pub(crate) async fn migrate( - &self, - cell: CellId, - plan: MigrationPlan, - now_ms: i64, - deadline: SqlDeadline, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send_worker_job( - cell, - WorkerCommand::Migrate { - cell, - plan, - now_ms, - deadline, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Applies one destination inbox delivery on the Cell's assigned worker. - pub(crate) async fn deliver_effect( - &self, - cell: CellId, - delivery: InboxDelivery, - now_ms: i64, - max_result_bytes: usize, - deadline: SqlDeadline, - handler: Handler, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send_worker_job( - cell, - WorkerCommand::DeliverEffect { - trace: tracing::Span::current(), - queued_at: Instant::now(), - cell, - delivery, - now_ms, - max_result_bytes, - deadline, - handler, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Runs one synchronous read on the Cell's assigned SQLite worker. - pub(crate) async fn query( - &self, - cell: CellId, - max_result_bytes: usize, - deadline: SqlDeadline, - handler: QueryHandler, - ) -> Result> { - let (reply, response) = oneshot::channel(); - self.send_worker_job( - cell, - WorkerCommand::Query { - cell, - max_result_bytes, - deadline, - handler, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Fetches a bounded sparse batch asynchronously, then installs it on its worker. - pub(crate) async fn hydrate( - &self, - cell: CellId, - pages: u32, - deadline: Instant, - ) -> Result { - let (reply, response) = oneshot::channel(); - let preparation = async { - self.send_worker_job( - cell, - WorkerCommand::PrepareHydration { - cell, - pages, - deadline: SqlDeadline::new(deadline), - reply, - }, - ) - .await?; - receive(response).await - }; - // Preparation only selects pages. Abandoning its waiter cannot install - // bytes or release foreground ownership, even if it was dispatched. - let read = match tokio::time::timeout_at(deadline.into(), preparation).await { - Ok(Err(Error::Deadline)) => return Ok(HydrationStep::Deferred(Duration::ZERO)), - Ok(result) => result?, - Err(_) => return Ok(HydrationStep::Deferred(Duration::ZERO)), - }; - let Some(read) = read else { - return Ok(HydrationStep::Progress(None)); - }; - let retained = match self - .inner - .resources - .try_reserve(ResourceCost::zero().with_retained_bytes(read.retained_bytes())) - { - Ok(retained) => retained, - // Background work yields to retained foreground bytes. No page was - // fetched or installed, so retry later without fencing the owner. - Err(Error::Capacity(_)) => return Ok(HydrationStep::Deferred(Duration::ZERO)), - Err(error) => return Err(error), - }; - let batch = match tokio::time::timeout_at(deadline.into(), read.fetch()).await { - Ok(Ok(batch)) => batch, - Ok(Err(error)) => match error.classify() { - crab_ltx::FailureClass::Retryable { after } => { - return Ok(HydrationStep::Deferred(after.unwrap_or_default())); - } - crab_ltx::FailureClass::Capacity => { - return Ok(HydrationStep::Deferred(Duration::ZERO)); - } - _ => return Err(error.into()), - }, - Err(_) => return Ok(HydrationStep::Deferred(Duration::ZERO)), - }; - let (reply, response) = oneshot::channel(); - // Move the payload's reservation into the dispatched install: dropping - // its waiter cannot release bytes still owned by the worker queue. - let install_deadline = SqlDeadline::new(deadline); - let installation = async { - self.send_worker_job( - cell, - WorkerCommand::InstallHydration { - cell, - batch, - retained, - deadline: install_deadline.clone(), - reply, - }, - ) - .await?; - receive(response).await - }; - tokio::pin!(installation); - let result = match tokio::time::timeout_at(deadline.into(), &mut installation).await { - Ok(result) => result, - Err(_) => { - // Queued installation cannot run after cancellation. Retain its - // admission until acknowledgement; started writes remain uncertain. - install_deadline.cancel_queued(); - let _ = installation.await; - Err(Error::Deadline) - } - }; - match result { - Err(Error::Deadline) if install_deadline.cancelled() => { - Ok(HydrationStep::Deferred(Duration::ZERO)) - } - result => result.map(|progress| HydrationStep::Progress(Some(progress))), - } - } - - pub(crate) async fn hydration(&self, cell: CellId) -> Result> { - let (reply, response) = oneshot::channel(); - self.send(cell, WorkerCommand::Hydration { cell, reply }) - .await?; - receive(response).await - } - - pub(crate) async fn persisted_work_inventory( - &self, - cell: CellId, - role: CatalogRole, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send_worker_job(cell, WorkerCommand::PersistedWork { cell, role, reply }) - .await?; - receive(response).await - } - - pub(crate) async fn transfer_work_inventory( - &self, - cell: CellId, - role: CatalogRole, - now_ms: i64, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send_worker_job( - cell, - WorkerCommand::TransferWork { - cell, - role, - now_ms, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Resolves one request ledger entry on the Cell's assigned worker. - pub(crate) async fn resolve( - &self, - cell: CellId, - identity: MutationIdentity, - operation_digest: Digest, - now_ms: i64, - max_result_bytes: usize, - deadline: SqlDeadline, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send_worker_job( - cell, - WorkerCommand::Resolve { - cell, - identity, - operation_digest, - now_ms, - max_result_bytes, - deadline, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Resolves one destination inbox identity on the Cell's assigned worker. - pub(crate) async fn resolve_effect( - &self, - cell: CellId, - delivery: InboxDelivery, - now_ms: i64, - max_result_bytes: usize, - deadline: SqlDeadline, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send_worker_job( - cell, - WorkerCommand::ResolveEffect { - cell, - delivery, - now_ms, - max_result_bytes, - deadline, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Binds the immutable proposal before any authority CAS can use it. - pub async fn bind_prepared( - &self, - cell: CellId, - prepared: crab_ltx::PreparedRoot, - ) -> Result<()> { - let (reply, response) = oneshot::channel(); - self.send( - cell, - WorkerCommand::BindPrepared { - cell, - prepared: Box::new(prepared), - reply, - }, - ) - .await?; - receive(response).await - } - - pub(crate) async fn bind_migration_prepared( - &self, - cell: CellId, - prepared: crab_ltx::PreparedRoot, - ) -> Result<()> { - let (reply, response) = oneshot::channel(); - self.send( - cell, - WorkerCommand::BindMigrationPrepared { - cell, - prepared: Box::new(prepared), - reply, - }, - ) - .await?; - receive(response).await - } - - /// Returns a clone of worker-owned publication state without releasing it. - pub async fn pending(&self, cell: CellId) -> Result> { - let (reply, response) = oneshot::channel(); - self.send(cell, WorkerCommand::Pending { cell, reply }) - .await?; - receive(response).await - } - - /// Releases one worker-retained result after exact root publication. - pub async fn confirm_published( - &self, - cell: CellId, - root: crab_ltx::RootRef, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send(cell, WorkerCommand::ConfirmPublished { cell, root, reply }) - .await?; - receive(response).await - } - - pub(crate) async fn confirm_durable(&self, cell: CellId, commit_sequence: u64) -> Result<()> { - let (reply, response) = oneshot::channel(); - self.send( - cell, - WorkerCommand::ConfirmDurable { - cell, - commit_sequence, - reply, - }, - ) - .await?; - receive(response).await - } - - pub(crate) async fn confirm_migration_published( - &self, - cell: CellId, - root: crab_ltx::RootRef, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send( - cell, - WorkerCommand::ConfirmMigrationPublished { cell, root, reply }, - ) - .await?; - receive(response).await - } - - /// Stops admission after a publication or worker invariant becomes unsafe. - pub(crate) async fn fence(&self, cell: CellId) -> Result<()> { - let (reply, response) = oneshot::channel(); - self.send(cell, WorkerCommand::Fence { cell, reply }) - .await?; - receive(response).await - } - - pub(crate) async fn state(&self, cell: CellId) -> Result { - let (reply, response) = oneshot::channel(); - self.send(cell, WorkerCommand::State { cell, reply }) - .await?; - receive(response).await - } - - pub(crate) async fn interrupt_handle( - &self, - cell: CellId, - ) -> Result { - let (reply, response) = oneshot::channel(); - self.send(cell, WorkerCommand::InterruptHandle { cell, reply }) - .await?; - receive(response).await - } - - /// Closes and removes one fully drained Cell from its worker. - pub async fn deactivate(&self, cell: CellId) -> Result<()> { - let (reply, response) = oneshot::channel(); - self.send(cell, WorkerCommand::Deactivate { cell, reply }) - .await?; - receive(response).await - } - - /// Closes a drained Cell and leaves a resume record for the root it holds. - pub(crate) async fn deactivate_resumable( - &self, - cell: CellId, - root: crate::control::RootRef, - code: Digest, - ) -> Result<()> { - let (reply, response) = oneshot::channel(); - self.send( - cell, - WorkerCommand::DeactivateResumable { - cell, - root, - code, - reply, - }, - ) - .await?; - receive(response).await - } - - /// Removes and closes a fenced Cell for authoritative-root recovery. - pub(crate) async fn discard(&self, cell: CellId) -> Result<()> { - let (reply, response) = oneshot::channel(); - self.send(cell, WorkerCommand::Discard { cell, reply }) - .await?; - receive(response).await - } - - /// Closes the empty pool and drains every admitted SQL job and worker thread. - /// - /// All Cells must first be drained and deactivated. Once shutdown starts, - /// every clone is permanently closed and a second call returns - /// [`Error::RuntimeClosed`]. - pub async fn shutdown(&self) -> Result<()> { - let threads = { - let mut lifecycle = self - .inner - .lifecycle - .lock() - .map_err(|_| Error::RuntimeClosed)?; - if lifecycle.closing { - return Err(Error::RuntimeClosed); - } - if self - .inner - .resources - .snapshot() - .map(|snapshot| snapshot.used.active_cells() != 0) - .unwrap_or(true) - { - return Err(Error::Control( - "SQL worker shutdown requires every Cell to be deactivated", - )); - } - lifecycle.closing = true; - for permits in &self.inner.worker_permits { - permits.close(); - } - lifecycle.workers.clear(); - std::mem::take(&mut lifecycle.threads) - }; - let joined = tokio::task::spawn_blocking(move || { - let mut panicked = false; - for thread in threads { - if thread.join().is_err() { - panicked = true; - } - } - if panicked { - Err(Error::WorkerPanic) - } else { - Ok(()) - } - }) - .await - .map_err(Error::WorkerJoin)?; - // Immutable replica SQL runs outside the dedicated worker threads. - // Wait for its existing admission charges before node withdrawal can - // let offline retention remove objects still being read. - let _drained = self - .inner - .resources - .reserve(ResourceCost::zero().with_worker_jobs(self.inner.worker_count)) - .await?; - joined - } - - async fn send(&self, cell: CellId, command: WorkerCommand) -> Result<()> { - let sender = { - let lifecycle = self - .inner - .lifecycle - .lock() - .map_err(|_| Error::RuntimeClosed)?; - if lifecycle.closing { - return Err(Error::RuntimeClosed); - } - lifecycle.workers[worker_index(cell, self.inner.worker_count)].clone() - }; - sender.send(command).await.map_err(|_| Error::RuntimeClosed) - } - - async fn send_worker_job(&self, cell: CellId, command: WorkerCommand) -> Result<()> { - let reservation = self.reserve_job(cell).await?; - self.send( - cell, - WorkerCommand::Reserved { - command: Box::new(command), - reservation, - }, - ) - .await - } - - pub(crate) async fn reserve_snapshot_job(&self) -> Result { - // Immutable connections have no writer-thread affinity. Borrow any SQL - // slot so refresh can progress beside an old view, within the node cap. - let permits = self - .inner - .worker_permits - .iter() - .map(|permit| Box::pin(Arc::clone(permit).acquire_owned())); - let (permit, _, _) = futures_util::future::select_all(permits).await; - self.job_reservation(permit.map_err(|_| Error::RuntimeClosed)?) - } - - async fn reserve_job(&self, cell: CellId) -> Result { - let worker_permits = { - let lifecycle = self - .inner - .lifecycle - .lock() - .map_err(|_| Error::RuntimeClosed)?; - if lifecycle.closing { - return Err(Error::RuntimeClosed); - } - // A queued job must wait for its own shard, without consuming - // the admission capacity an idle worker needs to make progress. - Arc::clone(&self.inner.worker_permits[worker_index(cell, self.inner.worker_count)]) - }; - let permit = worker_permits - .acquire_owned() - .await - .map_err(|_| Error::RuntimeClosed)?; - self.job_reservation(permit) - } - - fn job_reservation(&self, permit: OwnedSemaphorePermit) -> Result { - let lifecycle = self - .inner - .lifecycle - .lock() - .map_err(|_| Error::RuntimeClosed)?; - if lifecycle.closing { - return Err(Error::RuntimeClosed); - } - // Serialize the final admission with shutdown's close. A permit won - // just before closure must not create work after the drain barrier. - let reservation = self - .inner - .resources - .try_reserve(ResourceCost::zero().with_worker_jobs(1))?; - Ok(WorkerJobReservation { - _reservation: reservation, - _permit: permit, - }) - } - - pub(crate) fn reserve_activation(&self) -> Result { - let lifecycle = self - .inner - .lifecycle - .lock() - .map_err(|_| Error::RuntimeClosed)?; - if lifecycle.closing { - return Err(Error::RuntimeClosed); - } - let reservation = self - .inner - .resources - .try_reserve(ResourceCost::active_cell()) - .map_err(|error| match error { - Error::Capacity(_) => Error::Capacity("active Cells per node"), - error => error, - })?; - Ok(CellReservation { - _reservation: reservation, - }) - } - - pub(crate) fn configure_retained_capacity(&self, bytes: usize) -> Result<()> { - self.inner.resources.set_retained_limit(bytes) - } - - pub(crate) fn resource_ledger(&self) -> ResourceLedger { - self.inner.resources.clone() - } - - pub(crate) fn active_cells(&self) -> usize { - self.inner - .resources - .snapshot() - .map(|snapshot| snapshot.used.active_cells()) - .unwrap_or(self.inner.max_active_cells) - } - - pub(crate) fn active_cell_capacity(&self) -> usize { - self.inner.max_active_cells - } -} - -struct PoolInner { - lifecycle: Mutex, - resources: ResourceLedger, - max_active_cells: usize, - worker_count: usize, - worker_permits: Vec>, -} - -struct WorkerLifecycle { - workers: Vec>, - threads: Vec>, - closing: bool, -} - -impl Drop for PoolInner { - fn drop(&mut self) { - for permits in &self.worker_permits { - permits.close(); - } - let lifecycle = match self.lifecycle.get_mut() { - Ok(lifecycle) => lifecycle, - Err(poisoned) => poisoned.into_inner(), - }; - lifecycle.workers.clear(); - let threads = std::mem::take(&mut lifecycle.threads); - for thread in threads { - let _ = thread.join(); - } - } -} - -/// How one activation reaches its exact root before the worker serves it. -pub(crate) enum RestoredDatabase { - /// The exact root is restored into a fresh destination and read sparsely. - Paged(Box), - /// A local database already holds the root, so the origin is never read. - Local(Box), -} - -enum WorkerCommand { - Reserved { - command: Box, - reservation: WorkerJobReservation, - }, - Activate { - cell: CellId, - executor: Box, - reservation: CellReservation, - reply: oneshot::Sender>, - }, - ActivateRestored { - cell: CellId, - database: Box, - destination: PathBuf, - incarnation: crate::identity::IncarnationId, - schema: u32, - root: crab_ltx::RootRef, - reservation: CellReservation, - reply: oneshot::Sender>, - }, - Bootstrap(Box), - Execute { - trace: tracing::Span, - queued_at: Instant, - cell: CellId, - identity: MutationIdentity, - operation_digest: Digest, - now_ms: i64, - max_result_bytes: usize, - deadline: SqlDeadline, - handler: Handler, - reply: oneshot::Sender>, - }, - Migrate { - cell: CellId, - plan: MigrationPlan, - now_ms: i64, - deadline: SqlDeadline, - reply: oneshot::Sender>, - }, - DeliverEffect { - trace: tracing::Span, - queued_at: Instant, - cell: CellId, - delivery: InboxDelivery, - now_ms: i64, - max_result_bytes: usize, - deadline: SqlDeadline, - handler: Handler, - reply: oneshot::Sender>, - }, - Query { - cell: CellId, - max_result_bytes: usize, - deadline: SqlDeadline, - handler: QueryHandler, - reply: oneshot::Sender>>, - }, - PrepareHydration { - cell: CellId, - pages: u32, - deadline: SqlDeadline, - reply: oneshot::Sender>>, - }, - InstallHydration { - cell: CellId, - batch: crab_ltx::db::HydrationBatch, - retained: ResourceReservation, - deadline: SqlDeadline, - reply: oneshot::Sender>, - }, - Hydration { - cell: CellId, - reply: oneshot::Sender>>, - }, - PersistedWork { - cell: CellId, - role: CatalogRole, - reply: oneshot::Sender>, - }, - TransferWork { - cell: CellId, - role: CatalogRole, - now_ms: i64, - reply: oneshot::Sender>, - }, - Resolve { - cell: CellId, - identity: MutationIdentity, - operation_digest: Digest, - now_ms: i64, - max_result_bytes: usize, - deadline: SqlDeadline, - reply: oneshot::Sender>, - }, - ResolveEffect { - cell: CellId, - delivery: InboxDelivery, - now_ms: i64, - max_result_bytes: usize, - deadline: SqlDeadline, - reply: oneshot::Sender>, - }, - BindPrepared { - cell: CellId, - prepared: Box, - reply: oneshot::Sender>, - }, - BindMigrationPrepared { - cell: CellId, - prepared: Box, - reply: oneshot::Sender>, - }, - Pending { - cell: CellId, - reply: oneshot::Sender>>, - }, - ConfirmPublished { - cell: CellId, - root: crab_ltx::RootRef, - reply: oneshot::Sender>, - }, - ConfirmDurable { - cell: CellId, - commit_sequence: u64, - reply: oneshot::Sender>, - }, - ConfirmBootstrapPublished { - cell: CellId, - cuts: Box, - reply: oneshot::Sender>, - }, - ConfirmMigrationPublished { - cell: CellId, - root: crab_ltx::RootRef, - reply: oneshot::Sender>, - }, - Fence { - cell: CellId, - reply: oneshot::Sender>, - }, - State { - cell: CellId, - reply: oneshot::Sender>, - }, - InterruptHandle { - cell: CellId, - reply: oneshot::Sender>, - }, - Deactivate { - cell: CellId, - reply: oneshot::Sender>, - }, - DeactivateResumable { - cell: CellId, - root: crate::control::RootRef, - code: Digest, - reply: oneshot::Sender>, - }, - Discard { - cell: CellId, - reply: oneshot::Sender>, - }, -} - -pub(crate) struct WorkerJobReservation { - _reservation: ResourceReservation, - _permit: OwnedSemaphorePermit, -} - -struct WorkerBootstrap { - cell: CellId, - replica: crab_ltx::CellReplica, - destination: PathBuf, - incarnation: crate::identity::IncarnationId, - schema: u32, - initialize: Initializer, - reservation: CellReservation, - reply: oneshot::Sender>, -} - -struct ActiveCell { - executor: CellExecutor, - _reservation: CellReservation, -} - -pub(crate) struct CellReservation { - _reservation: ResourceReservation, -} - -fn worker_index(cell: CellId, workers: usize) -> usize { - let mut prefix = [0; 8]; - prefix.copy_from_slice(&cell.as_bytes()[..8]); - u64::from_be_bytes(prefix) as usize % workers -} - -async fn receive(response: oneshot::Receiver>) -> Result { - response.await.map_err(|_| Error::RuntimeClosed)? -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/cell/worker/run.rs b/crates/crab-cell-runtime/src/cell/worker/run.rs deleted file mode 100644 index e562be15b..000000000 --- a/crates/crab-cell-runtime/src/cell/worker/run.rs +++ /dev/null @@ -1,462 +0,0 @@ -//! The worker thread loop and its command handler. -//! -//! One OS thread owns each shard of Cells: it receives `WorkerCommand`s, -//! runs them against the shard's executors, and fences a Cell whose native -//! callback panicked instead of leaving a poisoned executor behind. - -use super::*; - -pub(super) fn run_worker(mut receiver: mpsc::Receiver) { - let mut cells = HashMap::new(); - while let Some(command) = receiver.blocking_recv() { - match command { - WorkerCommand::Reserved { - command, - reservation, - } => { - run_worker_command(*command, &mut cells, Some(reservation)); - } - command => run_worker_command(command, &mut cells, None), - } - } -} - -fn run_worker_command( - command: WorkerCommand, - cells: &mut HashMap, - mut reservation: Option, -) { - match command { - WorkerCommand::Reserved { - command, - reservation, - } => { - run_worker_command(*command, cells, Some(reservation)); - } - WorkerCommand::Activate { - cell, - executor, - reservation, - reply, - } => { - let result = match cells.entry(cell) { - std::collections::hash_map::Entry::Occupied(_) => Err(Error::CellAlreadyActive), - std::collections::hash_map::Entry::Vacant(entry) => { - entry.insert(ActiveCell { - executor: *executor, - _reservation: reservation, - }); - Ok(()) - } - }; - let _ = reply.send(result); - } - WorkerCommand::ActivateRestored { - cell, - database, - destination, - incarnation, - schema, - root, - reservation, - reply, - } => { - let result = match cells.entry(cell) { - std::collections::hash_map::Entry::Occupied(_) => Err(Error::CellAlreadyActive), - std::collections::hash_map::Entry::Vacant(entry) => match *database { - RestoredDatabase::Paged(database) => database.open_writable(&destination), - RestoredDatabase::Local(database) => Ok(*database), - } - .map_err(Error::from) - .and_then(|db| CellExecutor::from_restored(db, cell, incarnation, schema, root)) - .map(|executor| { - entry.insert(ActiveCell { - executor, - _reservation: reservation, - }); - }), - }; - let _ = reply.send(result); - } - WorkerCommand::Bootstrap(bootstrap) => { - let WorkerBootstrap { - cell, - replica, - destination, - incarnation, - schema, - initialize, - reservation, - reply, - } = *bootstrap; - let result = match catch_unwind(AssertUnwindSafe(|| match cells.entry(cell) { - std::collections::hash_map::Entry::Occupied(_) => Err(Error::CellAlreadyActive), - std::collections::hash_map::Entry::Vacant(entry) => replica - .open_new(&destination) - .map_err(Error::from) - .and_then(|db| { - CellExecutor::bootstrap(db, cell, incarnation, schema, initialize) - }) - .map(|(executor, cuts, next_due_ms)| { - entry.insert(ActiveCell { - executor, - _reservation: reservation, - }); - BootstrapExecution { cuts, next_due_ms } - }), - })) { - Ok(result) => result, - Err(_) => Err(Error::NativePanic), - }; - let _ = reply.send(result); - } - WorkerCommand::Execute { - trace, - queued_at, - cell, - identity, - operation_digest, - now_ms, - max_result_bytes, - deadline, - handler, - reply, - } => { - let _trace = trace.enter(); - let started = Instant::now(); - tracing::debug!( - target: "crab_cell_runtime::action", - event = "cell_worker_started", - worker_queue_us = queued_at.elapsed().as_micros(), - ); - let result = run_native_callback(cells, cell, deadline, move |active| { - match active.executor.execute( - identity, - operation_digest, - now_ms, - max_result_bytes, - handler, - )? { - CommandExecution::Recorded(outcome) => Ok(WorkerExecution::Recorded(outcome)), - CommandExecution::Pending => active - .executor - .latest_pending() - .cloned() - .map(Box::new) - .map(WorkerExecution::Pending) - .ok_or(Error::Fenced), - } - }); - tracing::debug!( - target: "crab_cell_runtime::action", - event = "cell_worker_completed", - worker_execute_us = started.elapsed().as_micros(), - succeeded = result.is_ok(), - ); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::Migrate { - cell, - plan, - now_ms, - deadline, - reply, - } => { - let result = run_native_callback(cells, cell, deadline, move |active| { - active.executor.migrate(plan, now_ms)?; - active - .executor - .pending_migration() - .cloned() - .ok_or(Error::Fenced) - }); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::DeliverEffect { - trace, - queued_at, - cell, - delivery, - now_ms, - max_result_bytes, - deadline, - handler, - reply, - } => { - let _trace = trace.enter(); - let started = Instant::now(); - tracing::debug!( - target: "crab_cell_runtime::action", - event = "cell_worker_started", - worker_queue_us = queued_at.elapsed().as_micros(), - ); - let result = run_native_callback(cells, cell, deadline, move |active| { - match active - .executor - .deliver_effect(delivery, now_ms, max_result_bytes, handler)? - { - CommandExecution::Recorded(outcome) => Ok(WorkerExecution::Recorded(outcome)), - CommandExecution::Pending => active - .executor - .latest_pending() - .cloned() - .map(Box::new) - .map(WorkerExecution::Pending) - .ok_or(Error::Fenced), - } - }); - tracing::debug!( - target: "crab_cell_runtime::action", - event = "cell_worker_completed", - worker_execute_us = started.elapsed().as_micros(), - succeeded = result.is_ok(), - ); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::Query { - cell, - max_result_bytes, - deadline, - handler, - reply, - } => { - let result = run_native_callback(cells, cell, deadline, move |active| { - active.executor.query(max_result_bytes, handler) - }); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::PrepareHydration { - cell, - pages, - deadline, - reply, - } => { - let result = run_native_callback(cells, cell, deadline, move |active| { - active.executor.prepare_hydration(pages) - }); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::InstallHydration { - cell, - batch, - retained, - deadline, - reply, - } => { - let result = run_native_callback(cells, cell, deadline, move |active| { - active.executor.install_hydration(batch) - }); - drop(retained); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::Hydration { cell, reply } => { - let result = cells - .get(&cell) - .ok_or(Error::CellNotActive) - .and_then(|active| active.executor.hydration()); - let _ = reply.send(result); - } - WorkerCommand::PersistedWork { cell, role, reply } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(|active| active.executor.persisted_work_inventory(role)); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::TransferWork { - cell, - role, - now_ms, - reply, - } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(|active| active.executor.transfer_work_inventory(role, now_ms)); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::Resolve { - cell, - identity, - operation_digest, - now_ms, - max_result_bytes, - deadline, - reply, - } => { - let result = run_native_callback(cells, cell, deadline, |active| { - active - .executor - .resolve(identity, operation_digest, now_ms, max_result_bytes) - }); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::ResolveEffect { - cell, - delivery, - now_ms, - max_result_bytes, - deadline, - reply, - } => { - let result = run_native_callback(cells, cell, deadline, |active| { - active - .executor - .resolve_effect(delivery, now_ms, max_result_bytes) - }); - drop(reservation.take()); - let _ = reply.send(result); - } - WorkerCommand::BindPrepared { - cell, - prepared, - reply, - } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.bind_prepared(&prepared)); - let _ = reply.send(result); - } - WorkerCommand::BindMigrationPrepared { - cell, - prepared, - reply, - } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.bind_migration_prepared(&prepared)); - let _ = reply.send(result); - } - WorkerCommand::Pending { cell, reply } => { - let result = cells - .get(&cell) - .ok_or(Error::CellNotActive) - .map(|cell| cell.executor.pending().cloned()); - let _ = reply.send(result); - } - WorkerCommand::ConfirmPublished { cell, root, reply } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.confirm_published(&root)); - let _ = reply.send(result); - } - WorkerCommand::ConfirmDurable { - cell, - commit_sequence, - reply, - } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.confirm_durable(commit_sequence)); - let _ = reply.send(result); - } - WorkerCommand::ConfirmBootstrapPublished { cell, cuts, reply } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.confirm_bootstrap_published(&cuts)); - let _ = reply.send(result); - } - WorkerCommand::ConfirmMigrationPublished { cell, root, reply } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.confirm_migration_published(&root)); - let _ = reply.send(result); - } - WorkerCommand::Fence { cell, reply } => { - let result = cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .map(|cell| cell.executor.fence()); - let _ = reply.send(result); - } - WorkerCommand::State { cell, reply } => { - let result = cells - .get(&cell) - .ok_or(Error::CellNotActive) - .map(|cell| cell.executor.worker_state()); - let _ = reply.send(result); - } - WorkerCommand::InterruptHandle { cell, reply } => { - let result = cells - .get(&cell) - .ok_or(Error::CellNotActive) - .map(|cell| cell.executor.interrupt_handle()); - let _ = reply.send(result); - } - WorkerCommand::Deactivate { cell, reply } => { - let result = match cells.get(&cell) { - None => Err(Error::CellNotActive), - Some(cell) if !cell.executor.drained() => Err(Error::PendingPublication), - Some(_) => cells - .remove(&cell) - .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.close()), - }; - let _ = reply.send(result); - } - WorkerCommand::DeactivateResumable { - cell, - root, - code, - reply, - } => { - let result = match cells.get(&cell) { - None => Err(Error::CellNotActive), - Some(cell) if !cell.executor.drained() => Err(Error::PendingPublication), - Some(_) => cells - .remove(&cell) - .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.close_resumable(root, code)), - }; - let _ = reply.send(result); - } - WorkerCommand::Discard { cell, reply } => { - let result = cells - .remove(&cell) - .ok_or(Error::CellNotActive) - .and_then(|cell| cell.executor.discard()); - let _ = reply.send(result); - } - } -} - -fn run_native_callback( - cells: &mut HashMap, - cell: CellId, - deadline: SqlDeadline, - callback: impl FnOnce(&mut ActiveCell) -> Result, -) -> Result { - deadline.start()?; - let result = catch_unwind(AssertUnwindSafe(|| { - crab_ltx::with_paged_io_deadline(deadline.at(), || { - cells - .get_mut(&cell) - .ok_or(Error::CellNotActive) - .and_then(callback) - }) - })); - match result { - Ok(result) => result, - Err(_) => { - if let Some(active) = cells.get_mut(&cell) { - active.executor.fence(); - } - Err(Error::NativePanic) - } - } -} diff --git a/crates/crab-cell-runtime/src/cell/worker/tests.rs b/crates/crab-cell-runtime/src/cell/worker/tests.rs deleted file mode 100644 index e73849082..000000000 --- a/crates/crab-cell-runtime/src/cell/worker/tests.rs +++ /dev/null @@ -1,721 +0,0 @@ -use super::*; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::Store; -use object_store::{ - memory::InMemory, - path::Path, - throttle::{ThrottleConfig, ThrottledStore}, -}; -use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; - -struct SparseActivation { - _source: tempfile::TempDir, - _destination: tempfile::TempDir, - cell: CellId, - incarnation: crate::identity::IncarnationId, - replica: CellReplica, - root: crab_ltx::RootRef, - database: crab_ltx::CellWritableDatabase, - destination: PathBuf, -} - -#[tokio::test(flavor = "multi_thread")] -async fn worker_and_runtime_reservations_share_one_node_ledger() { - let pool = SqlWorkerPool::new(1, 1).unwrap(); - pool.configure_retained_capacity(32).unwrap(); - let ledger = pool.resource_ledger(); - let active = pool.reserve_activation().unwrap(); - let retained = ledger - .try_reserve(ResourceCost::zero().with_retained_bytes(16)) - .unwrap(); - let snapshot = ledger.snapshot().unwrap(); - assert_eq!(snapshot.used.active_cells(), 1); - assert_eq!(snapshot.used.retained_bytes(), 16); - assert!( - ledger - .try_reserve(ResourceCost::zero().with_retained_bytes(17)) - .is_err() - ); - let job = ledger - .try_reserve(ResourceCost::zero().with_primitive_jobs(1)) - .unwrap(); - assert!( - ledger - .try_reserve(ResourceCost::zero().with_primitive_jobs(1)) - .is_err() - ); - drop(job); - drop(retained); - drop(active); - assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); - pool.shutdown().await.unwrap(); -} - -async fn sparse_activation(cell_byte: u8, store: Store, payload_bytes: usize) -> SparseActivation { - let cell = CellId::from_bytes([cell_byte; 32]); - let incarnation = crate::identity::IncarnationId::from_bytes([cell_byte + 16; 16]); - let layout = CellStorageLayout::new(store, Path::from("sparse-workers"), [9; 16]); - let replica = CellReplica::new( - layout, - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let source = tempfile::TempDir::new().unwrap(); - let managed = replica - .open_new(&source.path().join("source.sqlite")) - .unwrap(); - let (executor, cuts, _) = - CellExecutor::bootstrap(managed, cell, incarnation, 1, |transaction| { - transaction.execute_batch("CREATE TABLE payload(value BLOB NOT NULL)")?; - transaction.execute( - "INSERT INTO payload VALUES(randomblob(?1))", - [payload_bytes], - )?; - Ok(()) - }) - .unwrap(); - executor.close().unwrap(); - let root = replica.prepare(None, &cuts, 0, 1).await.unwrap().root(); - let verified = replica.open_root(&root).await.unwrap(); - let destination_dir = tempfile::TempDir::new().unwrap(); - let destination = destination_dir.path().join("active.sqlite"); - let database = verified - .paged() - .prepare_writable(&destination) - .await - .unwrap(); - SparseActivation { - _source: source, - _destination: destination_dir, - cell, - incarnation, - replica, - root, - database, - destination, - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn sparse_fault_pool_progresses_under_saturated_sql_workers() { - let backend = Arc::new(ThrottledStore::new( - InMemory::new(), - ThrottleConfig::default(), - )); - let read_bytes = Arc::new(AtomicU64::new(0)); - let observed = read_bytes.clone(); - let store = Store::new(backend.clone()).with_read_byte_observer(Arc::new(move |bytes| { - observed.fetch_add(bytes, Ordering::Relaxed); - })); - // Repeated-byte Cell prefixes select distinct workers modulo two. - let first = sparse_activation(0, store.clone(), 262_144).await; - let second = sparse_activation(1, store, 262_144).await; - assert_eq!(worker_index(first.cell, 2), 0); - assert_eq!(worker_index(second.cell, 2), 1); - read_bytes.store(0, Ordering::Relaxed); - backend.config_mut(|config| config.wait_get_per_call = Duration::from_millis(100)); - - let pool = SqlWorkerPool::new(2, 2).unwrap(); - pool.configure_retained_capacity(1 << 20).unwrap(); - let first_reservation = pool.reserve_activation().unwrap(); - let second_reservation = pool.reserve_activation().unwrap(); - let activation = tokio::time::timeout(Duration::from_secs(5), async { - tokio::join!( - pool.activate_restored( - first.cell, - crate::cell::worker::RestoredDatabase::Paged(Box::new(first.database)), - first.destination, - first.incarnation, - 1, - first.root, - first_reservation, - ), - pool.activate_restored( - second.cell, - crate::cell::worker::RestoredDatabase::Paged(Box::new(second.database)), - second.destination, - second.incarnation, - 1, - second.root, - second_reservation, - ), - ) - }) - .await - .expect("dedicated sparse I/O worker must progress while both SQL workers wait"); - activation.0.unwrap(); - activation.1.unwrap(); - assert!(read_bytes.load(Ordering::Relaxed) > 0); - - let initial = pool.hydration(first.cell).await.unwrap().unwrap(); - let HydrationStep::Progress(Some(progressed)) = pool - .hydrate(first.cell, 64, Instant::now() + Duration::from_secs(5)) - .await - .unwrap() - else { - panic!("hydration unexpectedly deferred") - }; - assert!(progressed.resolved >= initial.resolved); - let mut progress = progressed; - while !progress.complete() { - let HydrationStep::Progress(Some(next)) = pool - .hydrate(first.cell, 64, Instant::now() + Duration::from_secs(5)) - .await - .unwrap() - else { - panic!("hydration unexpectedly deferred") - }; - progress = next; - } - read_bytes.store(0, Ordering::Relaxed); - let length = pool - .query( - first.cell, - 1024, - SqlDeadline::new(Instant::now() + Duration::from_secs(5)), - Box::new(|connection| { - let length: i64 = - connection - .query_row("SELECT length(value) FROM payload", [], |row| row.get(0))?; - Ok(length.to_le_bytes().to_vec()) - }), - ) - .await - .unwrap(); - assert_eq!(i64::from_le_bytes(length.try_into().unwrap()), 262_144); - assert_eq!(read_bytes.load(Ordering::Relaxed), 0); - - pool.deactivate(first.cell).await.unwrap(); - pool.deactivate(second.cell).await.unwrap(); - pool.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn background_hydration_leaves_both_workers_query_admission_available() { - let backend = Arc::new(ThrottledStore::new( - InMemory::new(), - ThrottleConfig::default(), - )); - let armed = Arc::new(AtomicBool::new(false)); - let started = Arc::new(tokio::sync::Notify::new()); - let notify = started.clone(); - let watching = armed.clone(); - let store = Store::new(backend.clone()).with_read_request_observer(Arc::new(move |_| { - if watching.load(Ordering::Acquire) { - notify.notify_one(); - } - })); - let cold = sparse_activation(0, store, 4 << 20).await; - let resident = sparse_activation(1, Store::new(Arc::new(InMemory::new())), 262_144).await; - let queued = sparse_activation(2, Store::new(Arc::new(InMemory::new())), 262_144).await; - let pool = SqlWorkerPool::new(2, 3).unwrap(); - pool.configure_retained_capacity(1 << 20).unwrap(); - let cells = [cold.cell, resident.cell, queued.cell]; - for activation in [&cold, &resident, &queued] { - pool.activate_restored( - activation.cell, - RestoredDatabase::Paged(Box::new(activation.database.clone())), - activation.destination.clone(), - activation.incarnation, - 1, - activation.root, - pool.reserve_activation().unwrap(), - ) - .await - .unwrap(); - } - for cell in [resident.cell, queued.cell] { - while !pool.hydration(cell).await.unwrap().unwrap().complete() { - pool.hydrate(cell, 64, Instant::now() + Duration::from_secs(5)) - .await - .unwrap(); - } - } - - backend.config_mut(|config| config.wait_get_per_call = Duration::from_secs(3)); - armed.store(true, Ordering::Release); - let hydration = { - let pool = pool.clone(); - tokio::spawn(async move { - pool.hydrate(cold.cell, 64, Instant::now() + Duration::from_secs(10)) - .await - }) - }; - tokio::time::timeout(Duration::from_secs(2), started.notified()) - .await - .expect("hydration must reach a delayed origin read"); - let query = |cell| { - tokio::time::timeout( - Duration::from_secs(1), - pool.query( - cell, - 8, - SqlDeadline::new(Instant::now() + Duration::from_secs(10)), - Box::new(|connection| { - let length: i64 = - connection - .query_row("SELECT length(value) FROM payload", [], |row| row.get(0))?; - Ok(length.to_le_bytes().to_vec()) - }), - ), - ) - }; - let (same_worker, other_worker) = tokio::join!(query(queued.cell), query(resident.cell)); - backend.config_mut(|config| config.wait_get_per_call = Duration::ZERO); - hydration.await.unwrap().unwrap(); - for cell in cells { - pool.deactivate(cell).await.unwrap(); - } - pool.shutdown().await.unwrap(); - for result in [same_worker, other_worker] { - assert_eq!( - result - .expect("background hydration blocked a resident query") - .unwrap(), - 262_144_i64.to_le_bytes() - ); - } -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires isolated RustFS credentials; reports injected-delay worker interference"] -async fn rustfs_sparse_reads_report_worker_interference() { - rustfs_worker_interference(2).await; -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires isolated RustFS credentials; reports single-worker demand interference"] -async fn rustfs_single_worker_reports_sparse_read_interference() { - rustfs_worker_interference(1).await; -} - -async fn rustfs_worker_interference(worker_count: usize) { - fn payload_digest(connection: &crab_ltx::rusqlite::Connection) -> Result> { - let value: Vec = - connection.query_row("SELECT value FROM payload", [], |row| row.get(0))?; - Ok(blake3::hash(&value).as_bytes().to_vec()) - } - - let required = |name| std::env::var(name).unwrap_or_else(|_| panic!("missing {name}")); - let endpoint = required("CRAB_LTX_TEST_ENDPOINT"); - let store = crab_storage::build_explicit_store( - &required("CRAB_LTX_TEST_BUCKET"), - crab_storage::ObjectStoreCredentials::Aws { - access_key_id: required("AWS_ACCESS_KEY_ID"), - secret_access_key: required("AWS_SECRET_ACCESS_KEY"), - session_token: None, - region: "us-east-1".into(), - }, - Some(&endpoint), - endpoint.starts_with("http://"), - ) - .unwrap(); - let run = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_nanos(); - let prefix = format!("crab-runtime-tests/worker-interference/{run}"); - let backend = Arc::new(ThrottledStore::new( - object_store::prefix::PrefixStore::new(store.inner().clone(), Path::from(prefix.clone())), - ThrottleConfig::default(), - )); - let armed = Arc::new(AtomicBool::new(false)); - let started = Arc::new(tokio::sync::Notify::new()); - let notify = started.clone(); - let watching = armed.clone(); - let requests = Arc::new(AtomicU64::new(0)); - let observed = requests.clone(); - let bytes = Arc::new(AtomicU64::new(0)); - let transferred = bytes.clone(); - let store = Store::new(backend.clone()) - .with_read_byte_observer(Arc::new(move |count| { - transferred.fetch_add(count, Ordering::Relaxed); - })) - .with_read_request_observer(Arc::new(move |_| { - observed.fetch_add(1, Ordering::Relaxed); - if watching.swap(false, Ordering::AcqRel) { - notify.notify_one(); - } - })); - let cold = sparse_activation(0, store.clone(), 4 << 20).await; - let same = sparse_activation(2, store.clone(), 262_144).await; - let other = sparse_activation(1, store, 262_144).await; - let pool = SqlWorkerPool::new(worker_count, 3).unwrap(); - pool.configure_retained_capacity(1 << 20).unwrap(); - assert_eq!( - worker_index(cold.cell, worker_count), - worker_index(same.cell, worker_count) - ); - assert_eq!( - worker_index(cold.cell, worker_count) == worker_index(other.cell, worker_count), - worker_count == 1, - ); - for activation in [&cold, &same, &other] { - pool.activate_restored( - activation.cell, - RestoredDatabase::Paged(Box::new(activation.database.clone())), - activation.destination.clone(), - activation.incarnation, - 1, - activation.root, - pool.reserve_activation().unwrap(), - ) - .await - .unwrap(); - } - for cell in [same.cell, other.cell] { - while !pool.hydration(cell).await.unwrap().unwrap().complete() { - pool.hydrate(cell, 64, Instant::now() + Duration::from_secs(10)) - .await - .unwrap(); - } - } - let query = |cell| { - let pool = pool.clone(); - async move { - let started = Instant::now(); - let (timing, measured) = oneshot::channel(); - let output = pool - .query( - cell, - 8, - SqlDeadline::new(started + Duration::from_secs(10)), - Box::new(move |connection| { - let entered = Instant::now(); - let length: i64 = connection.query_row( - "SELECT length(value) FROM payload", - [], - |row| row.get(0), - )?; - let _ = timing.send((entered.duration_since(started), entered.elapsed())); - Ok(length.to_le_bytes().to_vec()) - }), - ) - .await - .unwrap(); - assert_eq!(output, 262_144_i64.to_le_bytes()); - let elapsed = started.elapsed(); - let (admission, sql) = measured.await.unwrap(); - serde_json::json!({ - "elapsed_us": elapsed.as_micros(), "admission_and_queue_us": admission.as_micros(), - "sql_us": sql.as_micros(), - }) - } - }; - let before = requests.load(Ordering::Relaxed); - let (same_baseline, other_baseline) = tokio::join!(query(same.cell), query(other.cell)); - assert_eq!( - requests.load(Ordering::Relaxed), - before, - "resident queries must use no origin reads" - ); - - backend.config_mut(|config| config.wait_get_per_call = Duration::from_millis(500)); - armed.store(true, Ordering::Release); - let hydration_started = Instant::now(); - let hydration_bytes = bytes.load(Ordering::Relaxed); - let hydration = { - let pool = pool.clone(); - tokio::spawn(async move { - pool.hydrate(cold.cell, 64, hydration_started + Duration::from_secs(10)) - .await - }) - }; - tokio::time::timeout(Duration::from_secs(2), started.notified()) - .await - .expect("hydration must reach the real RustFS path"); - let (hydrated, same_wait, other_wait) = - tokio::join!(hydration, query(same.cell), query(other.cell)); - let hydration_and_queries = hydration_started.elapsed(); - assert!(matches!( - hydrated.unwrap().unwrap(), - HydrationStep::Progress(Some(_)) - )); - armed.store(false, Ordering::Release); - backend.config_mut(|config| config.wait_get_per_call = Duration::ZERO); - let hydration_requests = requests.load(Ordering::Relaxed) - before; - let hydration_bytes = bytes.load(Ordering::Relaxed) - hydration_bytes; - pool.deactivate(cold.cell).await.unwrap(); - - let source = crab_ltx::rusqlite::Connection::open_with_flags( - cold._source.path().join("source.sqlite"), - crab_ltx::rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY, - ) - .unwrap(); - let expected_digest = payload_digest(&source).unwrap(); - drop(source); - let mut demand_samples = Vec::new(); - // Each trial starts with a fresh sparse file for the same immutable root. - // The origin and metadata caches stay warm; no hydrated local pages carry - // between trials. Alternate order to expose shared-host timing variation. - for (trial, delay_ms) in [0, 20, 20, 0, 0, 20].into_iter().enumerate() { - let destination = cold - ._destination - .path() - .join(format!("demand-{trial}.sqlite")); - let database = cold - .replica - .open_root(&cold.root) - .await - .unwrap() - .paged() - .prepare_writable(&destination) - .await - .unwrap(); - pool.activate_restored( - cold.cell, - RestoredDatabase::Paged(Box::new(database)), - destination, - cold.incarnation, - 1, - cold.root, - pool.reserve_activation().unwrap(), - ) - .await - .unwrap(); - let before_requests = requests.load(Ordering::Relaxed); - let before_bytes = bytes.load(Ordering::Relaxed); - backend.config_mut(|config| config.wait_get_per_call = Duration::from_millis(delay_ms)); - armed.store(true, Ordering::Release); - let demand_started = Instant::now(); - let demand = { - let pool = pool.clone(); - tokio::spawn(async move { - let digest = pool - .query( - cold.cell, - 32, - SqlDeadline::new(demand_started + Duration::from_secs(30)), - Box::new(payload_digest), - ) - .await - .unwrap(); - (digest, demand_started.elapsed()) - }) - }; - tokio::time::timeout(Duration::from_secs(2), started.notified()) - .await - .expect("the demand query must fetch from RustFS"); - let (demand, same_wait, other_wait) = - tokio::join!(demand, query(same.cell), query(other.cell)); - let (digest, elapsed) = demand.unwrap(); - assert_eq!(digest, expected_digest); - let demand_requests = requests.load(Ordering::Relaxed) - before_requests; - let demand_bytes = bytes.load(Ordering::Relaxed) - before_bytes; - armed.store(false, Ordering::Release); - backend.config_mut(|config| config.wait_get_per_call = Duration::ZERO); - - let before_warm = requests.load(Ordering::Relaxed); - let warm_started = Instant::now(); - let digest = pool - .query( - cold.cell, - 32, - SqlDeadline::new(warm_started + Duration::from_secs(10)), - Box::new(payload_digest), - ) - .await - .unwrap(); - assert_eq!(digest, expected_digest); - assert_eq!( - requests.load(Ordering::Relaxed), - before_warm, - "the repeated full-payload query must read only materialized pages" - ); - demand_samples.push(serde_json::json!({ - "trial": trial, "injected_get_delay_ms": delay_ms, - "demand_query_us": elapsed.as_micros(), "resident_payload_query_us": warm_started.elapsed().as_micros(), - "same_worker": same_wait, "comparison_cell": other_wait, - "origin_requests": demand_requests, "origin_bytes": demand_bytes, - })); - pool.deactivate(cold.cell).await.unwrap(); - } - for cell in [same.cell, other.cell] { - pool.deactivate(cell).await.unwrap(); - } - pool.shutdown().await.unwrap(); - eprintln!( - "worker-interference {}", - serde_json::json!({ - "schema": 3, - "prefix": prefix, - "build_profile": if cfg!(debug_assertions) { "debug" } else { "release" }, - "sqlite_version": crab_ltx::rusqlite::version(), - "payload_bytes": 4 << 20, - "sql_workers": worker_count, - "available_parallelism": std::thread::available_parallelism().unwrap().get(), - "worker_assignment": { - "cold": worker_index(cold.cell, worker_count), - "same_worker": worker_index(same.cell, worker_count), - "comparison_cell": worker_index(other.cell, worker_count), - }, - "cold_root": {"cell": cold.root.cell, "incarnation": cold.root.incarnation, - "digest": cold.root.digest, "txid": cold.root.position.txid, - "checksum": cold.root.position.checksum, "commit_sequence": cold.root.commit_sequence}, - "payload_digest": expected_digest, - "injected_get_delay_ms": 500, - "same_worker_baseline": same_baseline, - "comparison_cell_baseline": other_baseline, - "same_worker_during_hydration": same_wait, - "comparison_cell_during_hydration": other_wait, - "hydration_and_queries_us": hydration_and_queries.as_micros(), - "origin_requests_during_hydration": hydration_requests, - "origin_bytes_during_hydration": hydration_bytes, - "demand_samples": demand_samples, - }) - ); -} - -#[tokio::test(flavor = "multi_thread")] -async fn cancelled_hydration_releases_fetch_bytes_without_installing_pages() { - let backend = Arc::new(ThrottledStore::new( - InMemory::new(), - ThrottleConfig::default(), - )); - let armed = Arc::new(AtomicBool::new(false)); - let started = Arc::new(tokio::sync::Notify::new()); - let watching = armed.clone(); - let notify = started.clone(); - let store = Store::new(backend.clone()).with_read_request_observer(Arc::new(move |_| { - if watching.load(Ordering::Acquire) { - notify.notify_one(); - } - })); - let cold = sparse_activation(4, store, 4 << 20).await; - let pool = SqlWorkerPool::new(1, 1).unwrap(); - pool.configure_retained_capacity(1 << 20).unwrap(); - pool.activate_restored( - cold.cell, - RestoredDatabase::Paged(Box::new(cold.database)), - cold.destination, - cold.incarnation, - 1, - cold.root, - pool.reserve_activation().unwrap(), - ) - .await - .unwrap(); - let before = pool.hydration(cold.cell).await.unwrap(); - backend.config_mut(|config| config.wait_get_per_call = Duration::from_secs(3)); - armed.store(true, Ordering::Release); - let hydration = { - let pool = pool.clone(); - tokio::spawn(async move { - pool.hydrate(cold.cell, 64, Instant::now() + Duration::from_secs(10)) - .await - }) - }; - tokio::time::timeout(Duration::from_secs(2), started.notified()) - .await - .unwrap(); - let used = pool.resource_ledger().snapshot().unwrap().used; - assert!(used.retained_bytes() > 0); - assert_eq!( - used.worker_jobs(), - 0, - "fetch must release SQL worker admission" - ); - hydration.abort(); - assert!(hydration.await.unwrap_err().is_cancelled()); - assert_eq!( - pool.resource_ledger() - .snapshot() - .unwrap() - .used - .retained_bytes(), - 0 - ); - assert_eq!(pool.hydration(cold.cell).await.unwrap(), before); - let deadline = pool - .hydrate(cold.cell, 64, Instant::now() + Duration::from_millis(20)) - .await - .unwrap(); - assert!(matches!(deadline, HydrationStep::Deferred(_))); - assert_eq!(pool.hydration(cold.cell).await.unwrap(), before); - assert_eq!( - pool.resource_ledger() - .snapshot() - .unwrap() - .used - .retained_bytes(), - 0 - ); - let foreground_bytes = pool - .resource_ledger() - .try_reserve(ResourceCost::zero().with_retained_bytes(1 << 20)) - .unwrap(); - let pressure = pool - .hydrate(cold.cell, 64, Instant::now() + Duration::from_secs(1)) - .await - .unwrap(); - assert!(matches!(pressure, HydrationStep::Deferred(_))); - assert_eq!(pool.hydration(cold.cell).await.unwrap(), before); - drop(foreground_bytes); - armed.store(false, Ordering::Release); - backend.config_mut(|config| config.wait_get_per_call = Duration::ZERO); - let HydrationStep::Progress(Some(after)) = pool - .hydrate(cold.cell, 64, Instant::now() + Duration::from_secs(10)) - .await - .unwrap() - else { - panic!("hydration unexpectedly deferred") - }; - assert!(after.resolved > before.unwrap().resolved); - pool.deactivate(cold.cell).await.unwrap(); - pool.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn expired_worker_deadline_preserves_sparse_cell_for_retry() { - let fixture = sparse_activation(7, Store::new(Arc::new(InMemory::new())), 64 * 1024).await; - let pool = SqlWorkerPool::new(1, 1).unwrap(); - pool.configure_retained_capacity(1 << 20).unwrap(); - pool.activate_restored( - fixture.cell, - RestoredDatabase::Paged(Box::new(fixture.database)), - fixture.destination, - fixture.incarnation, - 1, - fixture.root, - pool.reserve_activation().unwrap(), - ) - .await - .unwrap(); - let before = pool.hydration(fixture.cell).await.unwrap().unwrap(); - let deadline = Instant::now(); - assert!(matches!( - pool.hydrate(fixture.cell, 64, deadline).await, - Ok(HydrationStep::Deferred(_)) - )); - let after = pool.hydration(fixture.cell).await.unwrap().unwrap(); - assert_eq!(after.resolved, before.resolved); - let entered = Arc::new(std::sync::atomic::AtomicBool::new(false)); - let executed = entered.clone(); - assert!(matches!( - pool.query( - fixture.cell, - 64, - SqlDeadline::new(Instant::now()), - Box::new(move |_| { - executed.store(true, Ordering::SeqCst); - Ok(Vec::new()) - }) - ) - .await, - Err(Error::Deadline) - )); - assert!(!entered.load(Ordering::SeqCst)); - let HydrationStep::Progress(Some(progress)) = pool - .hydrate(fixture.cell, 64, Instant::now() + Duration::from_secs(5)) - .await - .unwrap() - else { - panic!("hydration unexpectedly deferred") - }; - assert!(progress.resolved > before.resolved); - pool.deactivate(fixture.cell).await.unwrap(); - pool.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/src/client.rs b/crates/crab-cell-runtime/src/client.rs deleted file mode 100644 index 067f56ba4..000000000 --- a/crates/crab-cell-runtime/src/client.rs +++ /dev/null @@ -1,1069 +0,0 @@ -//! Client-side Cell envelopes and the connection that carries them to the owner. -use std::{ - collections::HashMap, - fmt, - future::Future, - marker::PhantomData, - pin::Pin, - sync::{ - Arc, - atomic::{AtomicU64, Ordering}, - }, - time::{Instant, SystemTime, UNIX_EPOCH}, -}; - -use crab_ltx::rusqlite::OptionalExtension; -use tokio::sync::Notify; -use tracing::Instrument as _; - -use crate::cell::actor::CellHandle; -use crate::cell::catalog::CatalogRole; -use crate::cell::executor::{MutationIdentity, Resolution, StoredOutcome}; -use crate::codec::{decode_wire, encode_wire}; -use crate::fleet::telemetry::{ - CellTelemetryHandle, PrimitiveOperationKind, PrimitiveOperationOutcome, -}; -use crate::identity::{CellId, CellTarget, Digest, IncarnationId, RequestId}; -use crate::primitives::workflow::{ActivityContext, ActivityExecution, ActivitySupport}; -use crate::registry::{ - Command, CommandInvocation, OperationDescriptor, Query, QueryInvocation, Registry, -}; -use crate::{Error, Result}; - -const CELL_COMMAND_TAG: u16 = 10; -const MAX_STATE_STREAM_CHUNKS: usize = 1_024; - -#[cfg(test)] -mod tests; - -mod backpressure; -mod local; -mod replica; -mod routing; -mod runtime; - -pub use replica::CellReadReplica; -pub use routing::ReplicaReadRouter; - -pub use local::command_operation_digest; -pub(crate) use local::{ - LocalCellTransport, decode_pending, encoded_command_operation_digest, local_description, - next_metadata, receipt, validate_description, -}; -use local::{decode_output, unix_time_ms, validate_minimum}; -pub use runtime::LocalCellResolver; -use runtime::{RuntimeCellTransport, RuntimeLocalResolver}; - -/// Execution policy for typed queries on a client capability. -/// -/// Commands, mutation resolution, state streams and primitive lease validation -/// always use the owner. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub enum ReadPolicy { - /// Execute FIFO on the current owner, optionally after a receipt. - #[default] - CurrentOwner, - /// Execute on an admitted snapshot, failing if no reader proves the minimum. - Replica, -} - -/// Immutable owner metadata used to fence a routed invocation. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct CellDescription { - /// Cell the description belongs to. - pub cell: CellId, - /// Incarnation that currently owns the Cell. - pub incarnation: IncarnationId, - /// Application code digest the owner installed. - pub code: Digest, - /// Schema version the owner installed. - pub schema: u32, -} - -/// Durable observation position returned with every typed result. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct Receipt { - /// Cell the observation came from. - pub cell: CellId, - /// Incarnation that produced it. - pub incarnation: IncarnationId, - /// Highest committed sequence the result observed. - pub commit_sequence: u64, -} - -/// Typed command result released only after authoritative publication. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct Committed { - /// Typed command output. - pub output: T, - /// Publication receipt the output can be observed at. - pub receipt: Receipt, -} - -/// Typed read result and the exact SQLite position it observed. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct Observed { - /// Typed read output. - pub output: T, - /// Position the read observed. - pub receipt: Receipt, -} - -/// Cancellation capability for one state-observing Cell stream. -#[derive(Clone)] -pub struct StateStreamCancellation { - cancelled: Arc, - notify: Arc, -} - -impl StateStreamCancellation { - /// Cancels the stream and releases any pending output wait. - pub fn cancel(&self) { - self.cancelled.store(true, Ordering::Release); - self.notify.notify_waiters(); - } - - /// Returns whether cancellation has been requested. - #[must_use] - pub fn is_cancelled(&self) -> bool { - self.cancelled.load(Ordering::Acquire) - } -} - -/// Serial, watermark-bound output capability for state-observing streams. -/// -/// Each call to [`Self::emit`] performs one actor-ordered query at or beyond -/// the previous receipt. The mutable borrow prevents two chunks from being -/// released out of order; the deadline and cancellation capability bound the -/// retained stream state. Dropping the stream is equivalent to cancellation. -pub struct CellStateStream { - client: CellClient, - target: CellTarget, - expected: CellDescription, - stream_id: RequestId, - minimum: Option, - deadline: Instant, - chunks: usize, - closed: bool, - cancellation: StateStreamCancellation, - marker: std::marker::PhantomData Q>, -} - -impl CellStateStream { - /// Returns the opaque stream identity used for lifecycle and telemetry. - #[must_use] - pub const fn id(&self) -> RequestId { - self.stream_id - } - - /// Returns the latest proven observation, if the stream emitted a chunk. - #[must_use] - pub const fn last_receipt(&self) -> Option { - self.minimum - } - - /// Returns a capability that cancels this stream from another task. - #[must_use] - pub fn cancellation(&self) -> StateStreamCancellation { - self.cancellation.clone() - } - - /// Emits one state-observing chunk after its watermark is proven. - pub async fn emit( - &mut self, - input: Q::Input, - ) -> std::result::Result, InvocationError> { - if self.closed || self.cancellation.is_cancelled() { - self.closed = true; - return Err(stream_error(Error::StreamCancelled)); - } - if self.chunks >= MAX_STATE_STREAM_CHUNKS { - self.closed = true; - return Err(stream_error(Error::Capacity("state stream chunks"))); - } - let remaining = self - .deadline - .checked_duration_since(Instant::now()) - .filter(|duration| !duration.is_zero()) - .ok_or_else(|| { - self.closed = true; - stream_error(Error::Deadline) - })?; - let query = self.client.query_with_description::( - &self.target, - self.expected, - self.minimum, - input, - ); - tokio::pin!(query); - let notified = self.cancellation.notify.notified(); - tokio::pin!(notified); - notified.as_mut().enable(); - if self.cancellation.is_cancelled() { - self.closed = true; - return Err(stream_error(Error::StreamCancelled)); - } - let result = match tokio::select! { - result = &mut query => result, - () = &mut notified => { - self.closed = true; - Err(stream_error(Error::StreamCancelled)) - } - () = tokio::time::sleep(remaining) => { - self.closed = true; - Err(stream_error(Error::Deadline)) - } - } { - Ok(result) => result, - Err(error) => { - self.closed = true; - return Err(error); - } - }; - if self.cancellation.is_cancelled() { - self.closed = true; - return Err(stream_error(Error::StreamCancelled)); - } - if Instant::now() >= self.deadline { - self.closed = true; - return Err(stream_error(Error::Deadline)); - } - let receipt = result.receipt; - if self - .minimum - .is_some_and(|minimum| receipt.commit_sequence < minimum.commit_sequence) - { - self.closed = true; - return Err(stream_error(Error::Command( - "state stream watermark moved backwards", - ))); - } - self.minimum = Some(receipt); - self.chunks = self.chunks.saturating_add(1); - Ok(result) - } - - /// Closes the stream. No later chunk can be emitted. - pub fn finish(&mut self) { - self.closed = true; - self.cancellation.cancel(); - } - - /// Returns whether the stream is closed or cancelled. - #[must_use] - pub fn is_closed(&self) -> bool { - self.closed || self.cancellation.is_cancelled() - } -} - -impl Drop for CellStateStream { - fn drop(&mut self) { - self.cancellation.cancel(); - } -} - -fn stream_error(error: Error) -> InvocationError { - InvocationError::NotStarted(error) -} - -/// Stable mutation evidence retained when acceptance cannot be resolved. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct PendingMutation { - target: CellTarget, - incarnation: IncarnationId, - identity: MutationIdentity, - operation_digest: Digest, - max_result_bytes: usize, -} - -impl PendingMutation { - /// Returns the identity that resolves this mutation. - #[must_use] - pub const fn identity(&self) -> MutationIdentity { - self.identity - } - - /// Returns the digest of the operation that was submitted. - #[must_use] - pub const fn operation_digest(&self) -> Digest { - self.operation_digest - } - - /// Returns the incarnation the mutation targeted. - #[must_use] - pub const fn incarnation(&self) -> IncarnationId { - self.incarnation - } - - /// Returns the Cell the mutation targeted. - #[must_use] - pub const fn target(&self) -> &CellTarget { - &self.target - } -} - -/// Typed command whose exact request evidence survives cancellation of execution. -/// -/// Clone before dispatch if the caller may be cancelled; only retry the clone -/// after resolving its evidence as absent. -#[must_use] -pub struct PreparedCommand { - client: CellClient, - request: EncodedCommand, - evidence: PendingMutation, - marker: PhantomData C>, -} - -impl Clone for PreparedCommand { - fn clone(&self) -> Self { - Self { - client: self.client.clone(), - request: self.request.clone(), - evidence: self.evidence.clone(), - marker: PhantomData, - } - } -} - -impl PreparedCommand { - /// Returns the exact request evidence to resolve after execution is cancelled or ambiguous. - #[must_use] - pub const fn evidence(&self) -> &PendingMutation { - &self.evidence - } - - /// Executes the prepared request once against its validated owner incarnation. - /// - /// Returns pending evidence when acceptance is unknown and rejects an expired identity. - pub async fn execute( - mut self, - ) -> std::result::Result, InvocationError> { - let now_ms = unix_time_ms().map_err(InvocationError::NotStarted)?; - self.evidence - .identity - .validate(now_ms) - .map_err(InvocationError::NotStarted)?; - self.request.now_ms = now_ms; - let span = tracing::debug_span!( - target: "crab_cell_runtime::action", - "cell_invocation", - cell = ?self.request.expected.cell, - incarnation = ?self.request.expected.incarnation, - mutation_request_id = ?self.request.identity.request_id, - module = C::MODULE, - operation_id = C::ID, - ); - async move { - let started = Instant::now(); - tracing::debug!(target: "crab_cell_runtime::action", event = "cell_invocation_started"); - let result = match self.client.transport.command(self.request).await { - Ok(outcome) => decode_pending::(&self.evidence, outcome), - Err(Error::OutcomeUnknown { - request_id, - operation_digest, - .. - }) if request_id == self.evidence.identity.request_id - && operation_digest == self.evidence.operation_digest => - { - Err(InvocationError::Pending(Box::new(self.evidence))) - } - Err(error) => Err(InvocationError::NotStarted(error)), - }; - let (outcome, receipt) = match &result { - Ok(committed) => ("committed", Some(committed.receipt)), - Err(InvocationError::Rejected(committed)) => ("rejected", Some(committed.receipt)), - Err(InvocationError::InvalidPublishedResult { receipt, .. }) => { - ("invalid_result", Some(*receipt)) - } - Err(InvocationError::Pending(_)) => ("pending", None), - Err(InvocationError::NotStarted(_)) => ("not_started", None), - }; - tracing::debug!( - target: "crab_cell_runtime::action", - event = "cell_invocation_completed", - outcome, - commit_sequence = receipt.map(|receipt| receipt.commit_sequence), - elapsed_us = started.elapsed().as_micros(), - ); - result - } - .instrument(span) - .await - } -} - -/// Outcome-aware typed invocation failure. -pub enum InvocationError { - /// The command committed a rejection; the committed value carries it. - Rejected(Box>), - /// The command may have committed; resolve the mutation before retrying. - Pending(Box), - /// The published result could not be decoded. - InvalidPublishedResult { - /// Receipt the published result was observed at. - receipt: Receipt, - /// Decoding failure that produced this error. - source: Box, - }, - /// The invocation failed before it was submitted. - NotStarted(Error), -} - -impl fmt::Debug for InvocationError { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Rejected(_) => formatter.write_str("InvocationError::Rejected(..)"), - Self::Pending(pending) => formatter - .debug_tuple("InvocationError::Pending") - .field(pending) - .finish(), - Self::InvalidPublishedResult { receipt, source } => formatter - .debug_struct("InvocationError::InvalidPublishedResult") - .field("receipt", receipt) - .field("source", source) - .finish(), - Self::NotStarted(error) => formatter - .debug_tuple("InvocationError::NotStarted") - .field(error) - .finish(), - } - } -} - -impl fmt::Display for InvocationError { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Rejected(_) => formatter.write_str("Cell command was durably rejected"), - Self::Pending(_) => formatter.write_str("Cell command outcome requires resolution"), - Self::InvalidPublishedResult { .. } => { - formatter.write_str("Cell command committed an invalid typed result") - } - Self::NotStarted(error) => write!(formatter, "Cell invocation did not start: {error}"), - } - } -} - -impl std::error::Error for InvocationError { - fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { - match self { - Self::InvalidPublishedResult { source, .. } => Some(source.as_ref()), - Self::NotStarted(error) => Some(error), - Self::Rejected(_) | Self::Pending(_) => None, - } - } -} - -/// Owned encoded command accepted by a local or authenticated peer transport. -#[derive(Clone)] -pub(super) struct EncodedCommand { - pub(super) target: CellTarget, - pub(super) expected: CellDescription, - pub(super) identity: MutationIdentity, - pub(super) operation_digest: Digest, - pub(super) now_ms: i64, - pub(super) module: &'static str, - pub(super) operation_id: u32, - pub(super) codec_version: u32, - pub(super) input: Vec, - pub(super) input_limit: u32, - pub(super) output_limit: u32, -} - -/// Owned encoded query accepted by a local or authenticated peer transport. -#[derive(Clone)] -pub(super) struct EncodedQuery { - pub(super) target: CellTarget, - pub(super) expected: CellDescription, - pub(super) minimum: Option, - pub(super) now_ms: i64, - pub(super) module: &'static str, - pub(super) operation_id: u32, - pub(super) codec_version: u32, - pub(super) input: Vec, - pub(super) input_limit: u32, - pub(super) output_limit: u32, -} - -/// Owned request-ledger lookup accepted by a routed transport. -#[derive(Clone)] -pub(super) struct EncodedResolve { - pub(super) target: CellTarget, - pub(super) expected: CellDescription, - pub(super) identity: MutationIdentity, - pub(super) operation_digest: Digest, - pub(super) now_ms: i64, - pub(super) max_result_bytes: usize, -} - -/// Encoded query output carrying the owner-observed commit position. -pub(super) struct EncodedObservation { - pub output: Vec, - pub receipt: Receipt, -} - -/// Internal routing boundary shared by local actors and the private peer client. -pub(super) trait CellTransport: Send + Sync + 'static { - fn describe( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>>; - - fn command( - &self, - command: EncodedCommand, - ) -> Pin> + Send + 'static>>; - - fn query( - &self, - query: EncodedQuery, - ) -> Pin> + Send + 'static>>; - - fn resolve( - &self, - resolve: EncodedResolve, - ) -> Pin> + Send + 'static>>; -} - -/// Cloneable typed application capability over one routing implementation. -#[derive(Clone)] -pub struct CellClient { - registry: Arc, - transport: Arc, - observed_description: Option, - blob_artifact_store: Option, - read_policy: ReadPolicy, - replicas: Option>, -} - -impl CellClient { - #[must_use] - fn new(registry: Arc, transport: Arc) -> Self { - Self { - registry, - transport, - observed_description: None, - blob_artifact_store: None, - read_policy: ReadPolicy::CurrentOwner, - replicas: None, - } - } - - /// Returns a capability with bounded waiting for owner capacity refusals. - /// - /// Clones share request and retained-input limits. Full client admission - /// still fails immediately. Mailbox work queues in FIFO order per Cell; - /// Describe uses shared client bounds only. Cells remain independent. - /// Only capacity refusals are retried; ambiguous commands require resolution. - /// The wait bound never cancels an accepted attempt. Replica reads retain - /// separate admission. - pub fn with_admission_backpressure( - &self, - requests: usize, - bytes: usize, - max_wait: std::time::Duration, - ) -> Result { - let mut client = self.clone(); - client.transport = Arc::new(backpressure::BackpressureTransport::new( - self.transport.clone(), - requests, - bytes, - max_wait, - )?); - Ok(client) - } - - /// Wires replica placement and authenticated execution at the host boundary. - /// - /// The peer registry must match this client's compiled application. Share - /// the router across callers; optionally supply this node's admitted local - /// snapshot resolver to avoid self-dials. Query policy remains unchanged. - pub fn with_read_replicas( - &self, - router: ReplicaReadRouter, - peer: crate::peer::ReplicaPeerClient, - local: Option<(crate::SessionId, Arc)>, - ) -> Result { - if peer.registry().release_digest() != self.registry.release_digest() { - return Err(Error::Registry( - "replica client registry differs from owner client", - )); - } - let mut client = self.clone(); - client.replicas = Some(Arc::new(routing::ReplicaClient { - router, - peer, - local, - })); - Ok(client) - } - - /// Returns a capability using the explicit policy for typed queries only. - /// - /// Replica queries without host wiring fail with `ReplicaUnavailable`; - /// they never fall back to the owner. Commands, streams and primitive lease - /// validation retain owner order. - #[must_use] - pub fn with_read_policy(&self, policy: ReadPolicy) -> Self { - let mut client = self.clone(); - client.read_policy = policy; - client - } - - /// Returns the compiled registry identity used to encode typed calls. - #[must_use] - pub fn registry_digest(&self) -> Digest { - self.registry.release_digest() - } - - /// Binds this client to an already observed Cell contract for a routed request. - /// - /// The host supplies a description read from authority or the owner. Calls - /// skip Describe and must target this exact Cell. Receiver authorization, - /// contract validation, and actor admission still apply; stale observations - /// refuse execution and require the host to resolve a fresh route. - #[must_use] - pub fn with_observed_description(mut self, description: CellDescription) -> Self { - self.observed_description = Some(description); - self - } - - /// Returns a client clone wired to the configured object-store Blob data. - #[must_use] - pub fn with_blob_artifact_store(&self, store: crate::BlobArtifactStore) -> Self { - let mut client = self.clone(); - client.blob_artifact_store = Some(store); - client - } - - pub(crate) fn blob_artifact_store(&self) -> Option { - self.blob_artifact_store.clone() - } - - /// Builds the canonical single-owner transport used by embedded routes. - #[must_use] - pub fn local(registry: Arc, handle: CellHandle) -> Self { - Self::local_with_telemetry(registry, handle, CellTelemetryHandle::default()) - } - - /// Builds the canonical single-owner transport with primitive-operation - /// telemetry reported from the executing thread. - #[must_use] - pub fn local_with_telemetry( - registry: Arc, - handle: CellHandle, - telemetry: CellTelemetryHandle, - ) -> Self { - let primary = handle.clone(); - let transport = Arc::new(LocalCellTransport { - registry: registry.clone(), - handles: Arc::new(HashMap::from([(handle.cell_id(), handle)])), - handle: primary, - telemetry, - }); - Self::new(registry, transport) - } - - /// Builds a bounded in-process transport for a set of locally owned Cells. - /// - /// This is the qualification and single-process embedding path. Production - /// multi-node routing still uses [`Self::peer`], while every target remains - /// checked against the exact local Cell identity before execution. - pub fn local_many( - registry: Arc, - handles: impl IntoIterator, - ) -> Result { - Self::local_many_with_telemetry(registry, handles, CellTelemetryHandle::default()) - } - - /// Builds a bounded in-process transport with primitive-operation telemetry - /// reported from the executing thread. - pub fn local_many_with_telemetry( - registry: Arc, - handles: impl IntoIterator, - telemetry: CellTelemetryHandle, - ) -> Result { - let mut local = HashMap::new(); - for handle in handles { - if local.insert(handle.cell_id(), handle).is_some() { - return Err(Error::Identity("duplicate local Cell handle")); - } - } - if local.is_empty() { - return Err(Error::Identity("local transport has no Cell handles")); - } - let primary = local - .values() - .next() - .cloned() - .ok_or(Error::Identity("local transport has no Cell handles"))?; - Ok(Self::new( - registry.clone(), - Arc::new(LocalCellTransport { - registry, - handles: Arc::new(local), - handle: primary, - telemetry, - }), - )) - } - - /// Routes to any Cell currently owned by this local runtime. - /// - /// The catalog and authority are checked for each invocation. This does - /// not acquire an idle Cell or forward to another node; callers must - /// arrange ownership before sending an operation. - #[must_use] - pub fn local_runtime( - registry: Arc, - runtime: crate::cell::actor::CellRuntime, - layout: crate::ltx::CellStorageLayout, - ) -> Self { - let transport = Arc::new(RuntimeCellTransport::new(registry.clone(), runtime, layout)); - Self::new(registry, transport) - } - - /// Routes a target to its current local owner or an authenticated peer. - /// - /// Every invocation rechecks catalog and authority state. The peer round - /// trip must resolve the current remote owner and verify its enrollment; - /// this constructor does not acquire an idle Cell. - #[must_use] - pub fn runtime_with_peer( - registry: Arc, - runtime: crate::cell::actor::CellRuntime, - layout: crate::ltx::CellStorageLayout, - signer: Arc, - principal: crate::peer::PeerPrincipal, - round_trip: Arc, - ) -> Self { - Self::peer(registry, signer, principal, round_trip) - .with_local_resolver(Arc::new(RuntimeLocalResolver { runtime, layout })) - } - - /// Resolves a local owner before delegating to this client's transport. - /// - /// The resolver owns product placement and admission policy. It runs before - /// describe, command, query, and resolution; errors never dispatch remotely. - /// Configure admission backpressure afterward so it bounds both routes. - #[must_use] - pub fn with_local_resolver(mut self, resolver: Arc) -> Self { - self.transport = Arc::new(RuntimeCellTransport::with_resolver( - self.registry.clone(), - resolver, - self.transport, - )); - self - } - - /// Builds a typed capability over authenticated private peer routing. - #[must_use] - pub fn peer( - registry: Arc, - signer: Arc, - principal: crate::peer::PeerPrincipal, - round_trip: Arc, - ) -> Self { - let transport = Arc::new(crate::peer::PeerClientTransport::new( - signer, principal, round_trip, - )); - Self::new(registry, transport) - } - - /// Opens a bounded, serial state-observing stream for one Cell target. - /// - /// The first emitted chunk establishes the response watermark. Every later - /// chunk is queried at or beyond the prior receipt and is therefore gated - /// by the same actor durability proof before it can be returned. - #[must_use = "await the stream setup result"] - pub async fn open_state_stream( - &self, - target: &CellTarget, - deadline: Instant, - ) -> std::result::Result, InvocationError> { - let remaining = deadline - .checked_duration_since(Instant::now()) - .filter(|duration| !duration.is_zero()) - .ok_or_else(|| InvocationError::NotStarted(Error::Deadline))?; - let expected = tokio::time::timeout(remaining, self.describe::(target)) - .await - .map_err(|_| InvocationError::NotStarted(Error::Deadline))??; - Ok(CellStateStream { - client: self.clone(), - target: target.clone(), - expected, - stream_id: RequestId::from_bytes(rand::random()), - minimum: None, - deadline, - chunks: 0, - closed: false, - cancellation: StateStreamCancellation { - cancelled: Arc::new(std::sync::atomic::AtomicBool::new(false)), - notify: Arc::new(Notify::new()), - }, - marker: std::marker::PhantomData, - }) - } - - pub(crate) fn require_namespace( - &self, - namespace: crate::NamespaceId, - module: &'static str, - role: CatalogRole, - ) -> Result { - let Some((owner, descriptor)) = self.registry.namespace_contract(namespace) else { - return Err(Error::Registry("namespace is not registered")); - }; - if owner != module || descriptor.role != role { - return Err(Error::Registry( - "namespace module or role does not match primitive", - )); - } - Ok(descriptor.shards) - } - - /// Builds a typed source-effect capability after validating the compiled - /// module and its source namespace. This keeps effect operation IDs inside - /// the registry instead of allowing callers to construct arbitrary ones. - pub fn effect_source( - &self, - target: CellTarget, - ) -> Result> { - let Some((module, _)) = self.registry.namespace_contract(target.namespace()) else { - return Err(Error::Registry("effect source namespace is not registered")); - }; - if module != M::MODULE || !self.registry.has_effect_runner(target.namespace()) { - return Err(Error::Registry("effect source module is not registered")); - } - Ok(crate::EffectSource::new(self.clone(), target)) - } - - pub(crate) fn activity_support( - &self, - module: &'static str, - definition: Digest, - ) -> Result> { - self.registry.activity_support(module, definition) - } - - pub(crate) async fn execute_activity( - &self, - module: &'static str, - definition: Digest, - activity: &str, - context: ActivityContext, - input: Vec, - blocking: Option, - ) -> Result { - self.registry - .execute_activity(module, definition, activity, context, input, blocking) - .await - } - - /// Executes one typed command with a digest derived from validated values. - pub async fn command( - &self, - target: &CellTarget, - identity: MutationIdentity, - input: C::Input, - ) -> std::result::Result, InvocationError> { - self.prepare_command::(target, identity, input) - .await? - .execute() - .await - } - - /// Prepares one exact typed command so its evidence survives cancellation during dispatch. - /// - /// Validates the owner contract and bounded input without dispatching a mutation. - pub async fn prepare_command( - &self, - target: &CellTarget, - identity: MutationIdentity, - input: C::Input, - ) -> std::result::Result, InvocationError> { - let started = Instant::now(); - let now_ms = unix_time_ms().map_err(InvocationError::NotStarted)?; - identity - .validate(now_ms) - .map_err(InvocationError::NotStarted)?; - let description = self.describe::(target).await?; - let operation = self - .registry - .command_contract::(target.namespace()) - .map_err(InvocationError::NotStarted)?; - validate_description(&self.registry, C::MODULE, description, operation) - .map_err(InvocationError::NotStarted)?; - let input = encode_wire(&input, operation.input_limit) - .map_err(Error::from) - .map_err(InvocationError::NotStarted)?; - let digest = command_operation_digest::(description, identity, &input) - .map_err(InvocationError::NotStarted)?; - let request = EncodedCommand { - target: target.clone(), - expected: description, - identity, - operation_digest: digest, - now_ms, - module: C::MODULE, - operation_id: C::ID, - codec_version: C::CODEC_VERSION, - input, - input_limit: operation.input_limit, - output_limit: operation.output_limit, - }; - tracing::debug!( - target: "crab_cell_runtime::action", - event = "cell_command_prepared", - cell = ?description.cell, - incarnation = ?description.incarnation, - mutation_request_id = ?identity.request_id, - module = C::MODULE, - operation_id = C::ID, - elapsed_us = started.elapsed().as_micros(), - ); - Ok(PreparedCommand { - client: self.clone(), - request, - evidence: PendingMutation { - target: target.clone(), - incarnation: description.incarnation, - identity, - operation_digest: digest, - max_result_bytes: operation.output_limit as usize, - }, - marker: PhantomData, - }) - } - - /// Runs a typed query under this capability's policy at or beyond a receipt. - /// - /// The default policy executes FIFO on the owner; explicit replica reads - /// return their actual snapshot position or fail closed. - pub async fn query( - &self, - target: &CellTarget, - minimum: Option, - input: Q::Input, - ) -> std::result::Result, InvocationError> { - if self.read_policy == ReadPolicy::Replica { - let replicas = self - .replicas - .as_ref() - .ok_or(InvocationError::NotStarted(Error::ReplicaUnavailable))?; - let local = replicas - .local - .as_ref() - .map(|(session, resolver)| (*session, resolver.as_ref())); - // Placement and replica admission carry large I/O futures. Keep - // that optional state off every caller's owner-query stack frame. - return Box::pin(replicas.router.query::( - &replicas.peer, - local, - target, - minimum, - input, - )) - .await - .map(|(observed, _)| observed) - .map_err(InvocationError::NotStarted); - } - let description = self.describe::(target).await?; - self.query_with_description::(target, description, minimum, input) - .await - } - - async fn query_with_description( - &self, - target: &CellTarget, - description: CellDescription, - minimum: Option, - input: Q::Input, - ) -> std::result::Result, InvocationError> { - let now_ms = unix_time_ms().map_err(InvocationError::NotStarted)?; - let operation = self - .registry - .query_contract::(target.namespace()) - .map_err(InvocationError::NotStarted)?; - validate_description(&self.registry, Q::MODULE, description, operation) - .map_err(InvocationError::NotStarted)?; - validate_minimum(description, minimum).map_err(InvocationError::NotStarted)?; - let input = encode_wire(&input, operation.input_limit) - .map_err(Error::from) - .map_err(InvocationError::NotStarted)?; - let observation = self - .transport - .query(EncodedQuery { - target: target.clone(), - expected: description, - minimum, - now_ms, - module: Q::MODULE, - operation_id: Q::ID, - codec_version: Q::CODEC_VERSION, - input, - input_limit: operation.input_limit, - output_limit: operation.output_limit, - }) - .await - .map_err(InvocationError::NotStarted)?; - if observation.receipt.cell != description.cell - || observation.receipt.incarnation != description.incarnation - || minimum.is_some_and(|minimum| { - observation.receipt.commit_sequence < minimum.commit_sequence - }) - { - return Err(InvocationError::NotStarted(Error::Command( - "query did not satisfy minimum receipt", - ))); - } - Ok(Observed { - output: decode_output(&observation.output, operation.output_limit)?, - receipt: observation.receipt, - }) - } - - /// Resolves a pending mutation without executing its handler again. - pub async fn resolve( - &self, - pending: &PendingMutation, - ) -> std::result::Result>> { - let description = self.describe::>(pending.target()).await?; - if description.incarnation != pending.incarnation { - return Err(InvocationError::NotStarted(Error::Command( - "pending mutation incarnation changed", - ))); - } - let now_ms = unix_time_ms().map_err(InvocationError::NotStarted)?; - self.transport - .resolve(EncodedResolve { - target: pending.target.clone(), - expected: description, - identity: pending.identity, - operation_digest: pending.operation_digest, - now_ms, - max_result_bytes: pending.max_result_bytes, - }) - .await - .map_err(InvocationError::NotStarted) - } - - async fn describe( - &self, - target: &CellTarget, - ) -> std::result::Result> { - let description = match self.observed_description { - Some(description) => description, - None => self - .transport - .describe(target.clone()) - .await - .map_err(InvocationError::NotStarted)?, - }; - if description.cell != target.cell_id() { - return Err(InvocationError::NotStarted(Error::Command( - "transport described a different Cell", - ))); - } - Ok(description) - } -} diff --git a/crates/crab-cell-runtime/src/client/backpressure.rs b/crates/crab-cell-runtime/src/client/backpressure.rs deleted file mode 100644 index b955c5f2a..000000000 --- a/crates/crab-cell-runtime/src/client/backpressure.rs +++ /dev/null @@ -1,247 +0,0 @@ -//! Bounded pacing for callers that can wait for owner admission. - -use std::{ - collections::HashMap, - future::Future, - pin::Pin, - sync::{Arc, Mutex, Weak}, - time::Duration, -}; - -use tokio::sync::Semaphore; - -use super::{ - CellDescription, CellTarget, CellTransport, EncodedCommand, EncodedObservation, EncodedQuery, - EncodedResolve, Resolution, StoredOutcome, unix_time_ms, -}; -use crate::cell::actor::{CELL_BYTES, CELL_REQUESTS}; -use crate::identity::CellId; -use crate::{Error, Result}; - -struct CellAdmission { - requests: Semaphore, - bytes: Semaphore, -} - -#[derive(Clone)] -pub(super) struct BackpressureTransport { - inner: Arc, - requests: Arc, - bytes: Arc, - cells: Arc>>>, - max_wait: Duration, -} - -impl BackpressureTransport { - pub(super) fn new( - inner: Arc, - requests: usize, - bytes: usize, - max_wait: Duration, - ) -> Result { - if requests == 0 - || requests > Semaphore::MAX_PERMITS - || bytes == 0 - || bytes > Semaphore::MAX_PERMITS.min(u32::MAX as usize) - || max_wait.is_zero() - { - return Err(Error::Command("invalid client admission bounds")); - } - Ok(Self { - inner, - requests: Arc::new(Semaphore::new(requests)), - bytes: Arc::new(Semaphore::new(bytes)), - cells: Arc::default(), - max_wait, - }) - } - - pub(super) async fn invoke>>( - &self, - cell: CellId, - input_bytes: usize, - output_bytes: Option, - mut attempt: impl FnMut() -> F, - ) -> Result { - let _request = self - .requests - .try_acquire() - .map_err(|_| Error::Capacity("client admission requests"))?; - // Retain the original envelope plus one attempted copy. Owner result - // reservations remain in the runtime; this bounds waiting input memory. - let bytes = input_bytes - .checked_mul(2) - .and_then(|bytes| bytes.checked_add(2_048)) - .and_then(|bytes| u32::try_from(bytes).ok()) - .ok_or(Error::Capacity("client admission bytes"))?; - let _bytes = self - .bytes - .try_acquire_many(bytes) - .map_err(|_| Error::Capacity("client admission bytes"))?; - let started = tokio::time::Instant::now(); - let cell = if let Some(output_bytes) = output_bytes { - let reservation = input_bytes - .checked_add(output_bytes) - .filter(|bytes| *bytes <= CELL_BYTES) - .and_then(|bytes| u32::try_from(bytes.max(1)).ok()) - .ok_or(Error::Capacity("client Cell admission bytes"))?; - let gate = { - let mut cells = self - .cells - .lock() - .map_err(|_| Error::Control("client admission lock poisoned"))?; - // Only admitted calls retain a gate. Pruning weak entries bounds - // this map by the shared call budget even as table IDs change. - cells.retain(|_, gate| gate.strong_count() != 0); - match cells.get(&cell).and_then(Weak::upgrade) { - Some(gate) => gate, - None => { - let gate = Arc::new(CellAdmission { - requests: Semaphore::new(CELL_REQUESTS), - bytes: Semaphore::new(CELL_BYTES), - }); - cells.insert(cell, Arc::downgrade(&gate)); - gate - } - } - }; - Some((gate, reservation)) - } else { - // Describe does not enter the owner mailbox. Queuing it behind - // writes would add an unrelated wait before every prepared command. - None - }; - // Match the owner's existing limits while allowing routing and reads - // to overlap. FIFO weighted permits keep large calls from starvation. - let _cell = if let Some((cell, reservation)) = &cell { - Some( - tokio::time::timeout(self.max_wait, async { - let requests = cell - .requests - .acquire() - .await - .map_err(|_| Error::RuntimeClosed)?; - let bytes = cell - .bytes - .acquire_many(*reservation) - .await - .map_err(|_| Error::RuntimeClosed)?; - Ok::<_, Error>((requests, bytes)) - }) - .await - .map_err(|_| Error::Capacity("client Cell admission wait"))??, - ) - } else { - None - }; - let mut delay = Duration::from_millis(10); - loop { - let result = attempt().await; - let Err(Error::Capacity(_)) = &result else { - return result; - }; - let Some(remaining) = self.max_wait.checked_sub(started.elapsed()) else { - return result; - }; - if remaining.is_zero() { - return result; - } - // Never cancel an accepted command to enforce an admission timeout. - // Unknown outcomes pass through unchanged and must be resolved. - tokio::time::sleep(delay.min(remaining)).await; - if started.elapsed() >= self.max_wait { - return result; - } - delay = (delay * 2).min(Duration::from_millis(100)); - } - } -} - -impl CellTransport for BackpressureTransport { - fn describe( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>> { - let transport = self.clone(); - Box::pin(async move { - transport - .invoke(target.cell_id(), target.partition().len(), None, || { - transport.inner.describe(target.clone()) - }) - .await - }) - } - - fn command( - &self, - command: EncodedCommand, - ) -> Pin> + Send + 'static>> { - let transport = self.clone(); - Box::pin(async move { - transport - .invoke( - command.target.cell_id(), - command.input.len(), - Some(command.output_limit as usize), - || { - let mut request = command.clone(); - let inner = Arc::clone(&transport.inner); - async move { - request.now_ms = unix_time_ms()?; - request.identity.validate(request.now_ms)?; - inner.command(request).await - } - }, - ) - .await - }) - } - - fn query( - &self, - query: EncodedQuery, - ) -> Pin> + Send + 'static>> { - let transport = self.clone(); - Box::pin(async move { - transport - .invoke( - query.target.cell_id(), - query.input.len(), - Some(query.output_limit as usize), - || { - let mut request = query.clone(); - let inner = Arc::clone(&transport.inner); - async move { - request.now_ms = unix_time_ms()?; - inner.query(request).await - } - }, - ) - .await - }) - } - - fn resolve( - &self, - resolve: EncodedResolve, - ) -> Pin> + Send + 'static>> { - let transport = self.clone(); - Box::pin(async move { - transport - .invoke( - resolve.target.cell_id(), - 48, - Some(resolve.max_result_bytes), - || { - let mut request = resolve.clone(); - let inner = Arc::clone(&transport.inner); - async move { - request.now_ms = unix_time_ms()?; - inner.resolve(request).await - } - }, - ) - .await - }) - } -} diff --git a/crates/crab-cell-runtime/src/client/local.rs b/crates/crab-cell-runtime/src/client/local.rs deleted file mode 100644 index 6c501ee7a..000000000 --- a/crates/crab-cell-runtime/src/client/local.rs +++ /dev/null @@ -1,367 +0,0 @@ -//! The in-process transport: a Cell client served by a handle in this process. -//! -//! Every call re-checks the receipt the caller expects against the handle's -//! current description before it touches the registry or the executor. - -use super::*; - -pub(crate) struct LocalCellTransport { - pub(crate) registry: Arc, - pub(crate) handles: Arc>, - pub(crate) handle: CellHandle, - pub(crate) telemetry: CellTelemetryHandle, -} - -impl CellTransport for LocalCellTransport { - fn describe( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>> { - let handles = Arc::clone(&self.handles); - Box::pin(async move { - let handle = local_handle(&handles, &target)?; - validate_local_target(&handle, &target)?; - Ok(local_description(&handle)) - }) - } - - fn command( - &self, - command: EncodedCommand, - ) -> Pin> + Send + 'static>> { - let registry = self.registry.clone(); - let handles = Arc::clone(&self.handles); - let telemetry = self.telemetry.clone(); - Box::pin(async move { - let handle = local_handle(&handles, &command.target)?; - validate_local_target(&handle, &command.target)?; - validate_expected(&handle, command.expected)?; - if command.input.len() > command.input_limit as usize { - return Err(Error::Command("encoded command input exceeds limit")); - } - let schema = handle.schema(); - let input_bytes = command.input.len(); - let output_limit = command.output_limit as usize; - handle - .execute( - command.identity, - command.operation_digest, - command.now_ms, - input_bytes, - output_limit, - move |transaction| { - let (sequence, now_ms) = next_metadata(transaction, command.now_ms)?; - let started = Instant::now(); - let result = registry.execute_command_with_issue_time( - transaction, - CommandInvocation { - module: command.module, - operation_id: command.operation_id, - codec_version: command.codec_version, - schema, - target: command.target.clone(), - sequence, - now_ms, - input: &command.input, - }, - command.identity.issued_at_ms, - ); - telemetry.primitive_operation( - command.module, - PrimitiveOperationKind::Command, - PrimitiveOperationOutcome::from(&result), - started.elapsed(), - ); - result - }, - ) - .await - }) - } - - fn query( - &self, - query: EncodedQuery, - ) -> Pin> + Send + 'static>> { - let registry = self.registry.clone(); - let handles = Arc::clone(&self.handles); - let telemetry = self.telemetry.clone(); - Box::pin(async move { - let handle = local_handle(&handles, &query.target)?; - validate_local_target(&handle, &query.target)?; - validate_expected(&handle, query.expected)?; - validate_minimum(query.expected, query.minimum)?; - if query.input.len() > query.input_limit as usize { - return Err(Error::Command("encoded query input exceeds limit")); - } - let sequence = Arc::new(AtomicU64::new(0)); - let observed_sequence = sequence.clone(); - let cell = handle.cell_id(); - let schema = handle.schema(); - let input_bytes = query.input.len(); - let output_limit = query.output_limit as usize; - let output = handle - .query(input_bytes, output_limit, move |connection| { - let (commit_sequence, now_ms) = current_metadata(connection, query.now_ms)?; - observed_sequence.store(commit_sequence, Ordering::Release); - let started = Instant::now(); - let result = registry.execute_query( - connection, - QueryInvocation { - module: query.module, - operation_id: query.operation_id, - codec_version: query.codec_version, - schema, - cell, - commit_sequence, - now_ms, - input: &query.input, - }, - ); - telemetry.primitive_operation( - query.module, - PrimitiveOperationKind::Query, - match &result { - Ok(_) => PrimitiveOperationOutcome::Success, - Err(_) => PrimitiveOperationOutcome::Failed, - }, - started.elapsed(), - ); - result - }) - .await?; - let commit_sequence = sequence.load(Ordering::Acquire); - if query - .minimum - .is_some_and(|minimum| commit_sequence < minimum.commit_sequence) - { - return Err(Error::Command("query did not satisfy minimum receipt")); - } - Ok(EncodedObservation { - output, - receipt: receipt(query.expected, commit_sequence), - }) - }) - } - - fn resolve( - &self, - resolve: EncodedResolve, - ) -> Pin> + Send + 'static>> { - let handles = Arc::clone(&self.handles); - Box::pin(async move { - let handle = local_handle(&handles, &resolve.target)?; - validate_local_target(&handle, &resolve.target)?; - validate_expected(&handle, resolve.expected)?; - handle - .resolve( - resolve.identity, - resolve.operation_digest, - resolve.now_ms, - resolve.max_result_bytes, - ) - .await - }) - } -} - -fn local_handle(handles: &HashMap, target: &CellTarget) -> Result { - handles - .get(&target.cell_id()) - .cloned() - .ok_or(Error::Control("target Cell is not locally owned")) -} - -/// Computes a typed command digest from validated metadata and encoded input. -pub fn command_operation_digest( - description: CellDescription, - identity: MutationIdentity, - input: &[u8], -) -> Result { - encoded_command_operation_digest(description, identity, C::ID, C::CODEC_VERSION, input) -} - -pub(crate) fn encoded_command_operation_digest( - description: CellDescription, - identity: MutationIdentity, - operation_id: u32, - codec_version: u32, - input: &[u8], -) -> Result { - let input_len = u32::try_from(input.len()) - .map_err(|_| Error::Command("command input exceeds canonical digest range"))?; - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.op.v1\0"); - hasher.update(description.cell.as_bytes()); - hasher.update(description.incarnation.as_bytes()); - hasher.update(identity.request_id.as_bytes()); - hasher.update(&identity.issued_at_ms.to_be_bytes()); - hasher.update(&identity.expires_at_ms.to_be_bytes()); - hasher.update(&CELL_COMMAND_TAG.to_be_bytes()); - hasher.update(&operation_id.to_be_bytes()); - hasher.update(&codec_version.to_be_bytes()); - hasher.update(&input_len.to_be_bytes()); - hasher.update(input); - Ok(Digest::from_bytes(*hasher.finalize().as_bytes())) -} - -pub(super) fn decode_output( - result: &[u8], - limit: u32, -) -> std::result::Result> { - decode_wire(result, limit) - .map_err(Error::from) - .map_err(InvocationError::NotStarted) -} - -fn decode_committed( - result: &[u8], - limit: u32, - receipt: Receipt, -) -> std::result::Result, InvocationError> { - let output = - decode_wire(result, limit).map_err(|error| InvocationError::InvalidPublishedResult { - receipt, - source: Box::new(Error::from(error)), - })?; - Ok(Committed { output, receipt }) -} - -pub(crate) fn decode_pending( - pending: &PendingMutation, - outcome: StoredOutcome, -) -> std::result::Result, InvocationError> { - let limit = u32::try_from(pending.max_result_bytes).map_err(|_| { - InvocationError::NotStarted(Error::Command("pending result limit overflow")) - })?; - let decode = |result: Vec, commit_sequence| { - decode_committed( - &result, - limit, - Receipt { - cell: pending.target.cell_id(), - incarnation: pending.incarnation, - commit_sequence, - }, - ) - }; - match outcome { - StoredOutcome::Success { - result, - commit_sequence, - } => decode(result, commit_sequence), - StoredOutcome::Rejected { - result, - commit_sequence, - } => Err(InvocationError::Rejected(Box::new(decode( - result, - commit_sequence, - )?))), - } -} - -pub(crate) fn validate_description( - registry: &Registry, - module: &str, - description: CellDescription, - operation: OperationDescriptor, -) -> Result<()> { - if !registry.supports_module_code(module, description.code, description.schema) { - return Err(Error::Command("Cell code does not match operation module")); - } - if !(operation.schema_min..=operation.schema_max).contains(&description.schema) { - return Err(Error::Command( - "registered operation does not support Cell schema", - )); - } - Ok(()) -} - -pub(super) fn validate_minimum( - description: CellDescription, - minimum: Option, -) -> Result<()> { - if minimum.is_some_and(|minimum| { - minimum.cell != description.cell || minimum.incarnation != description.incarnation - }) { - return Err(Error::Command("minimum receipt does not match Cell")); - } - Ok(()) -} - -fn validate_local_target(handle: &CellHandle, target: &CellTarget) -> Result<()> { - let entry = handle.catalog().entry(); - if target.cell_id() != handle.cell_id() - || target.namespace() != entry.namespace() - || target.partition() != entry.partition() - { - return Err(Error::Command("target does not match active Cell")); - } - Ok(()) -} - -fn validate_expected(handle: &CellHandle, expected: CellDescription) -> Result<()> { - if local_description(handle) != expected { - return Err(Error::Fenced); - } - Ok(()) -} - -pub(crate) fn local_description(handle: &CellHandle) -> CellDescription { - CellDescription { - cell: handle.cell_id(), - incarnation: handle.incarnation(), - code: handle.code(), - schema: handle.schema(), - } -} - -pub(crate) fn next_metadata( - transaction: &crab_ltx::rusqlite::Transaction<'_>, - now_ms: i64, -) -> Result<(u64, i64)> { - let (sequence, now_ms) = current_metadata(transaction, now_ms)?; - let sequence = sequence - .checked_add(1) - .filter(|sequence| *sequence <= i64::MAX as u64) - .ok_or(Error::Command("commit sequence overflow"))?; - Ok((sequence, now_ms)) -} - -pub(crate) fn current_metadata( - connection: &crab_ltx::rusqlite::Connection, - now_ms: i64, -) -> Result<(u64, i64)> { - if now_ms < 0 { - return Err(Error::Command("invalid runtime time")); - } - let (sequence, logical_time_ms) = connection - .query_row( - "SELECT commit_sequence, logical_time_ms FROM sys_meta WHERE singleton = 1", - [], - |row| Ok((row.get::<_, i64>(0)?, row.get::<_, i64>(1)?)), - ) - .optional()? - .ok_or(Error::Command("runtime metadata row missing"))?; - let sequence = - u64::try_from(sequence).map_err(|_| Error::Command("invalid commit sequence"))?; - // A queued request, clock rollback or owner change can carry an earlier - // sample. All handlers must observe at least this snapshot's committed - // time, or due work disappears and expired values become visible again. - Ok((sequence, now_ms.max(logical_time_ms))) -} - -pub(crate) fn receipt(description: CellDescription, commit_sequence: u64) -> Receipt { - Receipt { - cell: description.cell, - incarnation: description.incarnation, - commit_sequence, - } -} - -pub(super) fn unix_time_ms() -> Result { - let duration = SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_err(|_| Error::Command("system clock precedes Unix epoch"))?; - i64::try_from(duration.as_millis()).map_err(|_| Error::Command("system clock overflow")) -} diff --git a/crates/crab-cell-runtime/src/client/replica.rs b/crates/crab-cell-runtime/src/client/replica.rs deleted file mode 100644 index d9df2d2e7..000000000 --- a/crates/crab-cell-runtime/src/client/replica.rs +++ /dev/null @@ -1,390 +0,0 @@ -//! Explicit snapshot read path, gated against current Cell authority. - -use std::{ - path::Path, - sync::Arc, - time::{Duration, Instant}, -}; - -use crab_ltx::{CellReplica, ReadOnlyRoot}; -use tokio::sync::{Mutex, RwLock, Semaphore}; - -use super::*; -use crate::cell::actor::CellRuntime; -use crate::control::authority::CellAuthority; -use crate::control::{Control, ControlState, Owner}; -use crate::fleet::resource::ResourceReservation; -use crate::node::NodeDirectory; - -const QUERY_DEADLINE: Duration = Duration::from_secs(5); - -/// One immutable replica snapshot that serves explicit, position-tagged reads. -/// -/// The caller owns routing and authorization. Every successful -/// query checks authoritative control and the owner's live session after SQL -/// execution; a stale epoch or unavailable authority releases no output. -#[derive(Clone)] -pub struct CellReadReplica { - runtime: CellRuntime, - registry: Arc, - authority: CellAuthority, - directory: NodeDirectory, - replica: CellReplica, - target: CellTarget, - expected: CellDescription, - snapshot: Arc>, - refresh_gate: Arc>, - query_gate: Arc, -} - -#[derive(Clone)] -struct ReplicaSnapshot { - owner: Owner, - epoch: u64, - view: Arc, - _admission: Arc, -} - -impl CellReadReplica { - pub(crate) fn description(&self) -> CellDescription { - self.expected - } - - /// Opens the exact S3 root currently named by one live serving owner. - /// - /// The caller supplies a fresh private destination and an admitting node - /// runtime. Source Cell control must stay serving under the same owner - /// epoch through installation. - pub async fn open( - runtime: CellRuntime, - registry: Arc, - authority: CellAuthority, - directory: NodeDirectory, - replica: CellReplica, - target: CellTarget, - destination: &Path, - ) -> Result { - let admission = Arc::new(runtime.reserve_read_view()?); - let replica = runtime.replica_for_read(replica); - let cell = target.cell_id(); - let observed = authority.load(cell).await?.ok_or(Error::CellNotActive)?; - let control = observed.value(); - let owner = control.owner.as_ref().ok_or(Error::Fenced)?.clone(); - if control.state != ControlState::Serving || control.recovery.is_some() { - return Err(Error::Fenced); - } - let (module, _) = registry - .namespace_contract(target.namespace()) - .ok_or(Error::Registry("replica namespace is not registered"))?; - if !registry.supports_module_code(module, control.code, control.schema) { - return Err(Error::Registry("replica module or code is unsupported")); - } - if !directory.is_live(owner.session, unix_time_ms()?).await? { - return Err(Error::Fenced); - } - let root = control.ltx_root().ok_or(Error::Fenced)?; - let verified = replica.open_root(&root).await?; - if verified.schema() != control.schema { - return Err(Error::Fenced); - } - let expected = CellDescription { - cell, - incarnation: control.incarnation, - code: control.code, - schema: control.schema, - }; - let view = open_view(&runtime, verified, destination, Arc::clone(&admission)).await?; - let snapshot = ReplicaSnapshot { - owner, - epoch: control.epoch, - view, - _admission: admission, - }; - let opened = Self { - runtime, - registry, - authority, - directory, - replica, - target, - expected, - snapshot: Arc::new(RwLock::new(snapshot.clone())), - refresh_gate: Arc::new(Mutex::new(())), - query_gate: Arc::new(Semaphore::new(1)), - }; - opened.confirm_authority(&snapshot).await?; - Ok(opened) - } - - /// Returns the exact snapshot position this reader serves. - #[must_use] - pub async fn receipt(&self) -> Receipt { - let snapshot = self.snapshot.read().await; - self.snapshot_receipt(&snapshot) - } - - /// Returns the verified position and whether the original owner is still live. - /// - /// A false readiness bit is advisory warm state only; it never permits a - /// query or takeover. Changed authority or closed admission rejects it. - pub async fn readiness(&self) -> Result<(Receipt, bool)> { - self.runtime.ensure_running()?; - let snapshot = self.snapshot.read().await.clone(); - self.confirm_snapshot(&snapshot).await?; - let live = self - .directory - .is_live(snapshot.owner.session, unix_time_ms()?) - .await?; - Ok((self.snapshot_receipt(&snapshot), live)) - } - - /// Closes reader admission across every clone before eviction or writable activation. - pub fn close(&self) { - self.query_gate.close(); - } - - /// Installs a newer exact root without disrupting queries using the old view. - /// - /// The destination must be fresh and private. Concurrent refreshes are - /// serialized; a failed or stale refresh leaves the serving view intact. - pub async fn refresh(&self, destination: &Path) -> Result { - self.runtime.ensure_running()?; - let _refresh = self.refresh_gate.lock().await; - let current = self.snapshot.read().await.clone(); - self.confirm_authority(¤t).await?; - let observed = self - .authority - .load(self.expected.cell) - .await? - .ok_or(Error::Fenced)?; - let control = observed.value(); - if !self.same_owner_and_code(control, ¤t) { - return Err(Error::Fenced); - } - let root = control.ltx_root().ok_or(Error::Fenced)?; - if root.commit_sequence < current.view.root().commit_sequence { - return Err(Error::Fenced); - } - if root == current.view.root() { - return Ok(self.snapshot_receipt(¤t)); - } - let admission = Arc::new(self.runtime.reserve_read_view()?); - let verified = self.replica.open_root(&root).await?; - if verified.schema() != self.expected.schema { - return Err(Error::Fenced); - } - let replacement = ReplicaSnapshot { - owner: current.owner.clone(), - epoch: current.epoch, - view: open_view(&self.runtime, verified, destination, Arc::clone(&admission)).await?, - _admission: admission, - }; - self.confirm_authority(&replacement).await?; - let receipt = self.snapshot_receipt(&replacement); - *self.snapshot.write().await = replacement; - Ok(receipt) - } - - /// Executes one compiled typed query against this read-only snapshot. - /// - /// A minimum newer than this view fails rather than returning an older - /// value. Authority is checked after SQL before any result is released. - pub async fn query( - &self, - minimum: Option, - input: Q::Input, - ) -> Result> { - let operation = self.registry.query_contract::(self.target.namespace())?; - validate_description(&self.registry, Q::MODULE, self.expected, operation)?; - let input = encode_wire(&input, operation.input_limit)?; - let observed = self - .query_encoded(EncodedQuery { - target: self.target.clone(), - expected: self.expected, - minimum, - now_ms: unix_time_ms()?, - module: Q::MODULE, - operation_id: Q::ID, - codec_version: Q::CODEC_VERSION, - input, - input_limit: operation.input_limit, - output_limit: operation.output_limit, - }) - .await?; - Ok(Observed { - output: decode_wire(&observed.output, operation.output_limit)?, - receipt: observed.receipt, - }) - } - - pub(crate) async fn query_encoded(&self, query: EncodedQuery) -> Result { - self.runtime.ensure_running()?; - if self.query_gate.is_closed() { - return Err(Error::Fenced); - } - if query.target != self.target || query.expected != self.expected { - return Err(Error::Fenced); - } - validate_minimum(self.expected, query.minimum)?; - let (module, operation) = self.registry.routed_query_contract( - self.target.namespace(), - query.operation_id, - query.codec_version, - )?; - if module != query.module - || operation.input_limit != query.input_limit - || operation.output_limit != query.output_limit - { - return Err(Error::Registry("replica query contract changed")); - } - validate_description(&self.registry, module, self.expected, operation)?; - let snapshot = self.snapshot.read().await.clone(); - let observed = self.snapshot_receipt(&snapshot); - if let Some(minimum) = query - .minimum - .filter(|minimum| observed.commit_sequence < minimum.commit_sequence) - { - return Err(Error::ReplicaBehind { - observed_sequence: observed.commit_sequence, - minimum_sequence: minimum.commit_sequence, - }); - } - let input = query.input; - let deadline = Instant::now() + QUERY_DEADLINE; - let permit = tokio::time::timeout_at( - deadline.into(), - Arc::clone(&self.query_gate).acquire_owned(), - ) - .await - .map_err(|_| Error::Deadline)? - .map_err(|_| Error::Fenced)?; - let job = tokio::time::timeout_at(deadline.into(), self.runtime.reserve_sql_job()) - .await - .map_err(|_| Error::Deadline)??; - let interrupt = snapshot.view.connection()?.get_interrupt_handle(); - let active_snapshot = snapshot.clone(); - let registry = Arc::clone(&self.registry); - let cell = self.expected.cell; - let schema = self.expected.schema; - let sequence = observed.commit_sequence; - let now_ms = unix_time_ms()?; - let mut task = tokio::task::spawn_blocking(move || { - let _permit = permit; - let _job = job; - // Caller cancellation can drop the reader while SQL is running. - // Keep its view and admission together, releasing them before the - // job charge that node drain waits on. - let snapshot = active_snapshot; - let view = &snapshot.view; - let connection = view.connection()?; - view.take_io_error(); - crab_ltx::with_paged_io_deadline(deadline, || { - let (observed_sequence, now_ms) = local::current_metadata(&connection, now_ms)?; - if observed_sequence != sequence { - return Err(Error::Fenced); - } - registry.execute_query( - &connection, - QueryInvocation { - module, - operation_id: query.operation_id, - codec_version: query.codec_version, - schema, - cell, - commit_sequence: sequence, - now_ms, - input: &input, - }, - ) - }) - .map_err(|error| match view.take_io_error() { - Some(crab_ltx::CrabError::Deadline) => Error::Deadline, - Some(source) => Error::from(source), - None => error, - }) - }); - let result = match tokio::time::timeout_at(deadline.into(), &mut task).await { - Ok(result) => result.map_err(Error::WorkerJoin)?, - Err(_) => { - interrupt.interrupt(); - let _ = task.await; - return Err(Error::Deadline); - } - }?; - tokio::time::timeout_at(deadline.into(), self.confirm_authority(&snapshot)) - .await - .map_err(|_| Error::Deadline)??; - self.runtime.ensure_running()?; - Ok(EncodedObservation { - output: result, - receipt: observed, - }) - } - - fn snapshot_receipt(&self, snapshot: &ReplicaSnapshot) -> Receipt { - receipt(self.expected, snapshot.view.root().commit_sequence) - } - - async fn confirm_snapshot(&self, snapshot: &ReplicaSnapshot) -> Result<()> { - if self.query_gate.is_closed() { - return Err(Error::Fenced); - } - let current = self - .authority - .load(self.expected.cell) - .await? - .ok_or(Error::Fenced)?; - if !self.same_owner_and_code(current.value(), snapshot) { - return Err(Error::Fenced); - } - Ok(()) - } - - async fn confirm_authority(&self, snapshot: &ReplicaSnapshot) -> Result<()> { - self.confirm_snapshot(snapshot).await?; - if !self - .directory - .is_live(snapshot.owner.session, unix_time_ms()?) - .await? - { - return Err(Error::Fenced); - } - Ok(()) - } - - fn same_owner_and_code(&self, current: &Control, snapshot: &ReplicaSnapshot) -> bool { - current.state == ControlState::Serving - && current.recovery.is_none() - && current.epoch == snapshot.epoch - && current.incarnation == self.expected.incarnation - && current.code == self.expected.code - && current.schema == self.expected.schema - && current.owner.as_ref() == Some(&snapshot.owner) - && current - .root - .as_ref() - .is_some_and(|root| root.commit_sequence >= snapshot.view.root().commit_sequence) - } -} - -async fn open_view( - runtime: &CellRuntime, - verified: crab_ltx::VerifiedRoot, - destination: &Path, - admission: Arc, -) -> Result> { - let job = runtime.reserve_sql_job().await?; - let destination = destination.to_owned(); - // VFS faults need LTX's blocking pool for directory-cache I/O. SQLite must - // use separate SQL admission, retaining both charges if its waiter cancels. - tokio::task::spawn_blocking(move || { - let _job = job; - let _admission = admission; - crab_ltx::with_paged_io_deadline(Instant::now() + QUERY_DEADLINE, || { - verified.open_read_only(&destination).map(Arc::new) - }) - }) - .await - .map_err(Error::WorkerJoin)? - .map_err(Error::from) -} diff --git a/crates/crab-cell-runtime/src/client/routing.rs b/crates/crab-cell-runtime/src/client/routing.rs deleted file mode 100644 index c713627b7..000000000 --- a/crates/crab-cell-runtime/src/client/routing.rs +++ /dev/null @@ -1,331 +0,0 @@ -//! Shared placement, admission-aware selection and bounded replica query routing. - -use std::{collections::HashMap, sync::Arc, time::Duration}; - -use crate::control::{ControlState, authority::CellAuthority}; -use crate::identity::{CellTarget, NodeId, SessionId}; -use crate::node::{NodeAdvertisement, NodeDirectory}; -use crate::peer::{PeerReplicaResolver, ReplicaPeerClient}; -use crate::read_policy::ReadPolicyStore; -use crate::registry::Query; -use crate::{Error, Result}; - -use super::{CellDescription, EncodedQuery, Observed, Receipt, local::unix_time_ms}; -use crate::codec::{decode_wire, encode_wire}; - -/// Selects read replicas from authoritative policy and signed live membership. -/// -/// Clone this router across callers to share outstanding-attempt counts. These -/// counts describe this ingress only, not execution load from other ingresses. -#[derive(Clone)] -pub struct ReplicaReadRouter { - authority: CellAuthority, - policy: ReadPolicyStore, - directory: NodeDirectory, - load: Arc, -} - -impl ReplicaReadRouter { - /// Creates a router sharing the runtime's existing authority and directory. - #[must_use] - pub fn new(authority: CellAuthority, directory: NodeDirectory) -> Self { - Self { - policy: ReadPolicyStore::new(authority.layout().clone()), - authority, - directory, - load: Arc::new(ReplicaRouting::default()), - } - } - - /// Returns the authority-pinned description and current selected readers. - /// - /// This is placement evidence; each reader must still prove its snapshot - /// and fresh owner authority before releasing a query result. - pub async fn selected( - &self, - target: &CellTarget, - ) -> Result<(CellDescription, Vec)> { - let cell = target.cell_id(); - // These observations are independent. The incarnation check below - // rejects a policy from another Cell lifetime before it can route work. - let (control, policy) = - tokio::try_join!(self.authority.load(cell), self.policy.load(cell))?; - let control = control.ok_or(Error::ReplicaUnavailable)?; - let control = control.value(); - if control.state != ControlState::Serving || control.recovery.is_some() { - return Err(Error::Fenced); - } - let owner = control.owner.as_ref().ok_or(Error::Fenced)?; - let expected = CellDescription { - cell, - incarnation: control.incarnation, - code: control.code, - schema: control.schema, - }; - let Some(policy) = policy else { - return Ok((expected, Vec::new())); - }; - let policy = policy.value(); - if policy.incarnation() != control.incarnation || policy.desired_readers() == 0 { - return Ok((expected, Vec::new())); - } - let selected = self - .directory - .select_readers( - cell, - owner.session, - control.code, - usize::from(policy.desired_readers()), - unix_time_ms()?, - 10_000, - ) - .await?; - Ok((expected, selected)) - } - - /// Executes a typed read on a selected replica and returns its serving node. - /// - /// Selection and all attempts share one five-second deadline. Each attempt - /// receives an equal share of the remaining time and candidate count. A local - /// resolver may serve this node's admitted views without a self-dial. The - /// caller must perform its product authorization before invoking this route. - /// There is no owner fallback when replicas are absent, behind or fenced. - pub async fn query( - &self, - peer: &ReplicaPeerClient, - local: Option<(SessionId, &dyn PeerReplicaResolver)>, - target: &CellTarget, - minimum: Option, - input: Q::Input, - ) -> Result<(Observed, NodeId)> { - let deadline = tokio::time::Instant::now() + Duration::from_secs(5); - let query = async { - let (expected, mut selected) = self.selected(target).await?; - super::local::validate_minimum(expected, minimum)?; - let operation = peer.registry().query_contract::(target.namespace())?; - super::validate_description(peer.registry(), Q::MODULE, expected, operation)?; - let input = encode_wire(&input, operation.input_limit)?; - let mut behind = None; - let mut fenced = false; - while !selected.is_empty() { - // Selection and reservation are atomic across this ingress's - // Cells. The guard releases load on every exit, including timeout. - let (index, attempt) = self - .load - .reserve(selected.iter().map(NodeAdvertisement::node))?; - let node = selected.remove(index); - let reader_node = attempt.node; - let query = EncodedQuery { - target: target.clone(), - expected, - minimum, - now_ms: unix_time_ms()?, - module: Q::MODULE, - operation_id: Q::ID, - codec_version: Q::CODEC_VERSION, - input: input.clone(), - input_limit: operation.input_limit, - output_limit: operation.output_limit, - }; - // A blackholed peer or stalled local resolver must leave time - // for the other selected replicas. The outer deadline still - // bounds discovery and every attempt together. - let now = tokio::time::Instant::now(); - let remaining = deadline.saturating_duration_since(now); - let candidates = - u32::try_from(selected.len() + 1).map_err(|_| Error::ReplicaUnavailable)?; - let attempt_deadline = now + remaining / candidates; - let querying = async { - if let Some((_, resolver)) = - local.filter(|(session, _)| *session == node.session()) - { - match resolver.resolve(target.clone()).await { - Ok(reader) => reader.query_encoded(query).await, - Err(error) => Err(error), - } - } else { - peer.query_encoded(node, query, attempt_deadline).await - } - }; - let queried = tokio::time::timeout_at(attempt_deadline, querying) - .await - .unwrap_or(Err(Error::ReplicaUnavailable)); - drop(attempt); - match queried { - Ok(result) => { - if result.receipt.cell != expected.cell - || result.receipt.incarnation != expected.incarnation - || minimum.is_some_and(|minimum| { - result.receipt.commit_sequence < minimum.commit_sequence - }) - { - return Err(Error::Peer("read replica returned an invalid receipt")); - } - return Ok(( - Observed { - output: decode_wire(&result.output, operation.output_limit)?, - receipt: result.receipt, - }, - reader_node, - )); - } - Err(error @ Error::ReplicaBehind { .. }) => behind = Some(error), - Err(Error::Fenced) => fenced = true, - Err(error @ Error::PeerAuthorization(_)) => return Err(error), - Err(error) => { - tracing::debug!(cell = ?target.cell_id(), error = %error, "selected read replica unavailable") - } - } - } - Err(behind.unwrap_or(if fenced { - Error::Fenced - } else { - Error::ReplicaUnavailable - })) - }; - tokio::time::timeout_at(deadline, query) - .await - .unwrap_or(Err(Error::ReplicaUnavailable)) - } -} - -pub(super) struct ReplicaClient { - pub(super) router: ReplicaReadRouter, - pub(super) peer: ReplicaPeerClient, - pub(super) local: Option<(SessionId, Arc)>, -} - -#[derive(Default)] -struct ReplicaRouting { - state: std::sync::Mutex, -} - -#[derive(Default)] -struct ReplicaRoutingState { - cursor: usize, - in_flight: HashMap, -} - -struct ReplicaAttempt<'a> { - routing: &'a ReplicaRouting, - node: NodeId, -} - -impl ReplicaRouting { - fn reserve( - &self, - candidates: impl ExactSizeIterator, - ) -> Result<(usize, ReplicaAttempt<'_>)> { - let count = candidates.len(); - if count == 0 { - return Err(Error::ReplicaUnavailable); - } - let mut state = self - .state - .lock() - .map_err(|_| Error::Control("replica routing load lock poisoned"))?; - let start = state.cursor % count; - let (index, node) = candidates - .enumerate() - .min_by_key(|(index, node)| { - let distance = if *index >= start { - *index - start - } else { - count - (start - *index) - }; - (state.in_flight.get(node).copied().unwrap_or(0), distance) - }) - .ok_or(Error::ReplicaUnavailable)?; - state.cursor = index + 1; - *state.in_flight.entry(node).or_default() += 1; - Ok(( - index, - ReplicaAttempt { - routing: self, - node, - }, - )) - } -} - -impl Drop for ReplicaAttempt<'_> { - fn drop(&mut self) { - let mut state = self - .routing - .state - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner); - if let Some(active) = state.in_flight.get_mut(&self.node) { - *active -= 1; - if *active == 0 { - state.in_flight.remove(&self.node); - } - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - fn nodes() -> [NodeId; 3] { - [1, 2, 3].map(|byte| NodeId::from_bytes([byte; 16])) - } - - #[test] - fn idle_readers_share_ties_and_a_busy_reader_is_skipped() { - let routing = ReplicaRouting::default(); - let nodes = nodes(); - let mut counts = [0; 3]; - for _ in 0..12 { - let (index, _attempt) = routing.reserve(nodes.into_iter()).unwrap(); - counts[index] += 1; - } - assert_eq!(counts, [4, 4, 4]); - - let (_, busy) = routing.reserve(nodes.into_iter()).unwrap(); - assert_eq!(busy.node, nodes[0]); - counts = [0; 3]; - for _ in 0..12 { - let (index, _attempt) = routing.reserve(nodes.into_iter()).unwrap(); - counts[index] += 1; - } - assert_eq!(counts, [0, 6, 6]); - } - - #[tokio::test(flavor = "multi_thread")] - async fn cancelled_attempt_releases_load_for_overlapping_candidate_sets() { - let routing = Arc::new(ReplicaRouting::default()); - let nodes = nodes(); - let blocked_routing = Arc::clone(&routing); - let (started, ready) = tokio::sync::oneshot::channel(); - let blocked = tokio::spawn(async move { - let (_, attempt) = blocked_routing.reserve(nodes.into_iter()).unwrap(); - started.send(attempt.node).unwrap(); - std::future::pending::<()>().await; - drop(attempt); - }); - assert_eq!(ready.await.unwrap(), nodes[0]); - // A different Cell can share the busy physical reader. Its local load - // must carry across the two candidate sets without pinning membership. - let (index, attempt) = routing.reserve([nodes[0], nodes[2]].into_iter()).unwrap(); - assert_eq!(index, 1); - drop(attempt); - blocked.abort(); - assert!(blocked.await.unwrap_err().is_cancelled()); - assert!(routing.state.lock().unwrap().in_flight.is_empty()); - } - - #[tokio::test] - async fn expired_route_attempt_releases_its_load() { - let routing = ReplicaRouting::default(); - let nodes = nodes(); - let expired = tokio::time::timeout(Duration::from_millis(10), async { - let (_, _attempt) = routing.reserve(nodes.into_iter()).unwrap(); - std::future::pending::<()>().await; - }) - .await; - assert!(expired.is_err()); - assert!(routing.state.lock().unwrap().in_flight.is_empty()); - } -} diff --git a/crates/crab-cell-runtime/src/client/runtime.rs b/crates/crab-cell-runtime/src/client/runtime.rs deleted file mode 100644 index 788c67723..000000000 --- a/crates/crab-cell-runtime/src/client/runtime.rs +++ /dev/null @@ -1,128 +0,0 @@ -//! In-process routing for Cells admitted after a client was created. - -use super::*; -use crate::cell::actor::CellRuntime; -use crate::cell::catalog::CellCatalog; -use crate::control::authority::CellAuthority; -use crate::ltx::CellStorageLayout; - -/// Selects a local owner before an invocation can be forwarded to a peer. -/// -/// The product may acquire an idle, cataloged Cell through runtime admission. -/// Returning `None` delegates to the remote transport; errors stop dispatch. -pub trait LocalCellResolver: Send + Sync + 'static { - /// Returns the authorized local owner, or `None` when routing must continue remotely. - fn resolve( - &self, - target: CellTarget, - ) -> Pin>> + Send + 'static>>; -} - -#[derive(Clone)] -pub(super) struct RuntimeCellTransport { - registry: Arc, - resolver: Arc, - remote: Option>, -} - -impl RuntimeCellTransport { - pub(super) fn new( - registry: Arc, - runtime: CellRuntime, - layout: CellStorageLayout, - ) -> Self { - Self { - registry, - resolver: Arc::new(RuntimeLocalResolver { runtime, layout }), - remote: None, - } - } - - pub(super) fn with_resolver( - registry: Arc, - resolver: Arc, - remote: Arc, - ) -> Self { - Self { - registry, - resolver, - remote: Some(remote), - } - } - - async fn owner(&self, target: &CellTarget) -> Result> { - let local = self.resolver.resolve(target.clone()).await?; - let Some(handle) = local else { - return self - .remote - .clone() - .ok_or(Error::Control("target Cell is not locally owned")); - }; - Ok(Arc::new(LocalCellTransport { - registry: self.registry.clone(), - handles: Arc::new(HashMap::from([(handle.cell_id(), handle.clone())])), - handle, - telemetry: CellTelemetryHandle::default(), - })) - } -} - -#[derive(Clone)] -pub(super) struct RuntimeLocalResolver { - pub(super) runtime: CellRuntime, - pub(super) layout: CellStorageLayout, -} - -impl LocalCellResolver for RuntimeLocalResolver { - fn resolve( - &self, - target: CellTarget, - ) -> Pin>> + Send + 'static>> { - let resolver = self.clone(); - Box::pin(async move { - let catalog = CellCatalog::new(resolver.layout.clone(), target.tenant()) - .lookup(target.cell_id()) - .await? - .ok_or(Error::Control("target Cell is not cataloged"))?; - let control = CellAuthority::new(resolver.layout) - .load(target.cell_id()) - .await? - .ok_or(Error::Control("target Cell has no authority record"))?; - resolver.runtime.local_handle(catalog, &control).await - }) - } -} - -impl CellTransport for RuntimeCellTransport { - fn describe( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>> { - let client = self.clone(); - Box::pin(async move { client.owner(&target).await?.describe(target).await }) - } - - fn command( - &self, - command: EncodedCommand, - ) -> Pin> + Send + 'static>> { - let client = self.clone(); - Box::pin(async move { client.owner(&command.target).await?.command(command).await }) - } - - fn query( - &self, - query: EncodedQuery, - ) -> Pin> + Send + 'static>> { - let client = self.clone(); - Box::pin(async move { client.owner(&query.target).await?.query(query).await }) - } - - fn resolve( - &self, - resolve: EncodedResolve, - ) -> Pin> + Send + 'static>> { - let client = self.clone(); - Box::pin(async move { client.owner(&resolve.target).await?.resolve(resolve).await }) - } -} diff --git a/crates/crab-cell-runtime/src/client/tests.rs b/crates/crab-cell-runtime/src/client/tests.rs deleted file mode 100644 index 945bb2d54..000000000 --- a/crates/crab-cell-runtime/src/client/tests.rs +++ /dev/null @@ -1,769 +0,0 @@ -use std::{ - future::Future, - pin::Pin, - sync::{ - Arc, Mutex, OnceLock, - atomic::{AtomicUsize, Ordering}, - }, - time::{SystemTime, UNIX_EPOCH}, -}; -use tokio::sync::Notify; - -use super::{ - CellClient, CellDescription, CellTransport, EncodedCommand, EncodedObservation, EncodedQuery, - EncodedResolve, InvocationError, PendingMutation, decode_pending, -}; -use crate::Error; -use crate::cell::catalog::CatalogRole; -use crate::cell::executor::StoredOutcome; -use crate::cell::executor::{MutationIdentity, Resolution}; -use crate::client::Receipt; -use crate::identity::{ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId}; -use crate::identity::{IncarnationId, RequestId}; -use crate::peer::{PeerPrincipal, PeerRoundTrip, PeerSigner}; -use crate::registry::{ - BuildDescriptor, CellModule, Command, MigrationDescriptor, ModuleDescriptor, - NamespaceDescriptor, Query, RegistryBuilder, -}; -use crate::registry::{ - CommandContext, CommandResult, OperationDescriptor, QueryContext, RetainedCodeDescriptor, -}; - -const MODULE: &str = "pending-test"; -const ADMISSION_CELL: crate::CellId = crate::CellId::from_bytes([1; 32]); -const NAMESPACE: NamespaceId = NamespaceId::from_bytes([3; 16]); -const MIGRATION: &str = "CREATE TABLE pending_test(value BLOB NOT NULL)"; -const RETAINED_CODE: Digest = Digest::from_bytes([14; 32]); - -#[test] -fn pending_rejection_preserves_published_receipt() { - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NAMESPACE, - b"pending", - ) - .expect("valid pending target"); - let incarnation = IncarnationId::from_bytes([3; 16]); - let pending = PendingMutation { - target: target.clone(), - incarnation, - identity: MutationIdentity { - request_id: RequestId::from_bytes([4; 16]), - issued_at_ms: 1_000, - expires_at_ms: 61_000, - }, - operation_digest: Digest::from_bytes([5; 32]), - max_result_bytes: 64, - }; - let result = - crate::codec::encode_wire(&b"lease-lost".to_vec(), 64).expect("bounded rejection result"); - - let decoded = decode_pending::>( - &pending, - StoredOutcome::Rejected { - result, - commit_sequence: 42, - }, - ); - - assert!(matches!( - decoded, - Err(InvocationError::Rejected(committed)) - if committed.output == b"lease-lost" - && committed.receipt == Receipt { - cell: target.cell_id(), - incarnation, - commit_sequence: 42, - } - )); -} - -struct PendingCommand; - -impl Command for PendingCommand { - const MODULE: &'static str = MODULE; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - type Input = Vec; - type Output = Vec; - - fn execute( - _context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - Ok(CommandResult::Success(input)) - } -} - -struct StreamQuery; - -impl Query for StreamQuery { - const MODULE: &'static str = MODULE; - const ID: u32 = 2; - const CODEC_VERSION: u32 = 1; - type Input = u64; - type Output = Vec; - - fn execute(_context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - Ok(input.to_be_bytes().to_vec()) - } -} - -struct AcceptedButLost { - calls: Arc, -} - -impl PeerRoundTrip for AcceptedButLost { - fn send( - &self, - _target: CellTarget, - _request: Vec, - _remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let calls = Arc::clone(&self.calls); - Box::pin(async move { - calls.fetch_add(1, Ordering::SeqCst); - Err(Error::PeerTransportUnknown { - context: "test accepted request lost its response", - source: Box::new(Error::RuntimeClosed), - }) - }) - } -} - -#[tokio::test] -async fn ambiguous_peer_command_preserves_identity_without_retry() { - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NAMESPACE, - b"pending", - ) - .unwrap(); - let identity = MutationIdentity { - request_id: RequestId::from_bytes([7; 16]), - issued_at_ms: 1_000, - expires_at_ms: 61_000, - }; - let operation_digest = Digest::from_bytes([8; 32]); - let calls = Arc::new(AtomicUsize::new(0)); - let transport = crate::peer::PeerClientTransport::new( - Arc::new(PeerSigner::new( - SessionId::from_bytes([9; 16]), - Digest::from_bytes([10; 32]), - ed25519_dalek::SigningKey::from_bytes(&[11; 32]), - )), - PeerPrincipal { - issuer: "urn:crab:test".into(), - subject: "operator".into(), - actions: vec!["repository.write".into()], - }, - Arc::new(AcceptedButLost { - calls: Arc::clone(&calls), - }), - ); - - let result = CellTransport::command( - &transport, - EncodedCommand { - target: target.clone(), - expected: CellDescription { - cell: target.cell_id(), - incarnation: IncarnationId::from_bytes([12; 16]), - code: Digest::from_bytes([13; 32]), - schema: 1, - }, - identity, - operation_digest, - now_ms: 1_000, - module: MODULE, - operation_id: 1, - codec_version: 1, - input: b"input".to_vec(), - input_limit: 64, - output_limit: 64, - }, - ) - .await; - - assert!(matches!( - result, - Err(Error::OutcomeUnknown { - request_id, - operation_digest: digest, - .. - }) if request_id == identity.request_id && digest == operation_digest - )); - assert_eq!(calls.load(Ordering::SeqCst), 1); -} - -struct PendingModule; - -impl CellModule for PendingModule { - const NAME: &'static str = MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: OnceLock = OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: MODULE, - source_digest: Digest::from_bytes([4; 32]), - retained_codes: &[RetainedCodeDescriptor { - code: RETAINED_CODE, - schema_min: 1, - schema_max: 1, - }], - schema_min: 1, - schema_max: 1, - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: MIGRATION, - digest: Digest::from_bytes(*blake3::hash(MIGRATION.as_bytes()).as_bytes()), - }])), - commands: &[OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 64, - output_limit: 64, - }], - queries: &[OperationDescriptor { - id: 2, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 64, - output_limit: 64, - }], - workflow_definitions: &[], - activity_types: &[], - namespaces: &[NamespaceDescriptor { - id: NAMESPACE, - name: MODULE, - role: CatalogRole::Repository, - shards: 1, - effect_targets: &[], - dead_letter: None, - }], - }) - } - - fn register(self, registry: &mut RegistryBuilder) -> crate::Result<()> { - registry.bind_command::()?; - registry.bind_query::() - } -} - -struct StreamTransport { - description: CellDescription, - sequence: Arc, - fenced: Arc, - query_started: Option>, - query_release: Option>, -} - -impl CellTransport for StreamTransport { - fn describe( - &self, - _target: CellTarget, - ) -> Pin> + Send + 'static>> { - let description = self.description; - Box::pin(async move { Ok(description) }) - } - - fn command( - &self, - _command: EncodedCommand, - ) -> Pin> + Send + 'static>> { - Box::pin(async { Err(Error::Command("unexpected stream command")) }) - } - - fn query( - &self, - _query: EncodedQuery, - ) -> Pin> + Send + 'static>> { - let fenced = self.fenced.load(Ordering::Acquire) != 0; - let description = self.description; - let sequence = self.sequence.fetch_add(1, Ordering::SeqCst) as u64 + 1; - let query_started = self.query_started.clone(); - let query_release = self.query_release.clone(); - Box::pin(async move { - if let Some(query_started) = query_started { - query_started.notify_one(); - } - if let Some(query_release) = query_release { - query_release.notified().await; - } - if fenced { - return Err(Error::Fenced); - } - let output = crate::codec::encode_wire(&sequence.to_be_bytes().to_vec(), 64)?; - Ok(EncodedObservation { - output, - receipt: Receipt { - cell: description.cell, - incarnation: description.incarnation, - commit_sequence: sequence, - }, - }) - }) - } - - fn resolve( - &self, - _resolve: EncodedResolve, - ) -> Pin> + Send + 'static>> { - Box::pin(async { Err(Error::Command("unexpected stream resolve")) }) - } -} - -struct RefusedThenUnknownTransport { - description: CellDescription, - descriptions: AtomicUsize, - command_digest: Arc>>, - resolved_digest: Arc>>, - calls: Arc, -} - -impl CellTransport for RefusedThenUnknownTransport { - fn describe( - &self, - _target: CellTarget, - ) -> Pin> + Send + 'static>> { - let description = self.description; - let refused = self.descriptions.fetch_add(1, Ordering::SeqCst) < 2; - Box::pin(async move { - if refused { - Err(Error::Capacity("peer HTTP admission")) - } else { - Ok(description) - } - }) - } - - fn command( - &self, - command: EncodedCommand, - ) -> Pin> + Send + 'static>> { - let observed = self.command_digest.clone(); - if self.calls.fetch_add(1, Ordering::SeqCst) < 2 { - return Box::pin(async { Err(Error::Capacity("owner mailbox")) }); - } - Box::pin(async move { - *observed.lock().unwrap() = Some(command.operation_digest); - Err(Error::OutcomeUnknown { - request_id: command.identity.request_id, - operation_digest: command.operation_digest, - source: Box::new(Error::Capacity("accepted publication")), - }) - }) - } - - fn query( - &self, - _query: EncodedQuery, - ) -> Pin> + Send + 'static>> { - Box::pin(async { Err(Error::Command("unexpected query")) }) - } - - fn resolve( - &self, - resolve: EncodedResolve, - ) -> Pin> + Send + 'static>> { - let observed = self.resolved_digest.clone(); - Box::pin(async move { - *observed.lock().unwrap() = Some(resolve.operation_digest); - Ok(Resolution::Unknown) - }) - } -} - -#[tokio::test] -async fn capacity_wait_preserves_unknown_outcome_identity_and_digest_for_resolve() { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "pending-test".into(), - cargo_lock_digest: Digest::from_bytes([5; 32]), - }); - builder.register(PendingModule).unwrap(); - let registry = Arc::new(builder.finish().unwrap()); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NAMESPACE, - b"pending", - ) - .unwrap(); - let description = CellDescription { - cell: target.cell_id(), - incarnation: IncarnationId::from_bytes([6; 16]), - code: RETAINED_CODE, - schema: 1, - }; - let command_digest = Arc::new(Mutex::new(None)); - let resolved_digest = Arc::new(Mutex::new(None)); - let calls = Arc::new(AtomicUsize::new(0)); - let client = CellClient::new( - registry, - Arc::new(RefusedThenUnknownTransport { - description, - descriptions: AtomicUsize::new(0), - command_digest: command_digest.clone(), - resolved_digest: resolved_digest.clone(), - calls: calls.clone(), - }), - ) - .with_admission_backpressure(8, 16_384, std::time::Duration::from_secs(1)) - .unwrap(); - let now_ms = i64::try_from( - SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap(); - let identity = MutationIdentity { - request_id: RequestId::from_bytes([7; 16]), - issued_at_ms: now_ms, - expires_at_ms: now_ms + 60_000, - }; - - let observed = client.clone().with_observed_description(description); - for (index, client) in [client, observed].into_iter().enumerate() { - let prepared = client - .prepare_command::(&target, identity, b"input".to_vec()) - .await - .expect("prepare exact command"); - let evidence = prepared.evidence().clone(); - let pending = match prepared.execute().await { - Err(InvocationError::Pending(pending)) => pending, - outcome => panic!("unexpected command outcome: {outcome:?}"), - }; - assert_eq!(*pending, evidence); - assert_eq!(calls.load(Ordering::SeqCst), 3 + index); - assert_eq!(pending.identity(), identity); - assert_eq!( - Some(pending.operation_digest()), - *command_digest.lock().unwrap() - ); - assert_eq!(client.resolve(&pending).await.unwrap(), Resolution::Unknown); - assert_eq!( - Some(pending.operation_digest()), - *resolved_digest.lock().unwrap() - ); - } -} - -#[tokio::test] -async fn state_stream_advances_receipts_and_cancellation_is_terminal() { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "stream-test".into(), - cargo_lock_digest: Digest::from_bytes([15; 32]), - }); - builder.register(PendingModule).unwrap(); - let registry = Arc::new(builder.finish().unwrap()); - let target = CellTarget::new( - TenantId::from_bytes([16; 16]), - ApplicationId::from_bytes([17; 16]), - NAMESPACE, - b"stream", - ) - .unwrap(); - let description = CellDescription { - cell: target.cell_id(), - incarnation: IncarnationId::from_bytes([18; 16]), - code: RETAINED_CODE, - schema: 1, - }; - let fenced = Arc::new(AtomicUsize::new(0)); - let client = CellClient::new( - registry, - Arc::new(StreamTransport { - description, - sequence: Arc::new(AtomicUsize::new(0)), - fenced: fenced.clone(), - query_started: None, - query_release: None, - }), - ); - let mut stream = client - .open_state_stream::( - &target, - std::time::Instant::now() + std::time::Duration::from_secs(1), - ) - .await - .unwrap(); - let first = stream.emit(1).await.unwrap(); - assert_eq!(first.receipt.commit_sequence, 1); - let second = stream.emit(2).await.unwrap(); - assert_eq!(second.receipt.commit_sequence, 2); - assert_eq!(stream.last_receipt(), Some(second.receipt)); - - let cancellation = stream.cancellation(); - cancellation.cancel(); - assert!(matches!( - stream.emit(3).await, - Err(InvocationError::NotStarted(Error::StreamCancelled)) - )); - assert!(stream.is_closed()); - - let dropped_cancellation = { - let dropped_stream = client - .open_state_stream::( - &target, - std::time::Instant::now() + std::time::Duration::from_secs(1), - ) - .await - .unwrap(); - dropped_stream.cancellation() - }; - assert!(dropped_cancellation.is_cancelled()); - - assert!(matches!( - client - .open_state_stream::(&target, std::time::Instant::now()) - .await, - Err(InvocationError::NotStarted(Error::Deadline)) - )); - - fenced.store(1, Ordering::Release); - let mut fenced_stream = client - .open_state_stream::( - &target, - std::time::Instant::now() + std::time::Duration::from_secs(1), - ) - .await - .unwrap(); - assert!(matches!( - fenced_stream.emit(5).await, - Err(InvocationError::NotStarted(Error::Fenced)) - )); - assert!(fenced_stream.is_closed()); - - let query_started = Arc::new(Notify::new()); - let query_release = Arc::new(Notify::new()); - let waiting_client = CellClient::new( - client.registry.clone(), - Arc::new(StreamTransport { - description, - sequence: Arc::new(AtomicUsize::new(0)), - fenced: Arc::new(AtomicUsize::new(0)), - query_started: Some(query_started.clone()), - query_release: Some(query_release), - }), - ); - let waiting_stream = waiting_client - .open_state_stream::( - &target, - std::time::Instant::now() + std::time::Duration::from_secs(1), - ) - .await - .unwrap(); - let cancellation = waiting_stream.cancellation(); - let waiting = tokio::spawn(async move { - let mut waiting_stream = waiting_stream; - waiting_stream.emit(6).await - }); - tokio::time::timeout(std::time::Duration::from_secs(1), query_started.notified()) - .await - .unwrap(); - cancellation.cancel(); - assert!(matches!( - tokio::time::timeout(std::time::Duration::from_secs(1), waiting) - .await - .unwrap() - .unwrap(), - Err(InvocationError::NotStarted(Error::StreamCancelled)) - )); -} - -fn admission_transport( - requests: usize, - bytes: usize, - wait: std::time::Duration, -) -> super::backpressure::BackpressureTransport { - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NAMESPACE, - b"admission", - ) - .unwrap(); - super::backpressure::BackpressureTransport::new( - Arc::new(StreamTransport { - description: CellDescription { - cell: target.cell_id(), - incarnation: IncarnationId::from_bytes([3; 16]), - code: RETAINED_CODE, - schema: 1, - }, - sequence: Arc::default(), - fenced: Arc::default(), - query_started: None, - query_release: None, - }), - requests, - bytes, - wait, - ) - .unwrap() -} - -#[tokio::test] -async fn admission_backpressure_retries_capacity_but_not_fencing() { - let transport = admission_transport(2, 8_192, std::time::Duration::from_secs(1)); - let mut attempts = 0; - let value = transport - .invoke(ADMISSION_CELL, 32, Some(1), || { - attempts += 1; - std::future::ready(if attempts < 3 { - Err(Error::Capacity("owner mailbox")) - } else { - Ok(42) - }) - }) - .await - .unwrap(); - assert_eq!((value, attempts), (42, 3)); - let mut attempts = 0; - let result = transport - .invoke(ADMISSION_CELL, 32, Some(1), || { - attempts += 1; - std::future::ready(Err::<(), _>(Error::Fenced)) - }) - .await; - assert!(matches!(result, Err(Error::Fenced)) && attempts == 1); -} - -#[tokio::test] -async fn admission_backpressure_releases_shared_limits_after_cancellation() { - for (requests, bytes, resource) in [ - (1, 8_192, "client admission requests"), - (2, 2_048, "client admission bytes"), - ] { - let transport = admission_transport(requests, bytes, std::time::Duration::from_secs(1)); - let clone = transport.clone(); - let entered = Notify::new(); - let mut held = Box::pin(transport.invoke(ADMISSION_CELL, 0, Some(1), || { - entered.notify_one(); - std::future::pending::>() - })); - tokio::select! { - result = &mut held => panic!("unexpected completion: {result:?}"), - () = entered.notified() => {} - } - let result = clone - .invoke(ADMISSION_CELL, 0, Some(1), || std::future::ready(Ok(()))) - .await; - assert!(matches!(result, Err(Error::Capacity(found)) if found == resource)); - drop(held); - clone - .invoke(ADMISSION_CELL, 0, Some(1), || std::future::ready(Ok(()))) - .await - .unwrap(); - } -} - -#[tokio::test] -async fn admission_backpressure_expires_without_canceling_accepted_work() { - let transport = admission_transport(1, 8_192, std::time::Duration::from_millis(25)); - let mut attempts = 0; - let result = transport - .invoke(ADMISSION_CELL, 0, Some(1), || { - attempts += 1; - std::future::ready(Err::<(), _>(Error::Capacity("owner mailbox"))) - }) - .await; - assert!(matches!(result, Err(Error::Capacity("owner mailbox"))) && attempts <= 2); - let result = transport - .invoke(ADMISSION_CELL, 0, Some(1), || async { - tokio::time::sleep(std::time::Duration::from_millis(40)).await; - Ok(42) - }) - .await - .unwrap(); - assert_eq!(result, 42); -} - -#[tokio::test] -async fn admission_backpressure_is_fifo_per_cell_and_independent_across_cells() { - let transport = admission_transport(4, 16_384, std::time::Duration::from_secs(1)); - let mut held = Box::pin(transport.invoke( - ADMISSION_CELL, - 0, - Some(crate::cell::actor::CELL_BYTES), - std::future::pending::>, - )); - assert!(futures_util::poll!(&mut held).is_pending()); - let order = AtomicUsize::new(0); - let mut queued = Box::pin(transport.invoke( - ADMISSION_CELL, - 0, - Some(crate::cell::actor::CELL_BYTES), - || std::future::ready(Ok(order.fetch_add(1, Ordering::SeqCst))), - )); - assert!(futures_util::poll!(&mut queued).is_pending()); - transport - .invoke(crate::CellId::from_bytes([2; 32]), 0, Some(1), || { - std::future::ready(Ok(())) - }) - .await - .unwrap(); - drop(held); - let arrival = transport.invoke( - ADMISSION_CELL, - 0, - Some(crate::cell::actor::CELL_BYTES), - || std::future::ready(Ok(order.fetch_add(1, Ordering::SeqCst))), - ); - // Poll the new arrival first: the already queued call must still run first. - let (arrival, queued) = tokio::join!(biased; arrival, queued); - assert_eq!((queued.unwrap(), arrival.unwrap()), (0, 1)); -} - -#[tokio::test] -async fn admission_backpressure_overlaps_calls_without_overtaking_large_waiters() { - let transport = admission_transport(4, 16_384, std::time::Duration::from_secs(1)); - let half = crate::cell::actor::CELL_BYTES / 2; - let mut held = Box::pin(transport.invoke(ADMISSION_CELL, 0, Some(half), || { - std::future::pending::>() - })); - assert!(futures_util::poll!(&mut held).is_pending()); - transport - .invoke(ADMISSION_CELL, 0, Some(half), || std::future::ready(Ok(()))) - .await - .unwrap(); - let mut large = Box::pin(transport.invoke(ADMISSION_CELL, 0, Some(half + 1), || { - std::future::ready(Ok(())) - })); - assert!(futures_util::poll!(&mut large).is_pending()); - let mut small = - Box::pin(transport.invoke(ADMISSION_CELL, 0, Some(1), || std::future::ready(Ok(())))); - assert!(futures_util::poll!(&mut small).is_pending()); - // Canceling the older large waiter releases its claim on remaining bytes. - drop(large); - small.await.unwrap(); -} - -#[tokio::test] -async fn describe_admission_does_not_wait_for_the_owner_mailbox() { - let transport = admission_transport(2, 8_192, std::time::Duration::from_millis(25)); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NAMESPACE, - b"admission", - ) - .unwrap(); - let mut held = Box::pin(transport.invoke( - target.cell_id(), - 0, - Some(crate::cell::actor::CELL_BYTES), - std::future::pending::>, - )); - assert!(futures_util::poll!(&mut held).is_pending()); - assert_eq!( - transport.describe(target.clone()).await.unwrap().cell, - target.cell_id() - ); -} diff --git a/crates/crab-cell-runtime/src/codec.rs b/crates/crab-cell-runtime/src/codec.rs deleted file mode 100644 index 9d71b7dae..000000000 --- a/crates/crab-cell-runtime/src/codec.rs +++ /dev/null @@ -1,328 +0,0 @@ -// Global ceiling; each operation still declares its own, usually smaller, limit. -//! Bounded wire encoding shared by the Cell, peer, and client surfaces. -pub(crate) const MAX_WIRE_BYTES: usize = 4 * 1024 * 1024 + 64 * 1024; - -/// Canonical bounded wire-codec failure. -#[derive(Debug, thiserror::Error)] -pub enum CodecError { - /// The value is not canonical for its declared type. - #[error("invalid wire value: {0}")] - Invalid(&'static str), - /// A value or buffer exceeds the declared byte limit. - #[error("wire value exceeds its declared byte limit")] - Limit, - /// Wire text was not UTF-8. - #[error("wire text is not UTF-8")] - Utf8(#[from] std::str::Utf8Error), -} - -/// Explicit canonical codec implemented by every registered input and output. -pub trait WireValue: Sized + Send + 'static { - /// Appends the canonical encoding of `self`. - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError>; - /// Decodes one value, failing on a non-canonical or truncated encoding. - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result; -} - -/// Append-only encoder enforcing one operation's declared maximum size. -pub struct BoundedEncoder { - bytes: Vec, - limit: usize, -} - -impl BoundedEncoder { - /// Creates an encoder for one operation's declared limit, which must be - /// non-zero and no larger than the global wire ceiling. - pub fn new(limit: u32) -> Result { - let limit = usize::try_from(limit).map_err(|_| CodecError::Limit)?; - if limit == 0 || limit > MAX_WIRE_BYTES { - return Err(CodecError::Limit); - } - Ok(Self { - bytes: Vec::with_capacity(limit.min(256)), - limit, - }) - } - - /// Writes `value` as one canonical boolean tag. - pub fn write_bool(&mut self, value: bool) -> Result<(), CodecError> { - self.write_u8(u8::from(value)) - } - - /// Writes one byte. - pub fn write_u8(&mut self, value: u8) -> Result<(), CodecError> { - self.extend(&[value]) - } - - /// Writes `value` big-endian. - pub fn write_u32(&mut self, value: u32) -> Result<(), CodecError> { - self.extend(&value.to_be_bytes()) - } - - /// Writes `value` big-endian. - pub fn write_u64(&mut self, value: u64) -> Result<(), CodecError> { - self.extend(&value.to_be_bytes()) - } - - /// Writes `value` big-endian. - pub fn write_i64(&mut self, value: i64) -> Result<(), CodecError> { - self.extend(&value.to_be_bytes()) - } - - /// Writes `value` as its big-endian bits. - pub fn write_f64(&mut self, value: f64) -> Result<(), CodecError> { - if !value.is_finite() { - return Err(CodecError::Invalid("non-finite f64")); - } - let normalized = if value == 0.0 { 0.0 } else { value }; - self.extend(&normalized.to_bits().to_be_bytes()) - } - - /// Writes a `u32` length followed by the bytes. - pub fn write_bytes(&mut self, value: &[u8]) -> Result<(), CodecError> { - let length = u32::try_from(value.len()).map_err(|_| CodecError::Limit)?; - self.write_u32(length)?; - self.extend(value) - } - - /// Writes a `u32` length followed by the UTF-8 bytes. - pub fn write_text(&mut self, value: &str) -> Result<(), CodecError> { - self.write_bytes(value.as_bytes()) - } - - /// Writes an element count, rejecting one that does not fit a `u32`. - pub fn write_count(&mut self, count: usize) -> Result<(), CodecError> { - self.write_u32(u32::try_from(count).map_err(|_| CodecError::Limit)?) - } - - /// Returns the encoded bytes. - pub fn finish(self) -> Vec { - self.bytes - } - - fn extend(&mut self, value: &[u8]) -> Result<(), CodecError> { - let end = self - .bytes - .len() - .checked_add(value.len()) - .filter(|end| *end <= self.limit) - .ok_or(CodecError::Limit)?; - self.bytes.reserve(end - self.bytes.len()); - self.bytes.extend_from_slice(value); - Ok(()) - } -} - -/// Forward-only decoder that rejects truncation, trailing bytes and bad tags. -pub struct BoundedDecoder<'a> { - bytes: &'a [u8], - position: usize, -} - -impl<'a> BoundedDecoder<'a> { - /// Creates a decoder over `bytes`, which must be non-empty and within one - /// operation's declared limit. - pub fn new(bytes: &'a [u8], limit: u32) -> Result { - let limit = usize::try_from(limit).map_err(|_| CodecError::Limit)?; - if limit == 0 || limit > MAX_WIRE_BYTES || bytes.len() > limit { - return Err(CodecError::Limit); - } - Ok(Self { bytes, position: 0 }) - } - - /// Reads a canonical boolean tag. - pub fn read_bool(&mut self) -> Result { - match self.read_u8()? { - 0 => Ok(false), - 1 => Ok(true), - _ => Err(CodecError::Invalid("invalid bool tag")), - } - } - - /// Reads one byte. - pub fn read_u8(&mut self) -> Result { - Ok(self.take(1)?[0]) - } - - /// Reads a big-endian `u32`. - pub fn read_u32(&mut self) -> Result { - let bytes = self - .take(4)? - .try_into() - .map_err(|_| CodecError::Invalid("truncated u32"))?; - Ok(u32::from_be_bytes(bytes)) - } - - /// Reads a big-endian `u64`. - pub fn read_u64(&mut self) -> Result { - let bytes = self - .take(8)? - .try_into() - .map_err(|_| CodecError::Invalid("truncated u64"))?; - Ok(u64::from_be_bytes(bytes)) - } - - /// Reads a big-endian `i64`. - pub fn read_i64(&mut self) -> Result { - let bytes = self - .take(8)? - .try_into() - .map_err(|_| CodecError::Invalid("truncated i64"))?; - Ok(i64::from_be_bytes(bytes)) - } - - /// Reads a canonical finite `f64`, rejecting non-finite values and - /// negative zero. - pub fn read_f64(&mut self) -> Result { - let value = f64::from_bits(self.read_u64()?); - if !value.is_finite() || value.to_bits() == (-0.0_f64).to_bits() { - return Err(CodecError::Invalid("noncanonical f64")); - } - Ok(value) - } - - /// Reads a `u32` length and the bytes that follow it. - pub fn read_bytes(&mut self) -> Result<&'a [u8], CodecError> { - let length = usize::try_from(self.read_u32()?) - .map_err(|_| CodecError::Invalid("byte length overflow"))?; - self.take(length) - } - - /// Reads length-delimited UTF-8 text. - pub fn read_text(&mut self) -> Result<&'a str, CodecError> { - Ok(std::str::from_utf8(self.read_bytes()?)?) - } - - /// Reads an element count. - pub fn read_count(&mut self) -> Result { - usize::try_from(self.read_u32()?).map_err(|_| CodecError::Invalid("count overflow")) - } - - /// Fails when bytes remain unread after the last value. - pub fn finish(self) -> Result<(), CodecError> { - if self.position != self.bytes.len() { - return Err(CodecError::Invalid("trailing wire bytes")); - } - Ok(()) - } - - fn take(&mut self, length: usize) -> Result<&'a [u8], CodecError> { - let end = self - .position - .checked_add(length) - .filter(|end| *end <= self.bytes.len()) - .ok_or(CodecError::Invalid("truncated wire value"))?; - let value = &self.bytes[self.position..end]; - self.position = end; - Ok(value) - } -} - -macro_rules! fixed_wire { - ($type:ty, $write:ident, $read:ident) => { - impl WireValue for $type { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.$write(*self) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - decoder.$read() - } - } - }; -} - -fixed_wire!(bool, write_bool, read_bool); -fixed_wire!(u8, write_u8, read_u8); -fixed_wire!(u32, write_u32, read_u32); -fixed_wire!(u64, write_u64, read_u64); -fixed_wire!(i64, write_i64, read_i64); -fixed_wire!(f64, write_f64, read_f64); - -impl WireValue for Vec { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(self) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(decoder.read_bytes()?.to_vec()) - } -} - -impl WireValue for String { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_text(self) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(decoder.read_text()?.to_owned()) - } -} - -impl WireValue for Option { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - None => encoder.write_u8(0), - Some(value) => { - encoder.write_u8(1)?; - value.encode(encoder) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(None), - 1 => Ok(Some(T::decode(decoder)?)), - _ => Err(CodecError::Invalid("invalid option tag")), - } - } -} - -impl WireValue for () { - fn encode(&self, _encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - Ok(()) - } - - fn decode(_decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(()) - } -} - -pub(crate) fn decode_wire(input: &[u8], limit: u32) -> Result { - let mut decoder = BoundedDecoder::new(input, limit)?; - let value = T::decode(&mut decoder)?; - decoder.finish()?; - Ok(value) -} - -/// Reads a length-delimited value that must be exactly `N` bytes wide. -pub(crate) fn read_fixed( - decoder: &mut BoundedDecoder<'_>, - message: &'static str, -) -> Result<[u8; N], CodecError> { - decoder - .read_bytes()? - .try_into() - .map_err(|_| CodecError::Invalid(message)) -} - -/// Encodes and decodes one bounded wire value, asserting an exact round trip. -/// -/// Each primitive's in-src codec tests use this, so they all assert the same -/// contract instead of keeping a private copy of the assertion. -#[cfg(test)] -pub(crate) fn roundtrip(value: T) { - let mut encoder = BoundedEncoder::new(1024 * 1024).unwrap(); - value.encode(&mut encoder).unwrap(); - let bytes = encoder.finish(); - let mut decoder = BoundedDecoder::new(&bytes, 1024 * 1024).unwrap(); - assert_eq!(T::decode(&mut decoder).unwrap(), value); - decoder.finish().unwrap(); -} - -pub(crate) fn encode_wire(value: &T, limit: u32) -> Result, CodecError> { - let mut encoder = BoundedEncoder::new(limit)?; - value.encode(&mut encoder)?; - Ok(encoder.finish()) -} diff --git a/crates/crab-cell-runtime/src/control.rs b/crates/crab-cell-runtime/src/control.rs deleted file mode 100644 index 5863a90c5..000000000 --- a/crates/crab-cell-runtime/src/control.rs +++ /dev/null @@ -1,680 +0,0 @@ -//! Cell control record, transitions, and authority CAS ownership. - -pub mod authority; - -mod codec; - -#[cfg(test)] -mod tests; - -use serde::{Deserialize, Serialize}; - -// LTX owns the flag; the runtime only tests it, so validation cannot drift from -// the encoder that set the bit. -use crab_ltx::types::CHECKSUM_FLAG; - -use crate::identity::IncarnationId; -use crate::identity::{CellId, Digest, SessionId}; -use crate::{Error, Result}; - -const MAX_CONTROL_BYTES: usize = 8 * 1024; - -/// Exact immutable recovery root published by the current Cell control record. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct RootRef { - /// Digest of the published root record. - pub digest: Digest, - /// Transaction id the root publishes. - pub txid: u64, - /// Checksum the root publishes. - pub checksum: u64, - /// Root commit sequence. - pub commit_sequence: u64, -} - -impl RootRef { - /// Narrows a Cell-scoped LTX reference to the fields persisted in control JSON. - pub fn from_ltx( - cell: CellId, - incarnation: IncarnationId, - root: crab_ltx::RootRef, - ) -> Result { - if root.cell != *cell.as_bytes() || root.incarnation != *incarnation.as_bytes() { - return Err(Error::Control("prepared root changed Cell scope")); - } - Ok(Self { - digest: Digest::from_bytes(root.digest), - txid: root.position.txid, - checksum: root.position.checksum, - commit_sequence: root.commit_sequence, - }) - } - - /// Restores the typed Cell/incarnation scope inherited from its control record. - #[must_use] - pub fn to_ltx(&self, cell: CellId, incarnation: IncarnationId) -> crab_ltx::RootRef { - crab_ltx::RootRef { - cell: *cell.as_bytes(), - incarnation: *incarnation.as_bytes(), - digest: *self.digest.as_bytes(), - position: crab_ltx::Position { - txid: self.txid, - checksum: self.checksum, - }, - commit_sequence: self.commit_sequence, - } - } -} - -/// Enrolled process currently responsible for one Cell. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct Owner { - /// Boot session currently responsible for the Cell. - pub session: SessionId, - /// Endpoint the owner advertises for peer delivery. - pub endpoint: String, -} - -/// Exact recovered follower tail pinned before a dead owner's Cell can move. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct RecoveryOverlayRef { - /// Session whose sealed log the overlay was recovered from. - pub leader_session: SessionId, - /// Node-log epoch the overlay covers. - pub log_epoch: u64, - /// Digest of the recovery manifest that pins the overlay. - pub manifest_digest: Digest, - /// First node-log sequence the overlay covers. - pub first_node_sequence: u64, - /// Last node-log sequence the overlay covers. - pub last_node_sequence: u64, - /// Published root the overlay supersedes. - pub predecessor: RootRef, - /// Transaction id the recovered tail reaches. - pub final_txid: u64, - /// Checksum the recovered tail reaches. - pub final_checksum: u64, - /// Commit sequence the recovered tail reaches. - pub final_commit_sequence: u64, -} - -impl RecoveryOverlayRef { - fn validate(&self) -> Result<()> { - if self.leader_session.as_bytes().iter().all(|byte| *byte == 0) - || self - .manifest_digest - .as_bytes() - .iter() - .all(|byte| *byte == 0) - || self.log_epoch == 0 - || self.first_node_sequence == 0 - || self.first_node_sequence > self.last_node_sequence - || self.final_txid <= self.predecessor.txid - || self.final_checksum & CHECKSUM_FLAG == 0 - || self.final_commit_sequence <= self.predecessor.commit_sequence - || self.final_commit_sequence > i64::MAX as u64 - { - return Err(Error::Control("invalid recovery overlay")); - } - Ok(()) - } -} - -/// Durable Cell lifecycle state. -#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "lowercase")] -pub enum ControlState { - /// The Cell has no published root yet and is being initialized or recovered. - Recovering, - /// The Cell has an owner and serves requests. - Serving, - /// The Cell has no owner and another node may acquire it. - Idle, - /// The Cell is permanently retired. - Tombstoned, -} - -/// Strict, versioned owner/root authority stored with an object-store ETag. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct Control { - /// Cell this record describes. - pub cell: CellId, - /// Incarnation the record was created for. - pub incarnation: IncarnationId, - /// Ownership epoch, raised when an owner fences its predecessor. - pub epoch: u64, - /// Monotonic revision of this record. - pub revision: u64, - /// Owner-reported progress watermark. - pub progress: u64, - /// Durable lifecycle state. - pub state: ControlState, - /// Enrolled owner, absent while the Cell is idle. - pub owner: Option, - /// Published immutable root, absent before the first publication. - pub root: Option, - /// Recovered overlay pinned before the Cell moved. - pub recovery: Option, - /// Application code digest the owner installed. - pub code: Digest, - /// Schema version the owner installed. - pub schema: u32, - /// Logical time the owner asked to be renewed by. - pub next_due_ms: Option, -} - -/// Named transition whose complete predicate must pass before an ETag update. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum Transition { - /// Extends the current owner's lease without changing the root. - Renew, - /// Installs a new owner for an existing root. - Activate, - /// Advances the published root. - Publish, - /// Moves the Cell to a destination while the source keeps authority. - Migrate, - /// Returns the Cell to idle so another node may acquire it. - Release, - /// Pins a recovered overlay against the current published root. - AttachRecovery, - /// Publishes the recovered overlay as the new root. - PublishRecovery, - /// Fences the previous owner and installs this one. - Takeover, - /// Retires the Cell permanently. - Tombstone, -} - -impl Control { - /// Creates and validates the only legal first control record. - pub fn initial( - cell: CellId, - incarnation: IncarnationId, - owner: Owner, - code: Digest, - schema: u32, - ) -> Result { - let control = Self { - cell, - incarnation, - epoch: 1, - revision: 1, - progress: 1, - state: ControlState::Recovering, - owner: Some(owner), - root: None, - recovery: None, - code, - schema, - next_due_ms: None, - }; - control.validate()?; - Ok(control) - } - - /// Encodes canonical compact JSON after validating the complete record. - pub fn encode(&self) -> Result> { - self.validate()?; - let bytes = serde_json::to_vec(&codec::RawControl::from(self))?; - if bytes.len() > MAX_CONTROL_BYTES { - return Err(Error::Control("body exceeds 8 KiB")); - } - Ok(bytes) - } - - /// Decodes strict JSON, rejecting duplicate/unknown fields and noncanonical numbers. - pub fn decode(bytes: &[u8]) -> Result { - if bytes.len() > MAX_CONTROL_BYTES { - return Err(Error::Control("body exceeds 8 KiB")); - } - let raw: codec::RawControl = serde_json::from_slice(bytes)?; - let control = Self::try_from(raw)?; - control.validate()?; - if control.encode()? != bytes { - return Err(Error::Control("JSON is not canonical")); - } - Ok(control) - } - - /// Reattaches this control record's Cell scope to its immutable LTX root. - #[must_use] - pub fn ltx_root(&self) -> Option { - self.root - .as_ref() - .map(|root| root.to_ltx(self.cell, self.incarnation)) - } - - pub(crate) fn is_same_or_pure_renewal_of(&self, previous: &Self) -> bool { - let revisions = self.revision.checked_sub(previous.revision); - let progress = self.progress.checked_sub(previous.progress); - revisions.is_some() - && revisions == progress - && previous.cell == self.cell - && previous.incarnation == self.incarnation - && previous.epoch == self.epoch - && previous.state == self.state - && previous.owner == self.owner - && previous.root == self.root - && previous.recovery == self.recovery - && previous.code == self.code - && previous.schema == self.schema - && previous.next_due_ms == self.next_due_ms - } - - /// Builds the sole valid successor that proves this owner is still live. - pub(crate) fn renew(&self) -> Result { - let mut next = self.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Control("revision overflow"))?; - next.progress = next - .progress - .checked_add(1) - .ok_or(Error::Control("progress overflow"))?; - self.validate_transition(&next, Transition::Renew)?; - Ok(next) - } - - // Recovery becomes externally serving only after the restored root is verified. - pub(crate) fn activate(&self) -> Result { - let mut next = self.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Control("revision overflow"))?; - next.progress = next - .progress - .checked_add(1) - .ok_or(Error::Control("progress overflow"))?; - next.state = ControlState::Serving; - self.validate_transition(&next, Transition::Activate)?; - Ok(next) - } - - /// Builds a recovering successor owned by a different enrolled session. - pub fn takeover(&self, owner: Owner) -> Result { - let mut next = self.clone(); - next.epoch = next - .epoch - .checked_add(1) - .ok_or(Error::Control("epoch overflow"))?; - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Control("revision overflow"))?; - next.progress = next - .progress - .checked_add(1) - .ok_or(Error::Control("progress overflow"))?; - next.state = ControlState::Recovering; - next.owner = Some(owner); - self.validate_transition(&next, Transition::Takeover)?; - Ok(next) - } - - /// Pins an exact follower-recovered tail while retaining the dead owner. - pub fn attach_recovery(&self, recovery: RecoveryOverlayRef) -> Result { - let mut next = self.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Control("revision overflow"))?; - next.progress = next - .progress - .checked_add(1) - .ok_or(Error::Control("progress overflow"))?; - next.recovery = Some(recovery); - self.validate_transition(&next, Transition::AttachRecovery)?; - Ok(next) - } - - /// Publishes the root materialized from the currently pinned recovery tail. - pub fn publish_recovery( - &self, - prepared: &crab_ltx::PreparedRoot, - next_due_ms: Option, - ) -> Result { - let recovery = self - .recovery - .as_ref() - .ok_or(Error::Control("recovery overlay is not pinned"))?; - let expected = recovery.predecessor.to_ltx(self.cell, self.incarnation); - let root = prepared.root(); - if prepared.predecessor() != Some(expected) - || prepared.verified().schema() != self.schema - || root.position.txid != recovery.final_txid - || root.position.checksum != recovery.final_checksum - || root.commit_sequence != recovery.final_commit_sequence - { - return Err(Error::Control( - "prepared recovery does not match pinned overlay", - )); - } - let mut next = self.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Control("revision overflow"))?; - next.progress = next - .progress - .checked_add(1) - .ok_or(Error::Control("progress overflow"))?; - next.root = Some(RootRef::from_ltx(self.cell, self.incarnation, root)?); - next.recovery = None; - next.next_due_ms = next_due_ms; - self.validate_transition(&next, Transition::PublishRecovery)?; - Ok(next) - } - - /// Builds the sole valid publication successor for an uploaded root proposal. - pub fn publish_prepared( - &self, - prepared: &crab_ltx::PreparedRoot, - next_due_ms: Option, - ) -> Result { - if prepared.predecessor() != self.ltx_root() || prepared.verified().schema() != self.schema - { - return Err(Error::Control("prepared root does not continue control")); - } - let mut next = self.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Control("revision overflow"))?; - next.progress = next - .progress - .checked_add(1) - .ok_or(Error::Control("progress overflow"))?; - next.state = ControlState::Serving; - next.root = Some(RootRef::from_ltx( - self.cell, - self.incarnation, - prepared.root(), - )?); - next.next_due_ms = next_due_ms; - self.validate_transition(&next, Transition::Publish)?; - Ok(next) - } - - /// Builds the sole valid successor for one published schema migration. - pub fn migrate_prepared( - &self, - prepared: &crab_ltx::PreparedRoot, - next_due_ms: Option, - code: Digest, - schema: u32, - ) -> Result { - if prepared.predecessor() != self.ltx_root() - || prepared.verified().schema() != schema - || code.as_bytes().iter().all(|byte| *byte == 0) - || !valid_migration_version(self.code, self.schema, code, schema) - { - return Err(Error::Control( - "prepared migration does not continue control", - )); - } - let mut next = self.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Control("revision overflow"))?; - next.progress = next - .progress - .checked_add(1) - .ok_or(Error::Control("progress overflow"))?; - next.state = ControlState::Serving; - next.root = Some(RootRef::from_ltx( - self.cell, - self.incarnation, - prepared.root(), - )?); - next.code = code; - next.schema = schema; - next.next_due_ms = next_due_ms; - self.validate_transition(&next, Transition::Migrate)?; - Ok(next) - } - - /// Builds the sole valid successor that releases a drained Cell owner. - pub(crate) fn release(&self) -> Result { - let mut next = self.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Control("revision overflow"))?; - next.progress = next - .progress - .checked_add(1) - .ok_or(Error::Control("progress overflow"))?; - next.state = ControlState::Idle; - next.owner = None; - self.validate_transition(&next, Transition::Release)?; - Ok(next) - } - - /// Validates a named successor before its conditional object-store update. - pub fn validate_transition(&self, next: &Self, transition: Transition) -> Result<()> { - self.validate()?; - next.validate()?; - if self.cell != next.cell || self.incarnation != next.incarnation { - return Err(Error::Control("successor changed Cell scope")); - } - if self.revision.checked_add(1) != Some(next.revision) - || self.progress.checked_add(1) != Some(next.progress) - { - return Err(Error::Control("successor revision or progress")); - } - match transition { - Transition::Renew => { - if self.epoch != next.epoch - || self.state != next.state - || self.owner != next.owner - || self.root != next.root - || self.recovery != next.recovery - || self.code != next.code - || self.schema != next.schema - || self.next_due_ms != next.next_due_ms - { - return Err(Error::Control("renew changed protected fields")); - } - } - Transition::Activate => { - if self.state != ControlState::Recovering - || next.state != ControlState::Serving - || self.epoch != next.epoch - || self.owner != next.owner - || self.root.is_none() - || self.root != next.root - || self.recovery.is_some() - || next.recovery.is_some() - || self.code != next.code - || self.schema != next.schema - || self.next_due_ms != next.next_due_ms - { - return Err(Error::Control("invalid activation transition")); - } - } - Transition::Publish => { - if !matches!(self.state, ControlState::Recovering | ControlState::Serving) - || next.state != ControlState::Serving - || self.epoch != next.epoch - || self.owner != next.owner - || next.root.is_none() - || self.recovery.is_some() - || next.recovery.is_some() - || self.code != next.code - || self.schema != next.schema - || !valid_root_successor(self.root.as_ref(), next.root.as_ref()) - { - return Err(Error::Control("invalid publish transition")); - } - } - Transition::Migrate => { - if !matches!(self.state, ControlState::Recovering | ControlState::Serving) - || next.state != ControlState::Serving - || self.epoch != next.epoch - || self.owner != next.owner - || next.root.is_none() - || self.recovery.is_some() - || next.recovery.is_some() - || next.code.as_bytes().iter().all(|byte| *byte == 0) - || !valid_migration_version(self.code, self.schema, next.code, next.schema) - || !valid_root_successor(self.root.as_ref(), next.root.as_ref()) - { - return Err(Error::Control("invalid migration transition")); - } - } - Transition::Release => { - if !matches!(self.state, ControlState::Recovering | ControlState::Serving) - || next.state != ControlState::Idle - || next.owner.is_some() - || next.root.is_none() - || self.recovery.is_some() - || next.recovery.is_some() - || self.epoch != next.epoch - || self.root != next.root - || self.recovery != next.recovery - || self.code != next.code - || self.schema != next.schema - || self.next_due_ms != next.next_due_ms - { - return Err(Error::Control("invalid release transition")); - } - } - Transition::AttachRecovery => { - let Some(recovery) = next.recovery.as_ref() else { - return Err(Error::Control("invalid recovery attachment")); - }; - if self.state == ControlState::Tombstoned - || self.recovery.is_some() - || self.epoch != next.epoch - || self.state != next.state - || self.owner.as_ref().map(|owner| owner.session) - != Some(recovery.leader_session) - || self.owner != next.owner - || self.root.as_ref() != Some(&recovery.predecessor) - || self.root != next.root - || self.code != next.code - || self.schema != next.schema - || self.next_due_ms != next.next_due_ms - { - return Err(Error::Control("invalid recovery attachment")); - } - } - Transition::PublishRecovery => { - let Some(recovery) = self.recovery.as_ref() else { - return Err(Error::Control("invalid recovery publication")); - }; - if self.state != ControlState::Recovering - || next.state != ControlState::Recovering - || next.recovery.is_some() - || self.epoch != next.epoch - || self.owner != next.owner - || self.root.as_ref() != Some(&recovery.predecessor) - || next.root.as_ref().is_none_or(|root| { - root.txid != recovery.final_txid - || root.checksum != recovery.final_checksum - || root.commit_sequence != recovery.final_commit_sequence - }) - || self.code != next.code - || self.schema != next.schema - { - return Err(Error::Control("invalid recovery publication")); - } - } - Transition::Takeover => { - if self.state == ControlState::Tombstoned - || next.state != ControlState::Recovering - || next.owner.is_none() - || self.owner == next.owner - || self.epoch.checked_add(1) != Some(next.epoch) - || self.root != next.root - || self.recovery != next.recovery - || self.code != next.code - || self.schema != next.schema - || self.next_due_ms != next.next_due_ms - { - return Err(Error::Control("invalid takeover transition")); - } - } - Transition::Tombstone => { - if self.state == ControlState::Tombstoned - || next.state != ControlState::Tombstoned - || next.owner.is_some() - || self.epoch.checked_add(1) != Some(next.epoch) - || self.root != next.root - || self.recovery.is_some() - || next.recovery.is_some() - || self.code != next.code - || self.schema != next.schema - || self.next_due_ms != next.next_due_ms - { - return Err(Error::Control("invalid tombstone transition")); - } - } - } - Ok(()) - } - - fn validate(&self) -> Result<()> { - if self.epoch == 0 || self.revision == 0 || self.schema == 0 { - return Err(Error::Control("zero epoch, revision, or schema")); - } - if self.next_due_ms.is_some_and(|value| value < 0) { - return Err(Error::Control("negative next due time")); - } - if let Some(owner) = &self.owner - && (owner.endpoint.is_empty() - || owner.endpoint.len() > 512 - || owner.endpoint.chars().any(char::is_control)) - { - return Err(Error::Control("invalid owner endpoint")); - } - if let Some(root) = &self.root - && (root.txid == 0 - || root.checksum & CHECKSUM_FLAG == 0 - || root.commit_sequence > i64::MAX as u64) - { - return Err(Error::Control("invalid recovery root")); - } - if let Some(recovery) = &self.recovery { - recovery.validate()?; - if self.root.as_ref() != Some(&recovery.predecessor) { - return Err(Error::Control("recovery overlay does not match control")); - } - } - match self.state { - ControlState::Recovering if self.owner.is_some() => {} - ControlState::Serving if self.owner.is_some() && self.root.is_some() => {} - ControlState::Idle | ControlState::Tombstoned if self.owner.is_none() => {} - _ => return Err(Error::Control("owner/root do not match state")), - } - Ok(()) - } -} - -fn valid_root_successor(previous: Option<&RootRef>, next: Option<&RootRef>) -> bool { - let Some(next) = next else { - return false; - }; - let Some(previous) = previous else { - return true; - }; - if next.txid < previous.txid || next.commit_sequence < previous.commit_sequence { - return false; - } - next.txid != previous.txid - || (next.checksum == previous.checksum && next.commit_sequence == previous.commit_sequence) -} - -fn valid_migration_version( - current_code: Digest, - current_schema: u32, - next_code: Digest, - next_schema: u32, -) -> bool { - current_schema.checked_add(1) == Some(next_schema) - || (current_schema == next_schema && current_code != next_code) -} diff --git a/crates/crab-cell-runtime/src/control/authority.rs b/crates/crab-cell-runtime/src/control/authority.rs deleted file mode 100644 index fa7a8c24e..000000000 --- a/crates/crab-cell-runtime/src/control/authority.rs +++ /dev/null @@ -1,177 +0,0 @@ -//! Versioned control records and the ETag CAS that owns them. -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::{ETag, StorageError}; - -use crate::cell::catalog::CatalogProof; -use crate::control::{Control, Owner, Transition}; -use crate::identity::CellId; -use crate::identity::IncarnationId; -use crate::{Error, Result}; - -const MAX_CONTROL_BYTES: u64 = 8 * 1024; - -/// One exact control observation and its conditional-write token. -#[derive(Clone)] -pub struct VersionedControl { - value: Control, - token: ETag, -} - -impl VersionedControl { - /// Returns the control record this version carries. - #[must_use] - pub const fn value(&self) -> &Control { - &self.value - } -} - -/// Loads and conditionally changes Cell owner/root records. -/// -/// Catalog provisioning and takeover timing remain separate policy. Every write -/// must name the observed record and a validated transition; there is no blind -/// overwrite API. -#[derive(Clone)] -pub struct CellAuthority { - layout: CellStorageLayout, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, -} - -impl CellAuthority { - /// Binds the authority to one Cell storage layout. - #[must_use] - pub fn new(layout: CellStorageLayout) -> Self { - Self::with_telemetry( - layout, - crate::fleet::telemetry::CellTelemetryHandle::default(), - ) - } - - /// Binds the authority to one Cell storage layout and telemetry sink. - #[must_use] - pub fn with_telemetry( - layout: CellStorageLayout, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, - ) -> Self { - Self { layout, telemetry } - } - - /// Returns the Cell storage layout this authority reads and writes. - #[must_use] - pub const fn layout(&self) -> &CellStorageLayout { - &self.layout - } - - /// Strict-creates a bootstrap control only after catalog publication. - pub async fn create_initial( - &self, - catalog: &CatalogProof, - incarnation: IncarnationId, - owner: Owner, - ) -> Result { - let entry = catalog.entry(); - let initial = Control::initial( - entry.cell(), - incarnation, - owner, - entry.initial_code(), - entry.initial_schema(), - )?; - let path = self.layout.control_path(entry.cell().as_bytes()); - match self - .layout - .store() - .create_strict_with_etag(&path, Bytes::from(initial.encode()?)) - .await - { - Ok(token) => Ok(VersionedControl { - value: initial, - token, - }), - Err(create_error) => match self.load(entry.cell()).await? { - Some(current) if current.value == initial => Ok(current), - Some(_) => Err(Error::CellAlreadyActive), - None => Err(create_error.into()), - }, - } - } - - /// Reads one exact control object; absence is not inferred from listing. - pub async fn load(&self, cell: CellId) -> Result> { - let path = self.layout.control_path(cell.as_bytes()); - let started = std::time::Instant::now(); - let observed = self - .layout - .store() - .get_with_etag_bounded(&path, MAX_CONTROL_BYTES) - .await; - let (body, token) = match observed { - Ok(observed) => observed, - Err(StorageError::NotFound { .. }) => { - self.telemetry.control_read(started.elapsed(), true); - return Ok(None); - } - Err(error) => { - self.telemetry.control_read(started.elapsed(), false); - return Err(error.into()); - } - }; - self.telemetry.control_read(started.elapsed(), true); - let value = Control::decode(&body)?; - if value.cell != cell { - return Err(crate::Error::Control("control path does not match Cell ID")); - } - Ok(Some(VersionedControl { value, token })) - } - - pub(crate) async fn install_restored(&self, control: Control) -> Result { - if control.owner.is_some() - || !matches!( - control.state, - crate::control::ControlState::Idle | crate::control::ControlState::Tombstoned - ) - || (control.state == crate::control::ControlState::Idle && control.root.is_none()) - { - return Err(Error::Control("restored control is not safely unowned")); - } - let path = self.layout.control_path(control.cell.as_bytes()); - match self - .layout - .store() - .create_strict_with_etag(&path, Bytes::from(control.encode()?)) - .await - { - Ok(token) => Ok(VersionedControl { - value: control, - token, - }), - Err(create_error) => match self.load(control.cell).await? { - Some(current) if current.value == control => Ok(current), - Some(_) => Err(Error::Control( - "restored control conflicts with existing authority", - )), - None => Err(create_error.into()), - }, - } - } - - /// Applies one complete successor with the exact observed ETag. - /// - /// A storage conflict is returned to the coordinator for reload and full - /// predicate revalidation. It is never retried here with a stale successor. - pub async fn transition( - &self, - observed: &VersionedControl, - next: Control, - transition: Transition, - ) -> Result { - observed.value.validate_transition(&next, transition)?; - let path = self.layout.control_path(observed.value.cell.as_bytes()); - let token = self - .layout - .store() - .update(&path, Bytes::from(next.encode()?), observed.token.clone()) - .await?; - Ok(VersionedControl { value: next, token }) - } -} diff --git a/crates/crab-cell-runtime/src/control/codec.rs b/crates/crab-cell-runtime/src/control/codec.rs deleted file mode 100644 index 7adf31fab..000000000 --- a/crates/crab-cell-runtime/src/control/codec.rs +++ /dev/null @@ -1,202 +0,0 @@ -//! Canonical JSON codec for the Cell control record. - -use serde::{Deserialize, Serialize}; - -use super::{Control, ControlState, Owner, RecoveryOverlayRef, RootRef}; -use crate::identity::{CellId, Digest, IncarnationId, SessionId, decode_hex, encode_hex}; -use crate::{Error, Result}; - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(super) struct RawControl { - version: u32, - cell: String, - incarnation: String, - epoch: String, - revision: String, - progress: String, - state: ControlState, - owner: Option, - root: Option, - recovery: Option, - code: String, - schema: u32, - next_due_ms: Option, -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(super) struct RawOwner { - session: String, - endpoint: String, -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(super) struct RawRoot { - digest: String, - txid: String, - checksum: String, - commit_sequence: String, -} - -impl From<&RootRef> for RawRoot { - fn from(root: &RootRef) -> Self { - Self { - digest: encode_hex(root.digest.as_bytes()), - txid: root.txid.to_string(), - checksum: encode_hex(&root.checksum.to_be_bytes()), - commit_sequence: root.commit_sequence.to_string(), - } - } -} - -impl TryFrom for RootRef { - type Error = Error; - - fn try_from(root: RawRoot) -> Result { - Ok(Self { - digest: Digest::from_bytes(decode_hex(&root.digest)?), - txid: decimal_u64(&root.txid)?, - checksum: u64::from_be_bytes(decode_hex(&root.checksum)?), - commit_sequence: decimal_u64(&root.commit_sequence)?, - }) - } -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub(super) struct RawRecoveryOverlay { - leader_session: String, - log_epoch: String, - manifest_digest: String, - first_node_sequence: String, - last_node_sequence: String, - predecessor: RawRoot, - final_txid: String, - final_checksum: String, - final_commit_sequence: String, -} - -impl From<&Control> for RawControl { - fn from(control: &Control) -> Self { - Self { - version: 1, - cell: encode_hex(control.cell.as_bytes()), - incarnation: encode_hex(control.incarnation.as_bytes()), - epoch: control.epoch.to_string(), - revision: control.revision.to_string(), - progress: control.progress.to_string(), - state: control.state, - owner: control.owner.as_ref().map(|owner| RawOwner { - session: encode_hex(owner.session.as_bytes()), - endpoint: owner.endpoint.clone(), - }), - root: control.root.as_ref().map(|root| RawRoot { - digest: encode_hex(root.digest.as_bytes()), - txid: root.txid.to_string(), - checksum: encode_hex(&root.checksum.to_be_bytes()), - commit_sequence: root.commit_sequence.to_string(), - }), - recovery: control - .recovery - .as_ref() - .map(|recovery| RawRecoveryOverlay { - leader_session: encode_hex(recovery.leader_session.as_bytes()), - log_epoch: recovery.log_epoch.to_string(), - manifest_digest: encode_hex(recovery.manifest_digest.as_bytes()), - first_node_sequence: recovery.first_node_sequence.to_string(), - last_node_sequence: recovery.last_node_sequence.to_string(), - predecessor: RawRoot::from(&recovery.predecessor), - final_txid: recovery.final_txid.to_string(), - final_checksum: encode_hex(&recovery.final_checksum.to_be_bytes()), - final_commit_sequence: recovery.final_commit_sequence.to_string(), - }), - code: encode_hex(control.code.as_bytes()), - schema: control.schema, - next_due_ms: control.next_due_ms.map(|value| value.to_string()), - } - } -} - -impl TryFrom for Control { - type Error = Error; - - fn try_from(raw: RawControl) -> Result { - if raw.version != 1 { - return Err(Error::Control("unsupported version")); - } - Ok(Self { - cell: CellId::from_bytes(decode_hex(&raw.cell)?), - incarnation: IncarnationId::from_bytes(decode_hex(&raw.incarnation)?), - epoch: decimal_u64(&raw.epoch)?, - revision: decimal_u64(&raw.revision)?, - progress: decimal_u64(&raw.progress)?, - state: raw.state, - owner: raw - .owner - .map(|owner| { - Ok::(Owner { - session: SessionId::from_bytes(decode_hex(&owner.session)?), - endpoint: owner.endpoint, - }) - }) - .transpose()?, - root: raw - .root - .map(|root| { - Ok::(RootRef { - digest: Digest::from_bytes(decode_hex(&root.digest)?), - txid: decimal_u64(&root.txid)?, - checksum: u64::from_be_bytes(decode_hex(&root.checksum)?), - commit_sequence: decimal_u64(&root.commit_sequence)?, - }) - }) - .transpose()?, - recovery: raw - .recovery - .map(|recovery| { - Ok::(RecoveryOverlayRef { - leader_session: SessionId::from_bytes(decode_hex( - &recovery.leader_session, - )?), - log_epoch: decimal_u64(&recovery.log_epoch)?, - manifest_digest: Digest::from_bytes(decode_hex(&recovery.manifest_digest)?), - first_node_sequence: decimal_u64(&recovery.first_node_sequence)?, - last_node_sequence: decimal_u64(&recovery.last_node_sequence)?, - predecessor: RootRef::try_from(recovery.predecessor)?, - final_txid: decimal_u64(&recovery.final_txid)?, - final_checksum: u64::from_be_bytes(decode_hex(&recovery.final_checksum)?), - final_commit_sequence: decimal_u64(&recovery.final_commit_sequence)?, - }) - }) - .transpose()?, - code: Digest::from_bytes(decode_hex(&raw.code)?), - schema: raw.schema, - next_due_ms: raw - .next_due_ms - .map(|value| decimal_i64(&value)) - .transpose()?, - }) - } -} - -fn decimal_u64(value: &str) -> Result { - let number = value - .parse::() - .map_err(|_| Error::Control("invalid decimal u64"))?; - if number.to_string() != value { - return Err(Error::Control("noncanonical decimal u64")); - } - Ok(number) -} - -fn decimal_i64(value: &str) -> Result { - let number = value - .parse::() - .map_err(|_| Error::Control("invalid decimal i64"))?; - if number.to_string() != value { - return Err(Error::Control("noncanonical decimal i64")); - } - Ok(number) -} diff --git a/crates/crab-cell-runtime/src/control/tests.rs b/crates/crab-cell-runtime/src/control/tests.rs deleted file mode 100644 index 054029ec8..000000000 --- a/crates/crab-cell-runtime/src/control/tests.rs +++ /dev/null @@ -1,171 +0,0 @@ -//! Private control transition and codec behavior. - -use super::*; - -fn owner(byte: u8) -> Owner { - Owner { - session: SessionId::from_bytes([byte; 16]), - endpoint: format!("https://node-{byte}.internal:8081"), - } -} - -fn root(sequence: u64) -> RootRef { - RootRef { - digest: Digest::from_bytes([9; 32]), - txid: sequence, - checksum: CHECKSUM_FLAG | sequence, - commit_sequence: sequence, - } -} - -fn recovery(predecessor: RootRef) -> RecoveryOverlayRef { - RecoveryOverlayRef { - leader_session: SessionId::from_bytes([3; 16]), - log_epoch: 4, - manifest_digest: Digest::from_bytes([7; 32]), - first_node_sequence: 10, - last_node_sequence: 12, - predecessor, - final_txid: 9, - final_checksum: CHECKSUM_FLAG | 9, - final_commit_sequence: 9, - } -} - -fn initial() -> Control { - Control::initial( - CellId::from_bytes([1; 32]), - IncarnationId::from_bytes([2; 16]), - owner(3), - Digest::from_bytes([4; 32]), - 1, - ) - .unwrap() -} - -#[test] -fn canonical_control_roundtrips_and_rejects_alternate_encodings() { - let mut control = initial(); - control.root = Some(root(7)); - control.state = ControlState::Serving; - let bytes = control.encode().unwrap(); - assert_eq!(Control::decode(&bytes).unwrap(), control); - - let text = String::from_utf8(bytes).unwrap(); - assert!( - Control::decode( - text.replace("\"epoch\":\"1\"", "\"epoch\":\"01\"") - .as_bytes() - ) - .is_err() - ); - assert!( - Control::decode( - text.replace("\"schema\":1", "\"schema\":1,\"schema\":1") - .as_bytes() - ) - .is_err() - ); - assert!(Control::decode(format!("{{\"unknown\":1,{}}}", &text[1..]).as_bytes()).is_err()); -} - -#[test] -fn transitions_protect_scope_owner_and_published_root() { - let recovering = initial(); - let mut serving = recovering.clone(); - serving.state = ControlState::Serving; - serving.root = Some(root(1)); - serving.revision += 1; - serving.progress += 1; - recovering - .validate_transition(&serving, Transition::Publish) - .unwrap(); - let mut unlabelled_migration = serving.clone(); - unlabelled_migration.schema = 2; - unlabelled_migration.root = Some(root(2)); - unlabelled_migration.revision += 1; - unlabelled_migration.progress += 1; - assert!( - serving - .validate_transition(&unlabelled_migration, Transition::Publish) - .is_err() - ); - serving - .validate_transition(&unlabelled_migration, Transition::Migrate) - .unwrap(); - let mut code_only = serving.clone(); - code_only.code = Digest::from_bytes([8; 32]); - code_only.root = Some(root(2)); - code_only.revision += 1; - code_only.progress += 1; - serving - .validate_transition(&code_only, Transition::Migrate) - .unwrap(); - - let renewed = serving.renew().unwrap(); - assert!(serving.is_same_or_pure_renewal_of(&serving)); - assert!(renewed.is_same_or_pure_renewal_of(&serving)); - let mut skipped_progress = renewed.clone(); - skipped_progress.revision += 1; - assert!(!skipped_progress.is_same_or_pure_renewal_of(&serving)); - - let takeover = renewed.takeover(owner(5)).unwrap(); - - let activated = takeover.activate().unwrap(); - assert_eq!(activated.state, ControlState::Serving); - assert_eq!(activated.root, takeover.root); - assert!(takeover.activate().is_ok()); - assert!(activated.activate().is_err()); - - let rootless = initial(); - assert!(rootless.activate().is_err()); - - let mut corrupted_compaction = takeover.clone(); - corrupted_compaction.state = ControlState::Serving; - corrupted_compaction.root.as_mut().unwrap().checksum ^= 1; - corrupted_compaction.revision += 1; - corrupted_compaction.progress += 1; - assert!( - takeover - .validate_transition(&corrupted_compaction, Transition::Publish) - .is_err() - ); -} - -#[test] -fn recovery_attachment_is_canonical_and_survives_takeover() { - let mut serving = initial(); - serving.state = ControlState::Serving; - serving.root = Some(root(4)); - let attached = serving - .attach_recovery(recovery(serving.root.clone().unwrap())) - .unwrap(); - assert_eq!( - Control::decode(&attached.encode().unwrap()).unwrap(), - attached - ); - let mut normal_publish = attached.clone(); - normal_publish.revision += 1; - normal_publish.progress += 1; - normal_publish.root = Some(root(9)); - assert!( - attached - .validate_transition(&normal_publish, Transition::Publish) - .is_err() - ); - - let takeover = attached.takeover(owner(8)).unwrap(); - assert_eq!(takeover.state, ControlState::Recovering); - assert_eq!(takeover.recovery, attached.recovery); - assert!(takeover.activate().is_err()); - - let mut changed = attached.clone(); - changed.revision += 1; - changed.progress += 1; - changed.recovery.as_mut().unwrap().manifest_digest = Digest::from_bytes([8; 32]); - assert!( - attached - .validate_transition(&changed, Transition::Renew) - .is_err() - ); -} diff --git a/crates/crab-cell-runtime/src/coordination.rs b/crates/crab-cell-runtime/src/coordination.rs deleted file mode 100644 index f36f7204f..000000000 --- a/crates/crab-cell-runtime/src/coordination.rs +++ /dev/null @@ -1,759 +0,0 @@ -use std::{collections::BTreeMap, fmt}; - -#[cfg(test)] -mod sim; - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(crate) enum AdmissionKind { - Command, - Query, - Resolve, - Migration, -} - -/// The adapter intent associated with one in-flight effect. -/// -/// Keeping the intent in the pure state prevents a completion from one async -/// family (for example, a renewal) from releasing a different family that -/// happens to reuse its local effect identity. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(crate) enum CoordinationEffect { - Work(AdmissionKind), - Hydration, - Inventory, - Compaction, - Publication, - Proof, - Renewal, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(crate) enum RejectReason { - NotActive, - Fenced, - Draining, - Busy, - PublicationPending, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(crate) enum CoordinationInput { - Admit { - kind: AdmissionKind, - admission_matches: bool, - }, - #[cfg_attr( - not(test), - expect( - dead_code, - reason = "pure lookup transition is exercised by the simulator" - ) - )] - Lookup, - BeginWork { - kind: AdmissionKind, - publisher_ready: bool, - }, - /// Schedules the next adapter action from actor-owned queue observations. - /// - /// The actor supplies queue/publisher observations; lifecycle, busy, - /// renewal, and fencing policy remains owned by this pure state machine. - Schedule { - queue_empty: bool, - publisher_ready: bool, - publication_blocked: bool, - lease_live: bool, - }, - CompleteEffect { - effect_id: u64, - effect: CoordinationEffect, - }, - FinishWork { - fenced: bool, - }, - BeginDrain, - BeginShutdown, - BeginMigration, - FinishMigration { - fenced: bool, - }, - BeginRenewal { - queue_empty: bool, - publication_idle: bool, - lease_live: bool, - }, - FinishRenewal { - fenced: bool, - }, - BeginPublication, - #[cfg_attr( - not(test), - expect( - dead_code, - reason = "caller cancellation is exercised by the pure transition tests" - ) - )] - CallerCancel, - #[cfg_attr( - not(test), - expect( - dead_code, - reason = "follower proof is exercised by the pure transition tests" - ) - )] - FollowerProof { - accepted: bool, - }, - FinishPublication { - fenced: bool, - succeeded: bool, - }, - BeginHydration { - queue_empty: bool, - publication_idle: bool, - lease_live: bool, - }, - BeginInventory { - queue_empty: bool, - publication_idle: bool, - inventory_unknown: bool, - refreshing: bool, - lease_live: bool, - }, - BeginTransferPreflight { - queue_empty: bool, - publication_idle: bool, - lease_live: bool, - }, - ConfirmTransfer, - AbortTransfer, - BeginCompaction { - queue_empty: bool, - publication_idle: bool, - publisher_ready: bool, - due: bool, - lease_live: bool, - }, - FinishCompaction { - fenced: bool, - }, - FinishHydration { - complete: bool, - stale: bool, - }, - Fence, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(crate) enum CoordinationDecision { - Admit, - ResolveUnknown, - LocalHandle, - Reject(RejectReason), - Started, - EffectCompleted, - StaleEffect, - Ignored, - ReadyToDeactivate, - ReadyToDeactivateFenced, - StartQueuedWork, - Fence, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum Lifecycle { - Serving, - Fenced, - Draining, - Migrating, - Shutdown, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(crate) enum Residency { - Sparse, - Hydrating, - Resident, -} - -/// Pure volatile protocol state for one active Cell. -/// -/// The state owns lifecycle and in-flight flags. It does not contain Tokio, -/// storage, clocks, SQL handles, bytes, or provider errors; `actor.rs` maps -/// its decisions to those effects. -#[derive(Clone, Debug, PartialEq, Eq)] -pub(crate) struct CoordinationState { - lifecycle: Lifecycle, - busy: bool, - renewing: bool, - publications: usize, - follower_proof: bool, - residency: Residency, - shutdown_requested: bool, - transfer_preparing: bool, - next_effect_id: u64, - pending_effects: BTreeMap, -} - -impl CoordinationState { - #[cfg_attr( - not(test), - expect( - dead_code, - reason = "test and simulator constructor for the canonical kernel" - ) - )] - pub(crate) const fn serving(_publisher_ready: bool) -> Self { - Self { - lifecycle: Lifecycle::Serving, - busy: false, - renewing: false, - publications: 0, - follower_proof: false, - residency: Residency::Resident, - shutdown_requested: false, - transfer_preparing: false, - next_effect_id: 0, - pending_effects: BTreeMap::new(), - } - } - - pub(crate) const fn serving_with_residency( - _publisher_ready: bool, - residency: Residency, - ) -> Self { - Self { - lifecycle: Lifecycle::Serving, - busy: false, - renewing: false, - publications: 0, - follower_proof: false, - residency, - shutdown_requested: false, - transfer_preparing: false, - next_effect_id: 0, - pending_effects: BTreeMap::new(), - } - } - - pub(crate) fn is_fenced(&self) -> bool { - matches!(self.lifecycle, Lifecycle::Fenced) - } - - pub(crate) fn is_draining(&self) -> bool { - self.shutdown_requested - || matches!( - self.lifecycle, - Lifecycle::Draining | Lifecycle::Migrating | Lifecycle::Shutdown - ) - } - - pub(crate) fn is_shutdown(&self) -> bool { - self.shutdown_requested || matches!(self.lifecycle, Lifecycle::Shutdown) - } - - pub(crate) fn is_transfer_preparing(&self) -> bool { - self.transfer_preparing - } - - pub(crate) fn is_busy(&self) -> bool { - self.busy - } - - pub(crate) fn is_renewing(&self) -> bool { - self.renewing - } - - pub(crate) fn publication_count(&self) -> usize { - self.publications - } - - #[cfg_attr( - not(test), - expect( - dead_code, - reason = "follower proof is asserted by the pure transition tests" - ) - )] - pub(crate) fn follower_proof(&self) -> bool { - self.follower_proof - } - - pub(crate) fn residency(&self) -> Residency { - self.residency - } - - /// Allocates a collision-free effect identity for one active generation. - /// - /// The identity is local to this coordination state. Completions must be - /// submitted through `CompleteEffect` with the same intent; unknown, - /// mismatched, or duplicate identities cannot release a newer operation. - pub(crate) fn begin_effect(&mut self, effect: CoordinationEffect) -> u64 { - loop { - self.next_effect_id = self.next_effect_id.wrapping_add(1).max(1); - if !self.pending_effects.contains_key(&self.next_effect_id) { - self.pending_effects.insert(self.next_effect_id, effect); - return self.next_effect_id; - } - } - } - - pub(crate) fn effect_matches(&self, effect_id: u64, effect: CoordinationEffect) -> bool { - self.pending_effects.get(&effect_id) == Some(&effect) - } - - pub(crate) fn lookup(&self) -> CoordinationDecision { - if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_draining() || self.transfer_preparing { - CoordinationDecision::Reject(RejectReason::Draining) - } else { - CoordinationDecision::LocalHandle - } - } - - pub(crate) fn can_deactivate(&self) -> bool { - !self.busy && !self.renewing && self.publications == 0 && self.pending_effects.is_empty() - } - - /// Combines kernel-owned quiescence with adapter observations needed to - /// close a Cell. The adapter may report queue and publisher state, but it - /// cannot reimplement the lifecycle predicates above. - pub(crate) fn ready_to_deactivate(&self, queue_empty: bool, publisher_ready: bool) -> bool { - queue_empty && publisher_ready && self.can_deactivate() - } - - pub(crate) fn step(&mut self, input: CoordinationInput) -> CoordinationDecision { - match input { - CoordinationInput::Admit { - kind, - admission_matches, - } => self.admit(kind, admission_matches), - CoordinationInput::Lookup => self.lookup(), - CoordinationInput::BeginWork { - kind, - publisher_ready, - } => { - if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.busy { - CoordinationDecision::Reject(RejectReason::Busy) - } else if matches!(kind, AdmissionKind::Migration) - && (self.publications != 0 || !publisher_ready) - { - CoordinationDecision::Reject(RejectReason::PublicationPending) - } else { - self.busy = true; - CoordinationDecision::Started - } - } - CoordinationInput::Schedule { - queue_empty, - publisher_ready, - publication_blocked, - lease_live, - } => { - if !lease_live { - self.lifecycle = Lifecycle::Fenced; - self.busy = false; - self.renewing = false; - if self.ready_to_deactivate(queue_empty, publisher_ready) { - CoordinationDecision::ReadyToDeactivateFenced - } else { - CoordinationDecision::Fence - } - } else if (self.is_fenced() || self.is_draining()) - && self.ready_to_deactivate(queue_empty, publisher_ready) - { - if self.is_fenced() { - CoordinationDecision::ReadyToDeactivateFenced - } else { - CoordinationDecision::ReadyToDeactivate - } - } else if self.is_fenced() - || self.busy - || self.renewing - || self - .pending_effects - .values() - .any(|effect| matches!(effect, CoordinationEffect::Inventory)) - || queue_empty - || publication_blocked - { - CoordinationDecision::Ignored - } else { - CoordinationDecision::StartQueuedWork - } - } - CoordinationInput::CompleteEffect { effect_id, effect } => { - if self.pending_effects.get(&effect_id) == Some(&effect) { - self.pending_effects.remove(&effect_id); - CoordinationDecision::EffectCompleted - } else { - CoordinationDecision::StaleEffect - } - } - CoordinationInput::FinishWork { fenced } => { - self.busy = false; - if fenced || self.is_fenced() { - self.lifecycle = Lifecycle::Fenced; - return CoordinationDecision::Fence; - } - if self.can_deactivate() && self.is_draining() { - CoordinationDecision::ReadyToDeactivate - } else { - CoordinationDecision::Ignored - } - } - CoordinationInput::BeginDrain => { - if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_draining() { - CoordinationDecision::Reject(RejectReason::Draining) - } else { - self.lifecycle = Lifecycle::Draining; - if self.can_deactivate() { - CoordinationDecision::ReadyToDeactivate - } else { - CoordinationDecision::Started - } - } - } - CoordinationInput::BeginShutdown => { - self.transfer_preparing = false; - if self.is_shutdown() { - CoordinationDecision::Ignored - } else if self.is_fenced() { - self.shutdown_requested = true; - if self.can_deactivate() { - CoordinationDecision::ReadyToDeactivate - } else { - CoordinationDecision::Started - } - } else if matches!(self.lifecycle, Lifecycle::Migrating) { - self.shutdown_requested = true; - CoordinationDecision::Started - } else { - self.shutdown_requested = true; - self.lifecycle = Lifecycle::Shutdown; - if self.can_deactivate() { - CoordinationDecision::ReadyToDeactivate - } else { - CoordinationDecision::Started - } - } - } - CoordinationInput::BeginMigration => { - if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_draining() { - CoordinationDecision::Reject(RejectReason::Draining) - } else { - self.lifecycle = Lifecycle::Migrating; - CoordinationDecision::Started - } - } - CoordinationInput::FinishMigration { fenced } => { - self.busy = false; - if fenced || self.is_fenced() { - self.lifecycle = Lifecycle::Fenced; - return CoordinationDecision::Fence; - } - if matches!(self.lifecycle, Lifecycle::Migrating) { - self.lifecycle = if self.shutdown_requested { - Lifecycle::Shutdown - } else { - Lifecycle::Serving - }; - if self.can_deactivate() && self.is_draining() { - return CoordinationDecision::ReadyToDeactivate; - } - } - CoordinationDecision::Ignored - } - CoordinationInput::BeginRenewal { - queue_empty, - publication_idle, - lease_live, - } => { - if !lease_live { - self.lifecycle = Lifecycle::Fenced; - self.busy = false; - self.renewing = false; - CoordinationDecision::Fence - } else if self.is_fenced() - || self.is_draining() - || self.transfer_preparing - || self.renewing - { - CoordinationDecision::Reject(if self.is_fenced() { - RejectReason::Fenced - } else { - RejectReason::Draining - }) - } else if self.busy || !queue_empty || !publication_idle { - CoordinationDecision::Ignored - } else { - self.renewing = true; - CoordinationDecision::Started - } - } - CoordinationInput::FinishRenewal { fenced } => { - self.renewing = false; - if fenced || self.is_fenced() { - self.lifecycle = Lifecycle::Fenced; - CoordinationDecision::Fence - } else { - CoordinationDecision::Ignored - } - } - CoordinationInput::BeginPublication => { - if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_draining() && !self.busy { - CoordinationDecision::Reject(RejectReason::Draining) - } else { - self.publications = self.publications.saturating_add(1); - self.follower_proof = false; - CoordinationDecision::Started - } - } - CoordinationInput::CallerCancel => CoordinationDecision::Ignored, - CoordinationInput::FollowerProof { accepted } => { - if accepted && self.publications != 0 { - self.follower_proof = true; - } - CoordinationDecision::Ignored - } - CoordinationInput::FinishPublication { fenced, succeeded } => { - self.publications = self.publications.saturating_sub(1); - let succeeded = succeeded || self.follower_proof; - self.follower_proof = false; - if fenced || !succeeded { - self.lifecycle = Lifecycle::Fenced; - return CoordinationDecision::Fence; - } - if self.can_deactivate() && self.is_draining() { - CoordinationDecision::ReadyToDeactivate - } else { - CoordinationDecision::Ignored - } - } - CoordinationInput::BeginHydration { - queue_empty, - publication_idle, - lease_live, - } => { - if !lease_live { - self.lifecycle = Lifecycle::Fenced; - self.busy = false; - CoordinationDecision::Fence - } else if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_draining() || self.transfer_preparing { - CoordinationDecision::Reject(RejectReason::Draining) - } else if self.busy { - CoordinationDecision::Reject(RejectReason::Busy) - } else if !queue_empty || !publication_idle || self.residency != Residency::Sparse { - CoordinationDecision::Ignored - } else { - // Fetch owns an effect, not the foreground slot. The worker - // serializes installation with SQL; the effect keeps drain - // and transfer from releasing the activation during fetch. - self.residency = Residency::Hydrating; - CoordinationDecision::Started - } - } - CoordinationInput::BeginInventory { - queue_empty, - publication_idle, - inventory_unknown, - refreshing, - lease_live, - } => { - if !lease_live { - self.lifecycle = Lifecycle::Fenced; - self.busy = false; - self.renewing = false; - CoordinationDecision::Fence - } else if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_draining() || self.transfer_preparing { - CoordinationDecision::Reject(RejectReason::Draining) - } else if self.busy { - CoordinationDecision::Reject(RejectReason::Busy) - } else if !queue_empty || !publication_idle || !inventory_unknown || refreshing { - CoordinationDecision::Ignored - } else { - CoordinationDecision::Started - } - } - CoordinationInput::BeginTransferPreflight { - queue_empty: _, - publication_idle: _, - lease_live, - } => { - if !lease_live { - self.lifecycle = Lifecycle::Fenced; - self.busy = false; - self.renewing = false; - CoordinationDecision::Fence - } else if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_draining() || self.transfer_preparing { - CoordinationDecision::Reject(RejectReason::Draining) - } else if self.renewing - || self.pending_effects.values().any(|effect| { - matches!( - effect, - CoordinationEffect::Hydration - | CoordinationEffect::Inventory - | CoordinationEffect::Compaction - | CoordinationEffect::Renewal - ) - }) - { - CoordinationDecision::Ignored - } else { - // Admission closes before these observations become quiescent so - // work already accepted by the actor can finish without loss. - self.transfer_preparing = true; - CoordinationDecision::Started - } - } - CoordinationInput::ConfirmTransfer => { - if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_shutdown() || self.is_draining() { - CoordinationDecision::Reject(RejectReason::Draining) - } else if !self.transfer_preparing || !self.can_deactivate() { - CoordinationDecision::Ignored - } else { - self.transfer_preparing = false; - self.lifecycle = Lifecycle::Draining; - if self.can_deactivate() { - CoordinationDecision::ReadyToDeactivate - } else { - CoordinationDecision::Started - } - } - } - CoordinationInput::AbortTransfer => { - if self.is_fenced() { - CoordinationDecision::Reject(RejectReason::Fenced) - } else if self.is_shutdown() || self.is_draining() { - CoordinationDecision::Reject(RejectReason::Draining) - } else if self.transfer_preparing { - self.transfer_preparing = false; - CoordinationDecision::Ignored - } else { - CoordinationDecision::Ignored - } - } - CoordinationInput::BeginCompaction { - queue_empty, - publication_idle, - publisher_ready, - due, - lease_live, - } => { - if !lease_live { - self.lifecycle = Lifecycle::Fenced; - CoordinationDecision::Fence - } else if self.is_fenced() - || self.is_draining() - || self.transfer_preparing - || self.busy - || self.renewing - || !queue_empty - || !publication_idle - || !publisher_ready - || !due - || !self.pending_effects.is_empty() - { - CoordinationDecision::Ignored - } else { - // The publisher token is exclusive, and busy keeps new SQL - // from building unpublished cuts behind a long promotion. - self.busy = true; - CoordinationDecision::Started - } - } - CoordinationInput::FinishCompaction { fenced } => { - self.busy = false; - if fenced || self.is_fenced() { - self.lifecycle = Lifecycle::Fenced; - CoordinationDecision::Fence - } else if self.can_deactivate() && self.is_draining() { - CoordinationDecision::ReadyToDeactivate - } else { - CoordinationDecision::Ignored - } - } - CoordinationInput::FinishHydration { complete, stale } => { - // A command may have started while this fetch was in flight. - // Hydration completion must not release its exclusive slot. - if !stale { - self.residency = if complete { - Residency::Resident - } else { - Residency::Sparse - }; - } else { - self.residency = Residency::Sparse; - return CoordinationDecision::Fence; - } - CoordinationDecision::Ignored - } - CoordinationInput::Fence => { - self.lifecycle = Lifecycle::Fenced; - self.busy = false; - self.renewing = false; - self.transfer_preparing = false; - CoordinationDecision::Ignored - } - } - } - - fn admit(&self, kind: AdmissionKind, admission_matches: bool) -> CoordinationDecision { - if !admission_matches { - return CoordinationDecision::Reject(RejectReason::NotActive); - } - if self.is_fenced() { - return if matches!(kind, AdmissionKind::Resolve) { - CoordinationDecision::ResolveUnknown - } else { - CoordinationDecision::Reject(RejectReason::Fenced) - }; - } - if self.transfer_preparing { - return CoordinationDecision::Reject(RejectReason::Draining); - } - // Migration swaps the admission capability before its durable cut is - // published. The successor capability may queue work, but `busy` keeps - // it from executing until `FinishMigration` returns to Serving. - if matches!(self.lifecycle, Lifecycle::Migrating) && !self.shutdown_requested { - return CoordinationDecision::Admit; - } - if self.is_draining() { - return CoordinationDecision::Reject(RejectReason::Draining); - } - CoordinationDecision::Admit - } -} - -impl fmt::Display for RejectReason { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - let text = match self { - Self::NotActive => "not active", - Self::Fenced => "fenced", - Self::Draining => "draining", - Self::Busy => "busy", - Self::PublicationPending => "publication pending", - }; - formatter.write_str(text) - } -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/coordination/sim.rs b/crates/crab-cell-runtime/src/coordination/sim.rs deleted file mode 100644 index e150695ad..000000000 --- a/crates/crab-cell-runtime/src/coordination/sim.rs +++ /dev/null @@ -1,965 +0,0 @@ -use crate::coordination::{ - AdmissionKind, CoordinationDecision, CoordinationEffect, CoordinationInput, CoordinationState, - RejectReason, -}; -use std::env; - -const COMMANDS: usize = 2; -const EVENT_COUNT: usize = 42; - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum Event { - Admit, - BeginWork, - BeginPublication, - FinishPublication { succeeded: bool }, - FinishWork, - BeginRenewal, - FinishRenewal { fenced: bool }, - BeginDrain, - BeginShutdown, - BeginMigration, - FinishMigration, - BeginHydration, - FinishHydration { complete: bool, stale: bool }, - BeginCompaction, - FinishCompaction { fenced: bool }, - Fence, - AdvanceClock { milliseconds: u64 }, - CallerCancel, - LostResponse, - FollowerProof { accepted: bool }, - ExactCasAfterLostResponse, - RootAcceptedPruneFailed, - DuplicatePublicationCompletion, - DelayedEffect, - OwnerCrash, - OwnerRestart, - LeaseExpire, - CasConflict { exact: bool }, - Release, - PrimitiveObligation, - PrimitiveCompletion, - MovementQuiesce, - MovementDurability, - MovementRelease, - LostReleaseResponse, - MovementAcquire, - ReceiverCrash, - MembershipLoss, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum MovementPhase { - Idle, - Quiescing, - Durability, - Released, - Acquiring, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -struct SplitMix64(u64); - -impl SplitMix64 { - fn next(&mut self) -> u64 { - self.0 = self.0.wrapping_add(0x9e37_79b9_7f4a_7c15); - let mut value = self.0; - value = (value ^ (value >> 30)).wrapping_mul(0xbf58_476d_1ce4_e5b9); - value = (value ^ (value >> 27)).wrapping_mul(0x94d0_49bb_1331_11eb); - value ^ (value >> 31) - } -} - -#[derive(Clone, Debug, PartialEq, Eq)] -struct Simulation { - state: CoordinationState, - clock_ms: u64, - owner: Option, - epoch: u64, - published_sequence: u64, - retained: usize, - durable_publications: u64, - accepted: [bool; COMMANDS], - acknowledged: [bool; COMMANDS], - unknown: [bool; COMMANDS], - terminal_outcomes: [u8; COMMANDS], - cancelled: [bool; COMMANDS], - owner_live: bool, - released: bool, - last_epoch: u64, - last_published_sequence: u64, - work_effect: Option, - renewal_effect: Option, - hydration_effect: Option, - compaction_effect: Option, - publication_effects: Vec, - delayed_effects: usize, - primitive_obligations: usize, - movement: MovementPhase, - receiver_live: bool, - membership_live: bool, - trace: Vec, -} - -impl Simulation { - fn new() -> Self { - Self { - state: CoordinationState::serving(true), - clock_ms: 0, - owner: Some(0), - epoch: 1, - published_sequence: 0, - retained: 0, - durable_publications: 0, - accepted: [false; COMMANDS], - acknowledged: [false; COMMANDS], - unknown: [false; COMMANDS], - terminal_outcomes: [0; COMMANDS], - cancelled: [false; COMMANDS], - owner_live: true, - released: false, - last_epoch: 1, - last_published_sequence: 0, - work_effect: None, - renewal_effect: None, - hydration_effect: None, - compaction_effect: None, - publication_effects: Vec::new(), - delayed_effects: 0, - primitive_obligations: 0, - movement: MovementPhase::Idle, - receiver_live: true, - membership_live: true, - trace: Vec::new(), - } - } - - fn trace_bytes(&self) -> Vec { - self.trace.join("\n").into_bytes() - } - - fn restore_after_takeover(&mut self) { - let retained = self.retained; - self.state = CoordinationState::serving(true); - self.compaction_effect = None; - self.publication_effects.clear(); - for _ in 0..retained { - if matches!( - self.state.step(CoordinationInput::BeginPublication), - CoordinationDecision::Started - ) { - self.publication_effects - .push(self.state.begin_effect(CoordinationEffect::Publication)); - } - } - } - - fn complete_effect(&mut self, effect_id: Option, effect: CoordinationEffect) { - if let Some(effect_id) = effect_id { - let decision = self - .state - .step(CoordinationInput::CompleteEffect { effect_id, effect }); - assert!( - matches!( - decision, - CoordinationDecision::EffectCompleted | CoordinationDecision::StaleEffect - ), - "unexpected effect completion decision: {decision:?}" - ); - } - } - - fn apply(&mut self, event: Event) { - let before = self.semantic_digest(); - if self.released && !matches!(event, Event::OwnerRestart) { - self.trace.push(format!( - "event={event:?};before={before:016x};decision=Ignored;after={:016x};clock={};epoch={};published={};retained={};durable={};owner={:?}", - self.semantic_digest(), - self.clock_ms, - self.epoch, - self.published_sequence, - self.retained, - self.durable_publications, - self.owner, - )); - return; - } - let decision = match event { - Event::Admit => { - let command = usize::from(self.accepted[0]); - let decision = self.state.step(CoordinationInput::Admit { - kind: AdmissionKind::Command, - admission_matches: true, - }); - if matches!(decision, CoordinationDecision::Admit) && command < COMMANDS { - self.accepted[command] = true; - } - decision - } - Event::BeginWork => { - let decision = if self.owner_live { - self.state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Command, - publisher_ready: true, - }) - } else { - CoordinationDecision::Ignored - }; - if matches!(decision, CoordinationDecision::Started) { - self.work_effect = Some( - self.state - .begin_effect(CoordinationEffect::Work(AdmissionKind::Command)), - ); - } - decision - } - Event::BeginPublication => { - let decision = if self.owner_live { - self.state.step(CoordinationInput::BeginPublication) - } else { - CoordinationDecision::Ignored - }; - if matches!(decision, CoordinationDecision::Started) { - self.retained = self.retained.saturating_add(1); - self.publication_effects - .push(self.state.begin_effect(CoordinationEffect::Publication)); - } - decision - } - Event::FinishPublication { succeeded } => { - let before_publications = self.state.publication_count(); - let effect_id = self.publication_effects.pop(); - self.complete_effect(effect_id, CoordinationEffect::Publication); - let decision = self.state.step(CoordinationInput::FinishPublication { - fenced: false, - succeeded, - }); - let completed = before_publications.saturating_sub(self.state.publication_count()); - self.retained = self.retained.saturating_sub(completed); - if succeeded && completed != 0 { - self.published_sequence = - self.published_sequence.saturating_add(completed as u64); - self.durable_publications = - self.durable_publications.saturating_add(completed as u64); - } else if !succeeded && completed != 0 { - self.owner = None; - } - decision - } - Event::FinishWork => { - let effect_id = self.work_effect.take(); - self.complete_effect(effect_id, CoordinationEffect::Work(AdmissionKind::Command)); - self.state - .step(CoordinationInput::FinishWork { fenced: false }) - } - Event::BeginRenewal => { - let decision = self.state.step(CoordinationInput::BeginRenewal { - queue_empty: true, - publication_idle: true, - lease_live: true, - }); - if matches!(decision, CoordinationDecision::Started) { - self.renewal_effect = - Some(self.state.begin_effect(CoordinationEffect::Renewal)); - } - decision - } - Event::FinishRenewal { fenced } => { - let effect_id = self.renewal_effect.take(); - self.complete_effect(effect_id, CoordinationEffect::Renewal); - let decision = self.state.step(CoordinationInput::FinishRenewal { fenced }); - if fenced { - self.owner = None; - } - decision - } - Event::BeginDrain => self.state.step(CoordinationInput::BeginDrain), - Event::BeginShutdown => self.state.step(CoordinationInput::BeginShutdown), - Event::BeginMigration => self.state.step(CoordinationInput::BeginMigration), - Event::FinishMigration => self - .state - .step(CoordinationInput::FinishMigration { fenced: false }), - Event::BeginHydration => { - let decision = self.state.step(CoordinationInput::BeginHydration { - queue_empty: true, - publication_idle: true, - lease_live: true, - }); - if matches!(decision, CoordinationDecision::Started) { - self.hydration_effect = - Some(self.state.begin_effect(CoordinationEffect::Hydration)); - } - decision - } - Event::FinishHydration { complete, stale } => { - let effect_id = self.hydration_effect.take(); - self.complete_effect(effect_id, CoordinationEffect::Hydration); - self.state - .step(CoordinationInput::FinishHydration { complete, stale }) - } - Event::BeginCompaction => { - let decision = self.state.step(CoordinationInput::BeginCompaction { - queue_empty: true, - publication_idle: self.state.publication_count() == 0, - publisher_ready: self.compaction_effect.is_none(), - due: true, - lease_live: self.owner_live, - }); - if matches!(decision, CoordinationDecision::Started) { - self.compaction_effect = - Some(self.state.begin_effect(CoordinationEffect::Compaction)); - } - decision - } - Event::FinishCompaction { fenced } => { - if let Some(effect_id) = self.compaction_effect.take() { - self.complete_effect(Some(effect_id), CoordinationEffect::Compaction); - if fenced { - self.owner = None; - } - self.state - .step(CoordinationInput::FinishCompaction { fenced }) - } else { - CoordinationDecision::Ignored - } - } - Event::Fence => { - self.owner = None; - self.owner_live = false; - self.state.step(CoordinationInput::Fence) - } - Event::AdvanceClock { milliseconds } => { - self.clock_ms = self.clock_ms.saturating_add(milliseconds); - CoordinationDecision::Ignored - } - Event::CallerCancel => { - self.cancelled[0] = true; - CoordinationDecision::Ignored - } - Event::LostResponse => { - self.delayed_effects = self.delayed_effects.saturating_add(1); - CoordinationDecision::Ignored - } - Event::FollowerProof { accepted } => { - if accepted && self.retained != 0 { - self.durable_publications = self.durable_publications.saturating_add(1); - } - CoordinationDecision::Ignored - } - Event::ExactCasAfterLostResponse => { - if self.retained != 0 { - let effect_id = self.publication_effects.pop(); - self.complete_effect(effect_id, CoordinationEffect::Publication); - let decision = self.state.step(CoordinationInput::FinishPublication { - fenced: false, - succeeded: true, - }); - self.retained = self.retained.saturating_sub(1); - self.published_sequence = self.published_sequence.saturating_add(1); - self.durable_publications = self.durable_publications.saturating_add(1); - decision - } else { - CoordinationDecision::Ignored - } - } - Event::RootAcceptedPruneFailed => { - if self.retained == 0 { - CoordinationDecision::Ignored - } else { - let before_publications = self.state.publication_count(); - let effect_id = self.publication_effects.pop(); - self.complete_effect(effect_id, CoordinationEffect::Publication); - let decision = self.state.step(CoordinationInput::FinishPublication { - fenced: true, - succeeded: false, - }); - let completed = - before_publications.saturating_sub(self.state.publication_count()); - self.retained = self.retained.saturating_sub(completed); - if completed != 0 { - // The accepted root survives cleanup failure, but the actor - // fences before it can return a committed result. - self.published_sequence = self.published_sequence.saturating_add(1); - self.durable_publications = self.durable_publications.saturating_add(1); - if self.accepted[0] && !self.acknowledged[0] { - self.unknown[0] = true; - self.terminal_outcomes[0] = self.terminal_outcomes[0].saturating_add(1); - } - } - decision - } - } - Event::DuplicatePublicationCompletion => { - let before_publications = self.state.publication_count(); - self.complete_effect(None, CoordinationEffect::Publication); - let decision = self.state.step(CoordinationInput::FinishPublication { - fenced: false, - succeeded: true, - }); - let completed = before_publications.saturating_sub(self.state.publication_count()); - self.retained = self.retained.saturating_sub(completed); - if completed != 0 { - self.published_sequence = - self.published_sequence.saturating_add(completed as u64); - self.durable_publications = - self.durable_publications.saturating_add(completed as u64); - } - decision - } - Event::DelayedEffect => { - self.delayed_effects = self.delayed_effects.saturating_sub(1); - CoordinationDecision::Ignored - } - Event::OwnerCrash => { - self.owner = None; - self.owner_live = false; - self.state.step(CoordinationInput::Fence) - } - Event::OwnerRestart => { - if self.owner.is_none() { - self.epoch = self.epoch.saturating_add(1); - self.owner = Some(1); - self.owner_live = true; - self.released = false; - self.movement = MovementPhase::Idle; - self.restore_after_takeover(); - } - CoordinationDecision::Ignored - } - Event::LeaseExpire => { - self.owner = None; - self.owner_live = false; - self.state.step(CoordinationInput::Fence) - } - Event::CasConflict { exact } => { - if exact { - let effect_id = self.publication_effects.pop(); - self.complete_effect(effect_id, CoordinationEffect::Publication); - let before_publications = self.state.publication_count(); - let decision = self.state.step(CoordinationInput::FinishPublication { - fenced: false, - succeeded: true, - }); - let completed = - before_publications.saturating_sub(self.state.publication_count()); - self.retained = self.retained.saturating_sub(completed); - self.published_sequence = - self.published_sequence.saturating_add(completed as u64); - self.durable_publications = - self.durable_publications.saturating_add(completed as u64); - decision - } else { - self.owner = None; - self.owner_live = false; - self.state.step(CoordinationInput::Fence) - } - } - Event::Release => { - if self.state.ready_to_deactivate(true, true) - && self.retained == 0 - && self.primitive_obligations == 0 - && self - .accepted - .iter() - .enumerate() - .all(|(index, accepted)| !*accepted || self.terminal_outcomes[index] == 1) - { - self.owner = None; - self.owner_live = false; - self.released = true; - } - CoordinationDecision::Ignored - } - Event::PrimitiveObligation => { - if !self.released { - self.primitive_obligations = self.primitive_obligations.saturating_add(1); - } - CoordinationDecision::Ignored - } - Event::PrimitiveCompletion => { - self.primitive_obligations = self.primitive_obligations.saturating_sub(1); - CoordinationDecision::Ignored - } - Event::MovementQuiesce => { - let decision = self.state.step(CoordinationInput::BeginDrain); - if matches!( - decision, - CoordinationDecision::Started - | CoordinationDecision::ReadyToDeactivate - | CoordinationDecision::ReadyToDeactivateFenced - ) { - self.movement = MovementPhase::Quiescing; - } - decision - } - Event::MovementDurability => { - if self.movement == MovementPhase::Quiescing - && self.retained == 0 - && self.primitive_obligations == 0 - && self.state.ready_to_deactivate(true, true) - { - self.movement = MovementPhase::Durability; - } - CoordinationDecision::Ignored - } - Event::MovementRelease => { - if self.movement == MovementPhase::Durability - && self.retained == 0 - && self.primitive_obligations == 0 - && self.state.ready_to_deactivate(true, true) - { - self.owner = None; - self.owner_live = false; - self.movement = MovementPhase::Released; - } - CoordinationDecision::Ignored - } - // The authority transition already committed; only the caller's - // reply is lost. The receiver must still observe the released - // owner through the normal authority path. - Event::LostReleaseResponse => CoordinationDecision::Ignored, - Event::MovementAcquire => { - if self.movement == MovementPhase::Released - && self.receiver_live - && self.membership_live - { - self.movement = MovementPhase::Acquiring; - self.epoch = self.epoch.saturating_add(1); - self.owner = Some(1); - self.owner_live = true; - self.restore_after_takeover(); - self.movement = MovementPhase::Idle; - } - CoordinationDecision::Ignored - } - Event::ReceiverCrash => { - self.receiver_live = false; - if self.movement == MovementPhase::Acquiring { - self.movement = MovementPhase::Released; - } - CoordinationDecision::Ignored - } - Event::MembershipLoss => { - self.membership_live = false; - if self.movement == MovementPhase::Acquiring { - self.movement = MovementPhase::Released; - self.owner = None; - self.owner_live = false; - } - CoordinationDecision::Ignored - } - }; - if matches!(decision, CoordinationDecision::Admit) { - self.accepted[0] = true; - } - if self.state.is_fenced() { - self.owner = None; - self.owner_live = false; - } - if self.durable_publications != 0 - && self.accepted[0] - && !self.acknowledged[0] - && !self.unknown[0] - { - self.acknowledged[0] = true; - self.terminal_outcomes[0] = self.terminal_outcomes[0].saturating_add(1); - } - self.trace.push(format!( - "event={event:?};before={before:016x};decision={decision:?};after={:016x};clock={};epoch={};published={};retained={};durable={};owner={:?}", - self.semantic_digest(), - self.clock_ms, - self.epoch, - self.published_sequence, - self.retained, - self.durable_publications, - self.owner, - )); - } - - fn semantic_digest(&self) -> u64 { - let mut value = self.clock_ms ^ self.epoch.rotate_left(11); - value ^= self.published_sequence.rotate_left(23); - value ^= self.durable_publications.rotate_left(37); - value ^= (self.retained as u64).rotate_left(47); - value ^= u64::from(self.state.is_fenced()).rotate_left(53); - value ^= u64::from(self.state.is_draining()).rotate_left(59); - value ^= u64::from(self.owner.unwrap_or(u8::MAX)).rotate_left(7); - value ^= u64::from(self.owner_live).rotate_left(13); - value ^= u64::from(self.released).rotate_left(19); - value ^= (self.delayed_effects as u64).rotate_left(29); - value ^= (self.primitive_obligations as u64).rotate_left(41); - value ^= (self.movement as u64).rotate_left(31); - value ^= u64::from(self.receiver_live).rotate_left(43); - value ^= u64::from(self.membership_live).rotate_left(51); - for (index, accepted) in self.accepted.iter().enumerate() { - value ^= u64::from(*accepted).rotate_left(index as u32 + 1); - value ^= u64::from(self.acknowledged[index]).rotate_left(index as u32 + 9); - value ^= u64::from(self.unknown[index]).rotate_left(index as u32 + 25); - value ^= u64::from(self.terminal_outcomes[index]).rotate_left(index as u32 + 17); - } - value - } - - fn assert_invariants(&self) { - assert!(self.owner.is_none_or(|owner| owner < 2)); - assert!(self.epoch >= self.last_epoch); - assert!(self.published_sequence >= self.last_published_sequence); - assert!(self.published_sequence <= self.durable_publications); - assert!(self.terminal_outcomes.iter().all(|outcomes| *outcomes <= 1)); - assert!(self.unknown.iter().enumerate().all(|(index, unknown)| { - !*unknown - || (self.accepted[index] - && !self.acknowledged[index] - && self.durable_publications != 0 - && self.terminal_outcomes[index] == 1) - })); - assert!( - self.acknowledged - .iter() - .enumerate() - .all(|(index, acknowledged)| !*acknowledged || self.accepted[index]) - ); - assert!( - self.acknowledged - .iter() - .all(|acknowledged| !*acknowledged || self.durable_publications != 0) - ); - if self.state.is_fenced() { - assert!(matches!( - self.state.lookup(), - CoordinationDecision::Reject(RejectReason::Fenced) - )); - assert!(self.owner.is_none()); - assert!(!self.owner_live); - } - if self.state.is_draining() { - assert!(!matches!( - self.state.lookup(), - CoordinationDecision::LocalHandle - )); - } - assert!(!self.owner_live || self.owner.is_some()); - assert!(!self.released || self.owner.is_none()); - assert!(!self.released || self.primitive_obligations == 0); - assert!(!self.released || self.state.ready_to_deactivate(true, true)); - assert!( - !self.released - || self - .accepted - .iter() - .enumerate() - .all(|(index, accepted)| { !*accepted || self.terminal_outcomes[index] == 1 }) - ); - assert!( - self.retained == self.state.publication_count(), - "publication obligation was not retained: {:?}", - self.trace - ); - assert!( - self.owner.is_some() || self.retained == 0 || self.state.is_fenced(), - "owner lost with retained publication: {:?}", - self.trace - ); - if matches!( - self.movement, - MovementPhase::Quiescing - | MovementPhase::Durability - | MovementPhase::Released - | MovementPhase::Acquiring - ) { - assert!(!matches!( - self.state.lookup(), - CoordinationDecision::LocalHandle - )); - } - if matches!( - self.movement, - MovementPhase::Released | MovementPhase::Acquiring - ) { - assert!(self.owner.is_none() || self.movement == MovementPhase::Acquiring); - } - } - - fn record_progress(&mut self) { - self.last_epoch = self.epoch; - self.last_published_sequence = self.published_sequence; - } -} - -fn replay(seed: u64, steps: usize) -> Simulation { - let mut simulation = Simulation::new(); - let mut random = SplitMix64(seed); - for _ in 0..steps { - simulation.apply(event(random.next())); - simulation.assert_invariants(); - simulation.record_progress(); - } - simulation -} - -fn event(seed: u64) -> Event { - match seed as usize % EVENT_COUNT { - 0 => Event::Admit, - 1 => Event::BeginWork, - 2 => Event::BeginPublication, - 3 => Event::FinishPublication { succeeded: true }, - 4 => Event::FinishPublication { succeeded: false }, - 5 => Event::FinishWork, - 6 => Event::BeginRenewal, - 7 => Event::FinishRenewal { - fenced: seed & 1 == 0, - }, - 8 => Event::BeginDrain, - 9 => Event::BeginShutdown, - 10 => Event::BeginMigration, - 11 => Event::FinishMigration, - 12 => Event::BeginHydration, - 13 => Event::FinishHydration { - complete: seed & 1 == 0, - stale: seed & 2 == 0, - }, - 14 => Event::Fence, - 15 => Event::AdvanceClock { milliseconds: 17 }, - 16 => Event::CallerCancel, - 17 => Event::LostResponse, - 18 => Event::FollowerProof { - accepted: seed & 1 == 0, - }, - 19 => Event::ExactCasAfterLostResponse, - 20 => Event::DuplicatePublicationCompletion, - 21 => Event::DelayedEffect, - 22 => Event::OwnerCrash, - 23 => Event::OwnerRestart, - 24 => Event::LeaseExpire, - 25 => Event::CasConflict { - exact: seed & 1 == 0, - }, - 26 => Event::Release, - 27 => Event::PrimitiveObligation, - 28 => Event::PrimitiveCompletion, - 29 => Event::AdvanceClock { milliseconds: 1 }, - 30 => Event::CallerCancel, - 31 => Event::LostResponse, - 32 => Event::MovementQuiesce, - 33 => Event::MovementDurability, - 34 => Event::MovementRelease, - 35 => Event::LostReleaseResponse, - 36 => Event::MovementAcquire, - 37 => Event::ReceiverCrash, - 38 => Event::MembershipLoss, - 39 => Event::BeginCompaction, - 40 => Event::FinishCompaction { - fenced: seed & 1 == 0, - }, - _ => Event::RootAcceptedPruneFailed, - } -} - -fn explore(simulation: &Simulation, depth: usize) { - simulation.assert_invariants(); - if depth == 0 { - return; - } - for choice in 0..EVENT_COUNT as u64 { - let mut next = simulation.clone(); - next.apply(event(choice + depth as u64)); - next.assert_invariants(); - next.record_progress(); - explore(&next, depth - 1); - } -} - -#[test] -fn seeded_schedules_replay_byte_identically() { - for seed in [0, 1, 7, 41, 99, 0xfeed_1234] { - let first = replay(seed, 128); - let second = replay(seed, 128); - assert_eq!(first.trace_bytes(), second.trace_bytes()); - assert_eq!( - first, second, - "seed {seed} did not replay deterministically" - ); - } -} - -#[test] -fn broad_seed_corpus_is_replayable() { - for seed in 0..512 { - println!("coordination_seed={seed};steps=256"); - let first = replay(seed, 256); - let second = replay(seed, 256); - assert_eq!(first.trace_bytes(), second.trace_bytes(), "seed {seed}"); - } -} - -#[test] -fn replay_requested_seed_from_environment() { - let Ok(seed) = env::var("CRAB_COORDINATION_SEED") else { - return; - }; - let seed = seed - .parse::() - .expect("CRAB_COORDINATION_SEED must be an unsigned integer"); - let steps = env::var("CRAB_COORDINATION_STEPS") - .map(|value| { - value - .parse::() - .expect("CRAB_COORDINATION_STEPS must be an unsigned integer") - }) - .unwrap_or(128); - assert!( - steps <= 4_096, - "CRAB_COORDINATION_STEPS exceeds the replay bound" - ); - let simulation = replay(seed, steps); - println!( - "coordination_seed={seed};steps={steps};trace_bytes={}", - simulation.trace_bytes().len() - ); -} - -#[test] -fn bounded_exhaustive_schedules_preserve_protocol_invariants() { - explore(&Simulation::new(), 4); -} - -#[test] -fn broken_early_acknowledgement_is_detected() { - let mut simulation = Simulation::new(); - simulation.accepted[0] = true; - simulation.acknowledged[0] = true; - assert!(std::panic::catch_unwind(|| simulation.assert_invariants()).is_err()); -} - -#[test] -fn broken_accept_after_fence_is_detected() { - let mut simulation = Simulation::new(); - simulation.apply(Event::Fence); - simulation.accepted[0] = true; - assert!(simulation.state.is_fenced()); - assert!(matches!( - simulation.state.lookup(), - CoordinationDecision::Reject(RejectReason::Fenced) - )); - assert!(simulation.owner.is_none()); -} - -#[test] -fn broken_different_winner_is_detected() { - let mut simulation = Simulation::new(); - simulation.owner = Some(1); - assert!( - std::panic::catch_unwind(|| { - assert_eq!( - simulation.owner, - Some(0), - "different CAS winner was adopted" - ) - }) - .is_err() - ); -} - -#[test] -fn broken_early_release_is_detected() { - let mut simulation = Simulation::new(); - simulation.retained = 1; - simulation.owner = None; - assert!(std::panic::catch_unwind(|| simulation.assert_invariants()).is_err()); -} - -#[test] -fn duplicate_publication_completion_cannot_underflow_publication_state() { - let mut simulation = Simulation::new(); - simulation.apply(Event::BeginPublication); - simulation.apply(Event::FinishPublication { succeeded: true }); - simulation.apply(Event::DuplicatePublicationCompletion); - simulation.assert_invariants(); - assert_eq!(simulation.state.publication_count(), 0); -} - -#[test] -fn accepted_root_with_failed_prune_fences_before_success_reply() { - let mut simulation = Simulation::new(); - simulation.apply(Event::Admit); - simulation.apply(Event::BeginWork); - simulation.apply(Event::BeginPublication); - simulation.apply(Event::RootAcceptedPruneFailed); - - assert_eq!(simulation.published_sequence, 1); - assert_eq!(simulation.durable_publications, 1); - assert_eq!(simulation.retained, 0); - assert!(simulation.state.is_fenced()); - assert!(simulation.unknown[0]); - assert!(!simulation.acknowledged[0]); - assert_eq!(simulation.terminal_outcomes[0], 1); - simulation.apply(Event::OwnerRestart); - simulation.assert_invariants(); - assert_eq!(simulation.published_sequence, 1); - assert_eq!(simulation.terminal_outcomes[0], 1); -} - -#[test] -fn prune_failure_preserves_a_prior_follower_acknowledgement() { - let mut simulation = Simulation::new(); - simulation.apply(Event::Admit); - simulation.apply(Event::BeginWork); - simulation.apply(Event::BeginPublication); - simulation.apply(Event::FollowerProof { accepted: true }); - assert!(simulation.acknowledged[0]); - simulation.apply(Event::RootAcceptedPruneFailed); - - simulation.assert_invariants(); - assert!(simulation.state.is_fenced()); - assert_eq!(simulation.published_sequence, 1); - assert!(simulation.acknowledged[0]); - assert!(!simulation.unknown[0]); - assert_eq!(simulation.terminal_outcomes[0], 1); -} - -#[test] -fn movement_requires_quiesce_durability_release_and_live_acquire() { - let mut simulation = Simulation::new(); - simulation.apply(Event::MovementQuiesce); - assert_eq!(simulation.movement, MovementPhase::Quiescing); - simulation.apply(Event::MovementRelease); - assert_eq!(simulation.movement, MovementPhase::Quiescing); - simulation.apply(Event::MovementDurability); - assert_eq!(simulation.movement, MovementPhase::Durability); - simulation.apply(Event::MovementRelease); - assert_eq!(simulation.movement, MovementPhase::Released); - simulation.apply(Event::LostReleaseResponse); - assert_eq!(simulation.owner, None); - simulation.apply(Event::ReceiverCrash); - simulation.apply(Event::MovementAcquire); - assert_eq!(simulation.movement, MovementPhase::Released); - simulation.receiver_live = true; - simulation.membership_live = true; - simulation.apply(Event::MovementAcquire); - assert_eq!(simulation.movement, MovementPhase::Idle); - assert_eq!(simulation.owner, Some(1)); -} - -#[test] -fn membership_loss_during_movement_preserves_released_root() { - let mut simulation = Simulation::new(); - simulation.apply(Event::BeginPublication); - simulation.apply(Event::FinishPublication { succeeded: true }); - simulation.apply(Event::MovementQuiesce); - simulation.apply(Event::MovementDurability); - simulation.apply(Event::MovementRelease); - - let epoch = simulation.epoch; - let published_sequence = simulation.published_sequence; - let durable_publications = simulation.durable_publications; - simulation.apply(Event::MembershipLoss); - simulation.apply(Event::MovementAcquire); - - assert_eq!(simulation.movement, MovementPhase::Released); - assert_eq!(simulation.owner, None); - assert_eq!(simulation.epoch, epoch); - assert_eq!(simulation.published_sequence, published_sequence); - assert_eq!(simulation.durable_publications, durable_publications); - simulation.assert_invariants(); -} diff --git a/crates/crab-cell-runtime/src/coordination/tests.rs b/crates/crab-cell-runtime/src/coordination/tests.rs deleted file mode 100644 index d61ce1b9d..000000000 --- a/crates/crab-cell-runtime/src/coordination/tests.rs +++ /dev/null @@ -1,7 +0,0 @@ -use super::*; - -mod effects; -mod lifecycle; -mod publication; -mod scheduler; -mod transfer; diff --git a/crates/crab-cell-runtime/src/coordination/tests/effects.rs b/crates/crab-cell-runtime/src/coordination/tests/effects.rs deleted file mode 100644 index ae11b926e..000000000 --- a/crates/crab-cell-runtime/src/coordination/tests/effects.rs +++ /dev/null @@ -1,73 +0,0 @@ -//! Effect completion, deactivation, and adapter observations. - -use super::*; - -#[test] -fn effect_completion_is_exact_and_duplicate_completions_are_stale() { - let mut state = CoordinationState::serving(true); - let first = state.begin_effect(CoordinationEffect::Publication); - let second = state.begin_effect(CoordinationEffect::Proof); - assert_ne!(first, second); - assert!(state.effect_matches(first, CoordinationEffect::Publication)); - assert!(state.effect_matches(second, CoordinationEffect::Proof)); - - assert_eq!( - state.step(CoordinationInput::CompleteEffect { - effect_id: first, - effect: CoordinationEffect::Publication, - }), - CoordinationDecision::EffectCompleted - ); - assert!(!state.effect_matches(first, CoordinationEffect::Publication)); - assert!(state.effect_matches(second, CoordinationEffect::Proof)); - assert_eq!( - state.step(CoordinationInput::CompleteEffect { - effect_id: first, - effect: CoordinationEffect::Publication, - }), - CoordinationDecision::StaleEffect - ); - assert!(state.effect_matches(second, CoordinationEffect::Proof)); -} -#[test] -fn mismatched_effect_family_cannot_release_a_pending_operation() { - let mut state = CoordinationState::serving(true); - let effect_id = state.begin_effect(CoordinationEffect::Publication); - assert_eq!( - state.step(CoordinationInput::CompleteEffect { - effect_id, - effect: CoordinationEffect::Renewal, - }), - CoordinationDecision::StaleEffect - ); - assert!(state.effect_matches(effect_id, CoordinationEffect::Publication)); - assert_eq!( - state.step(CoordinationInput::CompleteEffect { - effect_id, - effect: CoordinationEffect::Publication, - }), - CoordinationDecision::EffectCompleted - ); -} -#[test] -fn deactivation_waits_for_effect_completion_owned_by_the_kernel() { - let mut state = CoordinationState::serving(true); - let effect_id = state.begin_effect(CoordinationEffect::Renewal); - state.step(CoordinationInput::BeginDrain); - assert!(!state.can_deactivate()); - assert_eq!( - state.step(CoordinationInput::CompleteEffect { - effect_id, - effect: CoordinationEffect::Renewal, - }), - CoordinationDecision::EffectCompleted - ); - assert!(state.can_deactivate()); -} -#[test] -fn deactivation_requires_adapter_observations_without_repeating_policy() { - let state = CoordinationState::serving(true); - assert!(!state.ready_to_deactivate(false, true)); - assert!(!state.ready_to_deactivate(true, false)); - assert!(state.ready_to_deactivate(true, true)); -} diff --git a/crates/crab-cell-runtime/src/coordination/tests/lifecycle.rs b/crates/crab-cell-runtime/src/coordination/tests/lifecycle.rs deleted file mode 100644 index 65d3708b3..000000000 --- a/crates/crab-cell-runtime/src/coordination/tests/lifecycle.rs +++ /dev/null @@ -1,147 +0,0 @@ -//! Admission, fencing, drain, cancellation, and shutdown transitions. - -use super::*; - -#[test] -fn serving_cell_admits_matching_work_and_local_lookup() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Command, - admission_matches: true, - }), - CoordinationDecision::Admit - ); - assert_eq!( - state.step(CoordinationInput::Lookup), - CoordinationDecision::LocalHandle - ); -} -#[test] -fn fence_rejects_new_work_and_returns_unknown_resolution() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::Fence); - assert_eq!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Command, - admission_matches: true, - }), - CoordinationDecision::Reject(RejectReason::Fenced) - ); - assert_eq!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Resolve, - admission_matches: true, - }), - CoordinationDecision::ResolveUnknown - ); - assert_eq!( - state.step(CoordinationInput::Lookup), - CoordinationDecision::Reject(RejectReason::Fenced) - ); -} -#[test] -fn drain_closes_admission_but_finishes_existing_work() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Command, - publisher_ready: true, - }), - CoordinationDecision::Started - ); - assert_eq!( - state.step(CoordinationInput::BeginDrain), - CoordinationDecision::Started - ); - assert_eq!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Query, - admission_matches: true, - }), - CoordinationDecision::Reject(RejectReason::Draining) - ); - assert_eq!( - state.step(CoordinationInput::FinishWork { fenced: false }), - CoordinationDecision::ReadyToDeactivate - ); -} -#[test] -fn fenced_work_completion_returns_fence_without_actor_redeciding() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Query, - publisher_ready: true, - }), - CoordinationDecision::Started - ); - assert_eq!( - state.step(CoordinationInput::FinishWork { fenced: true }), - CoordinationDecision::Fence - ); - assert!(state.is_fenced()); - assert!(!state.is_busy()); -} -#[test] -fn caller_cancellation_does_not_cancel_accepted_work() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Command, - publisher_ready: true, - }), - CoordinationDecision::Started - ); - assert_eq!( - state.step(CoordinationInput::CallerCancel), - CoordinationDecision::Ignored - ); - assert!(state.is_busy()); -} -#[test] -fn admitted_work_may_publish_after_drain_begins() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Command, - publisher_ready: true, - }), - CoordinationDecision::Started - ); - state.step(CoordinationInput::BeginDrain); - assert_eq!( - state.step(CoordinationInput::BeginPublication), - CoordinationDecision::Started - ); - state.step(CoordinationInput::FinishWork { fenced: false }); - assert_eq!(state.publication_count(), 1); - state.step(CoordinationInput::FinishPublication { - fenced: false, - succeeded: true, - }); - assert_eq!(state.publication_count(), 0); -} -#[test] -fn queued_work_can_finish_after_shutdown_closes_new_admission() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::BeginShutdown); - assert_eq!( - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Query, - publisher_ready: true, - }), - CoordinationDecision::Started - ); - assert_eq!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Query, - admission_matches: true, - }), - CoordinationDecision::Reject(RejectReason::Draining) - ); - assert_eq!( - state.step(CoordinationInput::FinishWork { fenced: false }), - CoordinationDecision::ReadyToDeactivate - ); -} diff --git a/crates/crab-cell-runtime/src/coordination/tests/publication.rs b/crates/crab-cell-runtime/src/coordination/tests/publication.rs deleted file mode 100644 index 79c1d7a52..000000000 --- a/crates/crab-cell-runtime/src/coordination/tests/publication.rs +++ /dev/null @@ -1,130 +0,0 @@ -//! Publication outcomes, CAS loss, durability proofs, and compaction. - -use super::*; - -#[test] -fn publication_failure_fences_and_cannot_be_released_as_success() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::BeginPublication); - assert_eq!(state.publication_count(), 1); - assert_eq!( - state.step(CoordinationInput::FinishPublication { - fenced: false, - succeeded: false, - }), - CoordinationDecision::Fence - ); - assert!(state.is_fenced()); - assert_eq!(state.publication_count(), 0); -} -#[test] -fn fenced_work_cannot_start_a_publication_after_completion() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Command, - publisher_ready: true, - }); - assert_eq!( - state.step(CoordinationInput::FinishWork { fenced: true }), - CoordinationDecision::Fence - ); - assert_eq!( - state.step(CoordinationInput::BeginPublication), - CoordinationDecision::Reject(RejectReason::Fenced) - ); -} -#[test] -fn lost_cas_fences_the_owner_and_releases_the_publication_obligation() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::BeginPublication); - assert_eq!( - state.step(CoordinationInput::FinishPublication { - fenced: true, - succeeded: false, - }), - CoordinationDecision::Fence - ); - assert!(state.is_fenced()); - assert_eq!(state.publication_count(), 0); - assert!(state.can_deactivate()); -} -#[test] -fn accepted_follower_proof_satisfies_publication_durability() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::BeginPublication); - state.step(CoordinationInput::FollowerProof { accepted: true }); - assert!(state.follower_proof()); - state.step(CoordinationInput::FinishPublication { - fenced: false, - succeeded: false, - }); - assert!(!state.is_fenced()); - assert_eq!(state.publication_count(), 0); -} -#[test] -fn compaction_waits_for_quiet_publication_and_blocks_drain_until_completion() { - let mut state = CoordinationState::serving(true); - let begin = |state: &mut CoordinationState, queue_empty, publication_idle| { - state.step(CoordinationInput::BeginCompaction { - queue_empty, - publication_idle, - publisher_ready: true, - due: true, - lease_live: true, - }) - }; - assert_eq!( - begin(&mut state, false, true), - CoordinationDecision::Ignored - ); - assert_eq!( - begin(&mut state, true, false), - CoordinationDecision::Ignored - ); - assert_eq!(begin(&mut state, true, true), CoordinationDecision::Started); - let effect = state.begin_effect(CoordinationEffect::Compaction); - assert_eq!( - state.step(CoordinationInput::BeginDrain), - CoordinationDecision::Started - ); - assert!(!state.ready_to_deactivate(true, false)); - assert_eq!( - state.step(CoordinationInput::CompleteEffect { - effect_id: effect, - effect: CoordinationEffect::Compaction, - }), - CoordinationDecision::EffectCompleted - ); - assert_eq!( - state.step(CoordinationInput::FinishCompaction { fenced: false }), - CoordinationDecision::ReadyToDeactivate - ); -} -#[test] -fn lease_loss_during_compaction_fences_without_releasing_the_effect() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginCompaction { - queue_empty: true, - publication_idle: true, - publisher_ready: true, - due: true, - lease_live: true, - }), - CoordinationDecision::Started - ); - let effect = state.begin_effect(CoordinationEffect::Compaction); - assert_eq!( - state.step(CoordinationInput::FinishCompaction { fenced: true }), - CoordinationDecision::Fence - ); - assert!(!state.ready_to_deactivate(true, true)); - assert_eq!( - state.step(CoordinationInput::CompleteEffect { - effect_id: effect, - effect: CoordinationEffect::Compaction, - }), - CoordinationDecision::EffectCompleted - ); - assert!(state.ready_to_deactivate(true, true)); -} diff --git a/crates/crab-cell-runtime/src/coordination/tests/scheduler.rs b/crates/crab-cell-runtime/src/coordination/tests/scheduler.rs deleted file mode 100644 index d72b9bdc7..000000000 --- a/crates/crab-cell-runtime/src/coordination/tests/scheduler.rs +++ /dev/null @@ -1,203 +0,0 @@ -//! Kernel scheduling decisions and observation requirements. - -use super::*; - -#[test] -fn scheduler_decision_stays_pure_and_respects_lifecycle_obligations() { - let mut serving = CoordinationState::serving(true); - assert_eq!( - serving.step(CoordinationInput::Schedule { - queue_empty: false, - publisher_ready: true, - publication_blocked: false, - lease_live: true, - }), - CoordinationDecision::StartQueuedWork - ); - - let mut busy = CoordinationState::serving(true); - busy.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Query, - publisher_ready: true, - }); - assert_eq!( - busy.step(CoordinationInput::Schedule { - queue_empty: false, - publisher_ready: true, - publication_blocked: false, - lease_live: true, - }), - CoordinationDecision::Ignored - ); - - let mut draining = CoordinationState::serving(true); - draining.step(CoordinationInput::BeginDrain); - assert_eq!( - draining.step(CoordinationInput::Schedule { - queue_empty: true, - publisher_ready: true, - publication_blocked: false, - lease_live: true, - }), - CoordinationDecision::ReadyToDeactivate - ); -} -#[test] -fn scheduler_blocks_publication_pressure_without_leaking_adapter_policy() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::Schedule { - queue_empty: false, - publisher_ready: true, - publication_blocked: true, - lease_live: true, - }), - CoordinationDecision::Ignored - ); - assert!(!state.is_busy()); -} -#[test] -fn scheduler_requires_publisher_observation_before_deactivation() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::BeginDrain); - assert_eq!( - state.step(CoordinationInput::Schedule { - queue_empty: true, - publisher_ready: false, - publication_blocked: false, - lease_live: true, - }), - CoordinationDecision::Ignored - ); - assert_eq!( - state.step(CoordinationInput::Schedule { - queue_empty: true, - publisher_ready: true, - publication_blocked: false, - lease_live: true, - }), - CoordinationDecision::ReadyToDeactivate - ); -} -#[test] -fn scheduler_fences_before_dispatch_when_the_node_lease_is_lost() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::Schedule { - queue_empty: false, - publisher_ready: true, - publication_blocked: false, - lease_live: false, - }), - CoordinationDecision::Fence - ); - assert!(state.is_fenced()); - assert!(!state.is_busy()); -} -#[test] -fn fenced_scheduler_can_release_after_all_local_obligations_drain() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::Schedule { - queue_empty: true, - publisher_ready: true, - publication_blocked: false, - lease_live: false, - }), - CoordinationDecision::ReadyToDeactivateFenced - ); - assert!(state.is_fenced()); -} -#[test] -fn background_effects_wait_for_foreground_and_publication_quiescence() { - let mut renewal = CoordinationState::serving(true); - renewal.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Query, - publisher_ready: true, - }); - assert_eq!( - renewal.step(CoordinationInput::BeginRenewal { - queue_empty: true, - publication_idle: true, - lease_live: true, - }), - CoordinationDecision::Ignored - ); - - let mut hydration = CoordinationState::serving_with_residency(true, Residency::Sparse); - assert_eq!( - hydration.step(CoordinationInput::BeginHydration { - queue_empty: false, - publication_idle: true, - lease_live: true, - }), - CoordinationDecision::Ignored - ); - assert_eq!(hydration.residency(), Residency::Sparse); -} -#[test] -fn inventory_refresh_requires_quiescence_and_live_lease() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginInventory { - queue_empty: false, - publication_idle: true, - inventory_unknown: true, - refreshing: false, - lease_live: true, - }), - CoordinationDecision::Ignored - ); - assert_eq!( - state.step(CoordinationInput::BeginInventory { - queue_empty: true, - publication_idle: true, - inventory_unknown: true, - refreshing: false, - lease_live: true, - }), - CoordinationDecision::Started - ); - state.begin_effect(CoordinationEffect::Inventory); - assert_eq!( - state.step(CoordinationInput::Schedule { - queue_empty: false, - publisher_ready: true, - publication_blocked: false, - lease_live: true, - }), - CoordinationDecision::Ignored - ); - - let mut fenced = CoordinationState::serving(true); - assert_eq!( - fenced.step(CoordinationInput::BeginInventory { - queue_empty: true, - publication_idle: true, - inventory_unknown: true, - refreshing: false, - lease_live: false, - }), - CoordinationDecision::Fence - ); - assert!(fenced.is_fenced()); -} -#[test] -fn stale_renewal_completion_is_not_a_new_admission() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginRenewal { - queue_empty: true, - publication_idle: true, - lease_live: true, - }), - CoordinationDecision::Started - ); - state.step(CoordinationInput::Fence); - assert_eq!( - state.step(CoordinationInput::FinishRenewal { fenced: false }), - CoordinationDecision::Fence - ); - assert!(state.is_fenced()); - assert!(!state.is_renewing()); -} diff --git a/crates/crab-cell-runtime/src/coordination/tests/transfer.rs b/crates/crab-cell-runtime/src/coordination/tests/transfer.rs deleted file mode 100644 index 1f50507e9..000000000 --- a/crates/crab-cell-runtime/src/coordination/tests/transfer.rs +++ /dev/null @@ -1,262 +0,0 @@ -//! Transfer preflight, hydration, and migration transitions. - -use super::*; - -#[test] -fn transfer_preflight_requires_a_quiescent_live_owner() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginTransferPreflight { - queue_empty: false, - publication_idle: true, - lease_live: true, - }), - CoordinationDecision::Started - ); - assert!(state.is_transfer_preparing()); - assert_eq!( - state.step(CoordinationInput::BeginCompaction { - queue_empty: true, - publication_idle: true, - publisher_ready: true, - due: true, - lease_live: true, - }), - CoordinationDecision::Ignored - ); - assert_eq!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Command, - admission_matches: true, - }), - CoordinationDecision::Reject(RejectReason::Draining) - ); - assert_eq!( - state.step(CoordinationInput::AbortTransfer), - CoordinationDecision::Ignored - ); - assert!(!state.is_transfer_preparing()); - assert_eq!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Command, - admission_matches: true, - }), - CoordinationDecision::Admit - ); - - let mut refreshing = CoordinationState::serving(true); - refreshing.begin_effect(CoordinationEffect::Inventory); - assert_eq!( - refreshing.step(CoordinationInput::BeginTransferPreflight { - queue_empty: true, - publication_idle: true, - lease_live: true, - }), - CoordinationDecision::Ignored - ); - - let mut fenced = CoordinationState::serving(true); - assert_eq!( - fenced.step(CoordinationInput::BeginTransferPreflight { - queue_empty: true, - publication_idle: true, - lease_live: false, - }), - CoordinationDecision::Fence - ); -} -#[test] -fn transfer_confirmation_enters_terminal_drain_and_fence_wins() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginTransferPreflight { - queue_empty: true, - publication_idle: true, - lease_live: true, - }), - CoordinationDecision::Started - ); - assert_eq!( - state.step(CoordinationInput::ConfirmTransfer), - CoordinationDecision::ReadyToDeactivate - ); - assert!(state.is_draining()); - assert!(!state.is_transfer_preparing()); - assert_eq!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Query, - admission_matches: true, - }), - CoordinationDecision::Reject(RejectReason::Draining) - ); - - let mut fenced = CoordinationState::serving(true); - fenced.step(CoordinationInput::BeginTransferPreflight { - queue_empty: true, - publication_idle: true, - lease_live: true, - }); - fenced.step(CoordinationInput::Fence); - assert_eq!( - fenced.step(CoordinationInput::ConfirmTransfer), - CoordinationDecision::Reject(RejectReason::Fenced) - ); - assert!(!fenced.is_transfer_preparing()); -} -#[test] -fn hydration_promotion_is_bounded_and_fenced_on_failure() { - let mut state = CoordinationState::serving_with_residency(true, Residency::Sparse); - assert_eq!( - state.step(CoordinationInput::BeginHydration { - queue_empty: true, - publication_idle: true, - lease_live: true, - }), - CoordinationDecision::Started - ); - assert_eq!(state.residency(), Residency::Hydrating); - state.step(CoordinationInput::FinishHydration { - complete: true, - stale: false, - }); - assert_eq!(state.residency(), Residency::Resident); - - let mut failed = CoordinationState::serving_with_residency(true, Residency::Sparse); - failed.step(CoordinationInput::BeginHydration { - queue_empty: true, - publication_idle: true, - lease_live: true, - }); - assert_eq!( - failed.step(CoordinationInput::FinishHydration { - complete: false, - stale: true, - }), - CoordinationDecision::Fence - ); - failed.step(CoordinationInput::Fence); - assert!(failed.is_fenced()); - assert_eq!(failed.residency(), Residency::Sparse); -} -#[test] -fn hydration_completion_preserves_concurrent_foreground_ownership() { - let mut state = CoordinationState::serving_with_residency(true, Residency::Sparse); - assert_eq!( - state.step(CoordinationInput::BeginHydration { - queue_empty: true, - publication_idle: true, - lease_live: true, - }), - CoordinationDecision::Started - ); - let hydration = state.begin_effect(CoordinationEffect::Hydration); - assert_eq!( - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Command, - publisher_ready: true, - }), - CoordinationDecision::Started - ); - let work = state.begin_effect(CoordinationEffect::Work(AdmissionKind::Command)); - state.step(CoordinationInput::CompleteEffect { - effect_id: hydration, - effect: CoordinationEffect::Hydration, - }); - state.step(CoordinationInput::FinishHydration { - complete: true, - stale: false, - }); - assert!(state.is_busy()); - assert_eq!( - state.step(CoordinationInput::BeginDrain), - CoordinationDecision::Started - ); - assert!(!state.can_deactivate()); - state.step(CoordinationInput::CompleteEffect { - effect_id: work, - effect: CoordinationEffect::Work(AdmissionKind::Command), - }); - assert_eq!( - state.step(CoordinationInput::FinishWork { fenced: false }), - CoordinationDecision::ReadyToDeactivate - ); -} - -#[test] -fn migration_queues_successor_work_until_publication_drains() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::BeginPublication); - assert_eq!( - state.step(CoordinationInput::BeginMigration), - CoordinationDecision::Started - ); - assert_eq!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Query, - admission_matches: true, - }), - CoordinationDecision::Admit - ); - assert_eq!( - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Migration, - publisher_ready: true, - }), - CoordinationDecision::Reject(RejectReason::PublicationPending) - ); -} -#[test] -fn migration_requires_a_live_publisher_observation() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::BeginMigration); - assert_eq!( - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Migration, - publisher_ready: false, - }), - CoordinationDecision::Reject(RejectReason::PublicationPending) - ); -} -#[test] -fn shutdown_supersedes_migration_and_keeps_the_drain_terminal() { - let mut state = CoordinationState::serving(true); - state.step(CoordinationInput::BeginMigration); - assert_eq!( - state.step(CoordinationInput::BeginShutdown), - CoordinationDecision::Started - ); - assert!(state.is_shutdown()); - assert!(matches!( - state.step(CoordinationInput::Admit { - kind: AdmissionKind::Query, - admission_matches: true, - }), - CoordinationDecision::Reject(RejectReason::Draining) - )); - assert_eq!( - state.step(CoordinationInput::FinishMigration { fenced: false }), - CoordinationDecision::ReadyToDeactivate - ); - assert!(state.is_shutdown()); -} -#[test] -fn fenced_migration_completion_does_not_reopen_serving_state() { - let mut state = CoordinationState::serving(true); - assert_eq!( - state.step(CoordinationInput::BeginMigration), - CoordinationDecision::Started - ); - assert_eq!( - state.step(CoordinationInput::BeginWork { - kind: AdmissionKind::Migration, - publisher_ready: true, - }), - CoordinationDecision::Started - ); - assert_eq!( - state.step(CoordinationInput::FinishMigration { fenced: true }), - CoordinationDecision::Fence - ); - assert!(state.is_fenced()); - assert!(!state.is_busy()); -} diff --git a/crates/crab-cell-runtime/src/error.rs b/crates/crab-cell-runtime/src/error.rs deleted file mode 100644 index 973802282..000000000 --- a/crates/crab-cell-runtime/src/error.rs +++ /dev/null @@ -1,205 +0,0 @@ -/// Result of a Cell runtime contract operation. -pub type Result = std::result::Result; - -/// Identity, control-codec and schema failures with original causes retained. -#[derive(Debug, thiserror::Error)] -pub enum Error { - /// An identity, digest, or partition argument violates its encoding rules. - #[error("invalid Cell identity: {0}")] - Identity(&'static str), - /// A control record or transition failed validation. - #[error("invalid Cell control record: {0}")] - Control(&'static str), - /// A catalog record, scan, or head failed validation. - #[error("invalid Cell catalog: {0}")] - Catalog(&'static str), - /// A compiled registry rejected a descriptor, binding, or release. - #[error("invalid compiled Cell registry: {0}")] - Registry(&'static str), - /// A Cell application release failed validation. - #[error("invalid Cell application release: {0}")] - Release(&'static str), - /// A backup pin failed validation. - #[error("invalid Cell backup pin: {0}")] - Backup(&'static str), - /// A retention request failed validation. - #[error("invalid Cell retention operation: {0}")] - Retention(&'static str), - /// Retention scratch storage failed locally. - #[error("Cell retention scratch storage failed")] - RetentionIo(#[source] std::io::Error), - /// The retention scratch worker could not be joined. - #[error("failed to join Cell retention scratch worker")] - RetentionWorkerJoin(#[source] tokio::task::JoinError), - /// The Cell ID already names a different catalog entry. - #[error("Cell ID collides with a different catalog entry")] - CatalogCollision, - /// The catalog shard cannot accept another entry. - #[error("Cell catalog shard reached its 65,536-entry limit")] - CatalogFull, - /// A persisted JSON document failed to encode or decode. - #[error("Cell runtime JSON failed")] - Json(#[from] serde_json::Error), - /// The wire codec rejected a value. - #[error("Cell wire codec failed")] - Codec(#[from] crate::codec::CodecError), - /// A peer request or reply violated the peer protocol. - #[error("invalid Cell peer protocol: {0}")] - Peer(&'static str), - /// A peer Protobuf message failed to decode. - #[error("Cell peer Protobuf decoding failed")] - PeerDecode(#[from] prost::DecodeError), - /// A peer transport failed before the operation was accepted. - #[error("Cell peer transport failed: {context}")] - PeerTransport { - /// Facility that failed, for logs and peer replies. - context: &'static str, - /// Transport or facility failure that produced this error. - #[source] - source: Box, - }, - /// The peer transport failed after it may have accepted the operation, so - /// the caller must resolve the request before retrying it. - #[error("Cell peer transport may have accepted the operation: {context}")] - PeerTransportUnknown { - /// Facility that failed, for logs and peer replies. - context: &'static str, - /// Transport or facility failure that produced this error. - #[source] - source: Box, - }, - /// A peer signature did not verify against the advertised key. - #[error("Cell peer signature verification failed")] - PeerSignature(#[source] ed25519_dalek::SignatureError), - /// The peer is authenticated but not authorized for this operation. - #[error("Cell peer authorization denied: {0}")] - PeerAuthorization(&'static str), - /// A node advertisement or directory record failed validation. - #[error("invalid Cell node advertisement: {0}")] - Node(&'static str), - /// Follower storage failed locally. - #[error("Cell follower storage failed")] - FollowerIo(#[from] std::io::Error), - /// The follower storage worker could not be joined. - #[error("failed to join Cell follower storage worker")] - FollowerWorkerJoin(#[source] tokio::task::JoinError), - /// SQLite rejected a statement, schema, or transaction operation. - #[error("Cell runtime SQLite schema failed")] - Sqlite(#[from] rusqlite::Error), - /// SQL text returned by the engine was not valid UTF-8. - #[error("Cell SQL returned invalid UTF-8 text")] - Utf8(#[from] std::str::Utf8Error), - /// Authoritative control storage failed. - #[error("Cell authority storage failed")] - Storage(#[from] crab_storage::StorageError), - /// LTX capture, verification, or publication failed. - #[error("Cell LTX publication failed")] - Ltx(#[from] crab_ltx::CrabError), - /// A command failed validation or its handler refused it. - #[error("invalid Cell command: {0}")] - Command(&'static str), - /// A read replica has not reached the requested Cell position. - #[error("Cell read replica is behind the requested receipt")] - ReplicaBehind { - /// Commit sequence the replica can currently serve. - observed_sequence: u64, - /// Minimum commit sequence required by the caller. - minimum_sequence: u64, - }, - /// No selected local read replica can currently serve the Cell. - #[error("Cell read replica is unavailable")] - ReplicaUnavailable, - /// The request ID was already used with different command bytes. - #[error("request ID was already used for different command bytes")] - RequestConflict, - /// The Cell holds a local commit that must be published durably first. - #[error("Cell has a local commit awaiting durable publication")] - PendingPublication, - /// The executor is fenced until origin recovery completes. - #[error("Cell executor is fenced pending origin recovery")] - Fenced, - /// The runtime is shutting down and refuses new work. - #[error("Cell runtime worker pool is closed")] - RuntimeClosed, - /// The state stream was cancelled before it finished. - #[error("Cell state stream was cancelled")] - StreamCancelled, - /// The Cell is not active on the worker that received the request. - #[error("Cell is not active on its assigned worker")] - CellNotActive, - /// The Cell is already active on this worker. - #[error("Cell is already active on its assigned worker")] - CellAlreadyActive, - /// The Cell is draining and refuses new commands. - #[error("Cell is draining and no longer accepts commands")] - CellDraining, - /// A SQLite operation exceeded its wall-clock deadline. - #[error("Cell SQLite operation exceeded its wall deadline")] - Deadline, - /// An accepted command's outcome is unknown; the caller must resolve the - /// request ID before retrying it. - #[error("accepted Cell command outcome is unknown")] - OutcomeUnknown { - /// Request whose outcome must be resolved. - request_id: crate::identity::RequestId, - /// Digest of the command bytes the request carried. - operation_digest: crate::Digest, - /// Failure observed while resolving the outcome. - #[source] - source: Box, - }, - /// An accepted effect's outcome is unknown; the caller must resolve the - /// effect ID before retrying it. - #[error("accepted Cell effect outcome is unknown")] - EffectOutcomeUnknown { - /// Effect whose outcome must be resolved. - effect_id: [u8; 32], - /// Digest of the operation the effect carried. - operation_digest: crate::Digest, - /// Failure observed while resolving the outcome. - #[source] - source: Box, - }, - /// The effect expired before it could be delivered. - #[error("Cell effect expired before delivery")] - EffectExpired, - /// A bounded runtime resource is exhausted; the message names it. - #[error("Cell runtime capacity exhausted: {0}")] - Capacity(&'static str), - /// The SQL worker could not be started. - #[error("failed to start Cell SQL worker")] - WorkerStart(#[source] Box), - /// The SQL worker supervisor could not be joined. - #[error("failed to join Cell SQL worker supervisor")] - WorkerJoin(#[source] tokio::task::JoinError), - /// A SQL worker panicked during shutdown. - #[error("a Cell SQL worker panicked during shutdown")] - WorkerPanic, - /// A native Cell callback panicked; its activation was fenced. - #[error("a native Cell callback panicked; its activation was fenced")] - NativePanic, - /// The blocking activity worker could not be started. - #[error("failed to start Cell blocking activity worker")] - ActivityWorkerStart(#[source] Box), - /// The blocking activity worker supervisor could not be joined. - #[error("failed to join Cell blocking activity worker supervisor")] - ActivityWorkerJoin(#[source] tokio::task::JoinError), - /// A blocking activity worker panicked during shutdown. - #[error("a Cell blocking activity worker panicked during shutdown")] - ActivityWorkerPanic, - /// A native blocking-activity handler panicked. - #[error("a native Cell blocking activity handler panicked")] - ActivityPanic, - /// The runtime was started outside an active Tokio runtime. - #[error("Cell runtime requires an active Tokio runtime")] - RuntimeStart(#[source] tokio::runtime::TryCurrentError), - /// A provider-owned node facility failed. - #[error("Cell node facility failed: {name}")] - Facility { - /// Facility that failed, for logs and peer replies. - name: &'static str, - /// Transport or facility failure that produced this error. - #[source] - source: Box, - }, -} diff --git a/crates/crab-cell-runtime/src/fleet.rs b/crates/crab-cell-runtime/src/fleet.rs deleted file mode 100644 index 5a321073f..000000000 --- a/crates/crab-cell-runtime/src/fleet.rs +++ /dev/null @@ -1,8 +0,0 @@ -//! Fleet placement, pressure, admission accounting, eviction, and scheduling. - -pub mod eviction; -pub mod placement; -pub mod pressure; -pub mod resource; -pub mod scheduler; -pub mod telemetry; diff --git a/crates/crab-cell-runtime/src/fleet/eviction.rs b/crates/crab-cell-runtime/src/fleet/eviction.rs deleted file mode 100644 index 38c760b39..000000000 --- a/crates/crab-cell-runtime/src/fleet/eviction.rs +++ /dev/null @@ -1,137 +0,0 @@ -//! Idle Cell eviction: observations, state, and victim selection. -use crate::fleet::resource::ResourceCost; -use crate::identity::CellId; - -/// Lifecycle class used when choosing a local Cell to evict. -/// -/// The selector only accepts cells that are already idle or quiescing. The -/// actor remains responsible for driving the selected cell through the -/// durability and close protocol before releasing its reservation. -#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] -pub(crate) enum EvictionState { - Idle, - Quiescing, -} - -/// Immutable actor observation consumed by the deterministic victim selector. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(crate) struct EvictionObservation { - pub(crate) cell: CellId, - pub(crate) state: EvictionState, - pub(crate) last_used_ms: i64, - pub(crate) cost: ResourceCost, - pub(crate) busy: bool, - pub(crate) retained_obligation: bool, - pub(crate) migrating: bool, - pub(crate) backup_pinned: bool, - pub(crate) leased_work: bool, - pub(crate) primitive_obligation: bool, - pub(crate) accounting_known: bool, -} - -impl EvictionObservation { - pub(crate) fn eligible(self) -> bool { - self.accounting_known - && !self.busy - && !self.retained_obligation - && !self.migrating - && !self.backup_pinned - && !self.leased_work - && !self.primitive_obligation - && self.last_used_ms >= 0 - } -} - -/// Selects at most `limit` safe victims with a total, permutation-invariant order. -pub(crate) fn select_victims(observations: &[EvictionObservation], limit: usize) -> Vec { - let mut candidates = observations - .iter() - .copied() - .filter(|observation| observation.eligible()) - .collect::>(); - candidates.sort_by(|left, right| { - left.last_used_ms - .cmp(&right.last_used_ms) - .then_with(|| right.state.cmp(&left.state)) - .then_with(|| right.cost.active_cells().cmp(&left.cost.active_cells())) - .then_with(|| right.cost.retained_bytes().cmp(&left.cost.retained_bytes())) - .then_with(|| right.cost.disk_bytes().cmp(&left.cost.disk_bytes())) - .then_with(|| left.cell.as_bytes().cmp(right.cell.as_bytes())) - }); - candidates - .into_iter() - .take(limit) - .map(|candidate| candidate.cell) - .collect() -} - -#[cfg(test)] -mod tests { - use super::*; - - fn observation(cell: u8, last_used_ms: i64) -> EvictionObservation { - EvictionObservation { - cell: CellId::from_bytes([cell; 32]), - state: EvictionState::Idle, - last_used_ms, - cost: ResourceCost::active_cell() - .with_retained_bytes(64) - .with_disk_bytes(128), - busy: false, - retained_obligation: false, - migrating: false, - backup_pinned: false, - leased_work: false, - primitive_obligation: false, - accounting_known: true, - } - } - - #[test] - fn unsafe_cells_are_never_selected() { - let mut busy = observation(1, 1); - busy.busy = true; - let mut retained = observation(2, 2); - retained.retained_obligation = true; - let mut migrating = observation(3, 3); - migrating.migrating = true; - let mut pinned = observation(4, 4); - pinned.backup_pinned = true; - let mut leased = observation(5, 5); - leased.leased_work = true; - let mut unknown = observation(6, 6); - unknown.accounting_known = false; - let mut primitive = observation(7, 7); - primitive.primitive_obligation = true; - assert!( - select_victims( - &[ - busy, retained, migrating, pinned, leased, unknown, primitive - ], - 10 - ) - .is_empty() - ); - } - - #[test] - fn selection_is_oldest_first_and_permutation_invariant() { - let first = observation(1, 20); - let second = observation(2, 10); - let third = observation(3, 30); - let left = select_victims(&[first, second, third], 2); - let right = select_victims(&[third, first, second], 2); - assert_eq!(left, right); - assert_eq!(left, vec![second.cell, first.cell]); - } - - #[test] - fn tie_break_prefers_more_reclaimable_cost_then_cell_id() { - let mut smaller = observation(1, 10); - let mut larger = observation(2, 10); - larger.cost = larger.cost.with_disk_bytes(512); - assert_eq!(select_victims(&[smaller, larger], 1), vec![larger.cell]); - smaller.cost = larger.cost; - assert_eq!(select_victims(&[smaller, larger], 1), vec![smaller.cell]); - } -} diff --git a/crates/crab-cell-runtime/src/fleet/placement.rs b/crates/crab-cell-runtime/src/fleet/placement.rs deleted file mode 100644 index dc39f6c62..000000000 --- a/crates/crab-cell-runtime/src/fleet/placement.rs +++ /dev/null @@ -1,722 +0,0 @@ -//! Signed-capacity placement planning and scoring. -use std::cmp::Ordering; -use std::collections::HashSet; - -use crate::identity::NodeId; -use crate::identity::{CellId, SessionId}; -use crate::node::{NodeAdvertisement, NodePlacementCapacity}; -use crate::{Error, Result}; - -const MAX_OBSERVATION_AGE_MS: i64 = 30_000; -const SCORE_SCALE: u128 = 1_000; -const MIN_TRANSFER_GAIN: u128 = 50_000; -const MIN_RESIDENCE_MS: i64 = 60_000; -const MAX_TRANSFERS_PER_TICK: usize = 2; -const MAX_TRANSFER_BYTES_PER_TICK: u64 = 8 * 1024 * 1024 * 1024; -/// A receiver reserves this whole-Cell fraction of its weighted target, while -/// the donor drains to its target exactly. The receiver's margin absorbs the -/// sample lag between a batch and its publication: a stale count can overshoot -/// by one batch, and the fleet cannot end up short by one deadband per donor. -const BALANCE_DEADBAND_PERCENT: u128 = 2; - -/// Pressure class supplied by the signed node observation. -/// -/// The signed placement block carries measured counters rather than the node's -/// hysteretic tier, so derivation currently reaches only the critical class; -/// soft and hard pressure remain local until the tier is signed. -#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] -pub enum PlacementPressure { - /// No pressure: the node accepts placements. - Normal, - /// Soft pressure: the node paces optional work. - Constrained, - /// Hard pressure on one dimension: the node sheds load. - Shedding, - /// Critical pressure on both dimensions: the node takes no new ownership. - Critical, -} - -/// Authenticated, bounded node values consumed by the pure placement planner. -/// -/// Authentication and timestamp validation remain the advertisement owner's -/// responsibility. The planner treats a missing/invalid observation as -/// ineligible rather than interpreting it as zero load. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct PlacementObservation { - /// Node that produced the observation. - pub node: NodeId, - /// Boot session that signed it. - pub session: SessionId, - /// Logical time the sample was taken. - pub observed_at_ms: i64, - /// Total memory the node reports. - pub memory_capacity_bytes: u64, - /// Free memory at sample time. - pub free_memory_bytes: u64, - /// Total scratch disk the node reports. - pub disk_capacity_bytes: u64, - /// Free scratch disk at sample time. - pub free_disk_bytes: u64, - /// Cells the node currently owns. - pub active_cells: u32, - /// Cells the node admits. - pub max_active_cells: u32, - /// Jobs the node is running. - pub running_jobs: u32, - /// Jobs the node admits. - pub job_capacity: u32, - /// Publications waiting to be acknowledged. - pub publication_backlog: u32, - /// Hydrations waiting to run. - pub hydration_backlog: u32, - /// Primitive maintenance items waiting. - pub primitive_backlog: u32, - /// Pressure class carried by the signed observation. - pub pressure: PlacementPressure, - /// Whether the node is shedding ownership. - pub draining: bool, - /// Whether the signature and identity checks passed. - pub authenticated: bool, - /// Whether the node currently owns the Cell being placed. - pub current_owner: bool, -} - -impl PlacementObservation { - /// Converts the signed runtime block carried by a live advertisement into - /// planner input without substituting host-wide or zero-valued guesses. - pub fn from_signed_advertisement( - advertisement: &NodeAdvertisement, - now_ms: i64, - current_owner: bool, - ) -> Result { - let placement = advertisement - .placement_capacity() - .ok_or(Error::Node("placement snapshot is missing"))?; - let capacity = advertisement.capacity(); - if now_ms < 0 - || advertisement.expires_at_ms() <= now_ms - || !advertisement.has_signed_placement() - { - return Err(Error::Node("placement advertisement is stale")); - } - Ok(Self::from_signed_capacity( - advertisement, - placement, - capacity, - now_ms, - current_owner, - )) - } - - fn from_signed_capacity( - advertisement: &NodeAdvertisement, - placement: NodePlacementCapacity, - capacity: crate::node::NodeCapacity, - _now_ms: i64, - current_owner: bool, - ) -> Self { - Self { - node: advertisement.node(), - session: advertisement.session(), - observed_at_ms: advertisement.issued_at_ms(), - memory_capacity_bytes: placement.memory_capacity_bytes, - free_memory_bytes: capacity.free_memory_bytes, - disk_capacity_bytes: placement.disk_capacity_bytes, - free_disk_bytes: capacity.free_disk_bytes, - active_cells: placement.active_cells, - max_active_cells: placement.max_active_cells, - running_jobs: placement.running_jobs, - job_capacity: placement.job_capacity, - publication_backlog: placement.publication_backlog, - hydration_backlog: placement.hydration_backlog, - primitive_backlog: placement.primitive_backlog, - // Only an empty ledger is visible to a peer while the hysteretic - // tier stays local. - pressure: if capacity.free_memory_bytes == 0 || capacity.free_disk_bytes == 0 { - PlacementPressure::Critical - } else { - PlacementPressure::Normal - }, - draining: capacity.free_memory_bytes == 0 || capacity.free_disk_bytes == 0, - authenticated: true, - current_owner, - } - } -} - -/// Stable reason for accepting or rejecting one placement candidate. -#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] -pub enum PlacementEligibility { - /// The node may receive the Cell. - Eligible, - /// The observation failed signature or identity checks. - Unauthenticated, - /// The observation is older than the planner's freshness window. - Stale, - /// The node is draining. - Draining, - /// The node reports critical pressure. - CriticalPressure, - /// Free memory is below the placement reserve. - NoMemoryHeadroom, - /// Free disk is below the placement reserve. - NoDiskHeadroom, - /// The node has no room for another Cell. - NoCellCapacity, - /// The node has no room for another job. - NoJobCapacity, -} - -/// Score and eligibility explanation for one candidate node. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct PlacementScore { - /// Candidate node. - pub node: NodeId, - /// Boot session the score applies to. - pub session: SessionId, - /// Weighted headroom score; larger is better. - pub score: u128, - /// Why the node was accepted or rejected. - pub eligibility: PlacementEligibility, -} - -/// Actor-sampled demand for one locally owned Cell. A missing or unsettled -/// sample cannot be used as a transfer hint; the actor must recheck on release. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct CellTransferDemand { - /// Cell the sample describes. - pub cell: CellId, - /// Session that currently owns the Cell. - pub source: SessionId, - /// Ownership generation the sample was taken under. - pub generation: u64, - /// Resident memory the Cell uses. - pub memory_bytes: u64, - /// Scratch disk the Cell uses. - pub disk_bytes: u64, - /// Job credits the Cell holds. - pub job_credits: u32, - /// Logical time the Cell became resident. - pub resident_since_ms: i64, - /// Logical time the Cell last moved, when it has. - pub last_moved_at_ms: Option, - /// Consecutive settled samples observed for this Cell. - pub stable_observations: u8, - /// Whether the sample is settled enough to act on. - pub settled: bool, -} - -/// Advisory transfer proposal. It conveys neither release nor receiver admission. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct CellTransferIntent { - /// Cell proposed for movement. - pub cell: CellId, - /// Session that currently owns it. - pub source: SessionId, - /// Ownership generation the proposal was made under. - pub generation: u64, - /// Session proposed to receive it. - pub destination: SessionId, - /// Logical time the proposal was derived from. - pub observed_at_ms: i64, - /// Resident memory the receiver must admit. - pub memory_bytes: u64, - /// Scratch disk the receiver must admit. - pub disk_bytes: u64, - /// Job credits the receiver must admit. - pub job_credits: u32, -} - -/// Weighted ownership balance of one complete fleet snapshot. -/// -/// This is a count rule, not a resource rule: a node's weight is its declared -/// cell capacity, so a node sized for twice the Cells carries twice the -/// target. The densest member by Cells per unit of weight is the only donor, -/// and one snapshot has at most one donor, so the fleet hands over from one -/// place instead of every node pushing at once. A wrong count costs movement, -/// never authority: every release still crosses the actor gate and the exact -/// control CAS. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct FleetBalance { - /// Boot session elected to donate from this snapshot. - pub donor: SessionId, - /// Cells the donor may release now, bounded by the fleet batch and by the - /// receivers' room below their deadband. - pub surplus: usize, - /// Sessions below their weighted target that can absorb a donation, - /// least dense first. - pub receivers: Vec, -} - -/// Deterministic weighted placement policy. It never mutates authority. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct PlacementPlanner { - max_observation_age_ms: i64, -} - -impl Default for PlacementPlanner { - fn default() -> Self { - Self { - max_observation_age_ms: MAX_OBSERVATION_AGE_MS, - } - } -} - -impl PlacementPlanner { - /// Creates a planner with a bounded observation freshness window. - pub fn new(max_observation_age_ms: i64) -> Result { - if max_observation_age_ms <= 0 { - return Err(Error::Control("placement freshness window is invalid")); - } - Ok(Self { - max_observation_age_ms, - }) - } - - /// Ranks candidates for one Cell with a total, permutation-invariant order. - pub fn rank( - &self, - cell: CellId, - now_ms: i64, - observations: &[PlacementObservation], - ) -> Result> { - if now_ms < 0 { - return Err(Error::Control("placement time is invalid")); - } - let mut scores = observations - .iter() - .map(|observation| self.score(cell, now_ms, *observation)) - .collect::>(); - scores.sort_by(|left, right| { - right - .score - .cmp(&left.score) - .then_with(|| right.eligibility.cmp(&left.eligibility)) - .then_with(|| left.node.as_bytes().cmp(right.node.as_bytes())) - .then_with(|| left.session.as_bytes().cmp(right.session.as_bytes())) - }); - Ok(scores) - } - - /// Returns the best eligible node, if any. - pub fn choose( - &self, - cell: CellId, - now_ms: i64, - observations: &[PlacementObservation], - ) -> Result> { - Ok(self - .rank(cell, now_ms, observations)? - .into_iter() - .find(|score| score.eligibility == PlacementEligibility::Eligible)) - } - - /// Computes the fleet's weighted ownership balance from one complete, - /// authenticated snapshot. - /// - /// `since_ms` is the instant the caller's previous movement batch was - /// dispatched. A sample taken at or before it may already be stale about - /// that batch, so the whole view fails closed: a mixed total would lower - /// every target and move Cells that come straight back. `None` therefore - /// means "move nothing on the count rule" — an empty, duplicated, stale, or - /// already balanced fleet. Pressure shedding and drains move Cells through - /// their own gates and are never weakened by this method. - pub fn fleet_balance( - &self, - now_ms: i64, - observations: &[PlacementObservation], - since_ms: i64, - ) -> Result> { - if now_ms < 0 || observations.len() > 10_000 { - return Err(Error::Control("fleet balance snapshot is invalid")); - } - let mut nodes = HashSet::new(); - let mut sessions = HashSet::new(); - let mut owned = 0u128; - let mut weight = 0u128; - for observation in observations { - if !nodes.insert(observation.node) || !sessions.insert(observation.session) { - return Err(Error::Control("fleet balance snapshot duplicates a node")); - } - if !self.sampled_after(now_ms, *observation, since_ms) { - return Ok(None); - } - owned += u128::from(observation.active_cells); - weight += u128::from(observation.max_active_cells); - } - if weight == 0 { - return Ok(None); - } - let target = |observation: &PlacementObservation| -> u128 { - (owned * u128::from(observation.max_active_cells)).div_ceil(weight) - }; - let Some(donor) = observations - .iter() - .max_by(|left, right| denser(left, right)) - else { - return Ok(None); - }; - let donor_surplus = u128::from(donor.active_cells).saturating_sub(target(donor)); - let mut receivers = observations - .iter() - .filter(|observation| { - observation.session != donor.session - && room(observation, target(observation)) > 0 - && observation.pressure < PlacementPressure::Shedding - && self.eligibility(now_ms, **observation) == PlacementEligibility::Eligible - }) - .collect::>(); - receivers.sort_by(|left, right| denser(left, right)); - let room = receivers - .iter() - .map(|observation| room(observation, target(observation))) - .sum::(); - let count = usize::try_from(donor_surplus.min(room)).unwrap_or(MAX_TRANSFERS_PER_TICK); - let count = count.min(MAX_TRANSFERS_PER_TICK); - if count == 0 { - return Ok(None); - } - Ok(Some(FleetBalance { - donor: donor.session, - surplus: count, - receivers: receivers - .into_iter() - .map(|observation| observation.session) - .collect(), - })) - } - - /// Plans at most two settled transfers and 8 GiB of projected restore - /// bytes from one authenticated fleet snapshot. Receiver capacity is - /// projected across selected intents, then rechecked during activation. - /// - /// `balance` is the ownership balance of the same snapshot. When it elects - /// one of the demands' sources as the donor, its receivers may absorb that - /// bounded batch without the headroom-gain gate: the fleet owes those - /// Cells, and their idle headroom need not also be worse. Every other gate - /// — settlement, residence, cooldown, eligibility, and projected receiver - /// capacity — still applies to a balancing move. - pub fn plan_transfers( - &self, - now_ms: i64, - observations: &[PlacementObservation], - demands: &[CellTransferDemand], - balance: Option<&FleetBalance>, - ) -> Result> { - if now_ms < 0 || observations.len() > 10_000 || demands.len() > 10_000 { - return Err(Error::Control("transfer snapshot is invalid")); - } - let mut nodes = HashSet::new(); - let mut sessions = HashSet::new(); - for observation in observations { - if !nodes.insert(observation.node) || !sessions.insert(observation.session) { - return Err(Error::Control("transfer snapshot duplicates a node")); - } - } - let mut cells = HashSet::new(); - for demand in demands { - if !cells.insert(demand.cell) { - return Err(Error::Control("transfer snapshot duplicates a Cell")); - } - } - let mut projected = observations.to_vec(); - for observation in &mut projected { - observation.current_owner = false; - } - let accepting = - balance.map(|balance| balance.receivers.iter().copied().collect::>()); - let owned = observations - .iter() - .map(|node| u128::from(node.active_cells)) - .sum::(); - let weight = observations - .iter() - .map(|node| u128::from(node.max_active_cells)) - .sum::(); - let receiver_room = |node: &PlacementObservation| { - let target = if weight == 0 { - 0 - } else { - (owned * u128::from(node.max_active_cells)).div_ceil(weight) - }; - room(node, target) - }; - let mut candidates = demands.to_vec(); - candidates.sort_by(|left, right| { - let priority = |demand: &CellTransferDemand| { - observations - .iter() - .find(|node| node.session == demand.source) - .map_or(0, |node| { - u8::from(node.draining) * 2 - + u8::from(node.pressure >= PlacementPressure::Shedding) - }) - }; - priority(right) - .cmp(&priority(left)) - .then_with(|| left.cell.as_bytes().cmp(right.cell.as_bytes())) - }); - let mut intents = Vec::new(); - let mut bytes = 0u64; - let mut donations = 0usize; - for demand in candidates { - if intents.len() == MAX_TRANSFERS_PER_TICK { - break; - } - let Some(source) = observations - .iter() - .find(|node| node.session == demand.source) - else { - continue; - }; - if !demand.settled - || demand.generation == 0 - || demand.memory_bytes == 0 - || demand.disk_bytes == 0 - || demand.job_credits == 0 - || demand.resident_since_ms < 0 - || demand.resident_since_ms > now_ms - || demand - .last_moved_at_ms - .is_some_and(|at| at < 0 || at > now_ms) - || !source.authenticated - || self.eligibility(now_ms, *source) == PlacementEligibility::Stale - { - continue; - } - let urgent = source.draining || source.pressure >= PlacementPressure::Shedding; - if !urgent - && (demand.stable_observations < 2 - || now_ms - demand.resident_since_ms < MIN_RESIDENCE_MS - || demand - .last_moved_at_ms - .is_some_and(|at| now_ms - at < MIN_RESIDENCE_MS)) - { - continue; - } - let Some(next_bytes) = bytes.checked_add(demand.disk_bytes) else { - continue; - }; - if next_bytes > MAX_TRANSFER_BYTES_PER_TICK { - continue; - } - let mut owned_source = *source; - owned_source.current_owner = true; - let source_score = self.score(demand.cell, now_ms, owned_source).score; - let donates = balance.is_some_and(|balance| { - balance.donor == demand.source && donations < balance.surplus - }); - let destination = self - .rank(demand.cell, now_ms, &projected)? - .into_iter() - .find(|score| { - score.session != demand.source - && score.eligibility == PlacementEligibility::Eligible - && projected - .iter() - .find(|node| node.session == score.session) - .is_some_and(|node| { - node.free_memory_bytes >= demand.memory_bytes - && node.free_disk_bytes >= demand.disk_bytes - && node.max_active_cells.saturating_sub(node.active_cells) >= 1 - && node.job_capacity.saturating_sub(node.running_jobs) - >= demand.job_credits - && (urgent - || (donates - && accepting.as_ref().is_some_and(|accepting| { - accepting.contains(&score.session) - }) - && receiver_room(node) > 0) - || score.score - >= source_score.saturating_add(MIN_TRANSFER_GAIN)) - }) - }); - let Some(destination) = destination else { - continue; - }; - let Some(receiver) = projected - .iter_mut() - .find(|node| node.session == destination.session) - else { - continue; - }; - // Charge a donation only while its projected receiver has room. - // Fleet-wide room alone lets a preferred receiver take the whole - // batch and overshoot its share, despite other empty receivers. - let donation = !urgent - && donates - && accepting - .as_ref() - .is_some_and(|accepting| accepting.contains(&destination.session)) - && receiver_room(receiver) > 0; - receiver.free_memory_bytes -= demand.memory_bytes; - receiver.free_disk_bytes -= demand.disk_bytes; - receiver.active_cells += 1; - receiver.running_jobs += demand.job_credits; - bytes = next_bytes; - if donation { - donations += 1; - } - intents.push(CellTransferIntent { - cell: demand.cell, - source: demand.source, - generation: demand.generation, - destination: destination.session, - observed_at_ms: source.observed_at_ms, - memory_bytes: demand.memory_bytes, - disk_bytes: demand.disk_bytes, - job_credits: demand.job_credits, - }); - } - Ok(intents) - } - - fn score( - &self, - cell: CellId, - now_ms: i64, - observation: PlacementObservation, - ) -> PlacementScore { - let eligibility = self.eligibility(now_ms, observation); - if eligibility != PlacementEligibility::Eligible { - return PlacementScore { - node: observation.node, - session: observation.session, - score: 0, - eligibility, - }; - } - let memory = ratio( - observation.free_memory_bytes, - observation.memory_capacity_bytes, - ); - let disk = ratio(observation.free_disk_bytes, observation.disk_capacity_bytes); - let cells = ratio( - u64::from( - observation - .max_active_cells - .saturating_sub(observation.active_cells), - ), - u64::from(observation.max_active_cells), - ); - let jobs = ratio( - u64::from( - observation - .job_capacity - .saturating_sub(observation.running_jobs), - ), - u64::from(observation.job_capacity), - ); - let backlog = u128::from( - observation - .publication_backlog - .saturating_add(observation.hydration_backlog) - .saturating_add(observation.primitive_backlog), - ); - let sticky = u128::from(observation.current_owner) * 50; - let hash_bonus = placement_hash(cell, observation.node, observation.session) % 100; - let score = (memory * 400 - + disk * 250 - + cells * 200 - + jobs * 100 - + sticky * SCORE_SCALE - + hash_bonus) - .saturating_sub(backlog.min(100) * SCORE_SCALE); - PlacementScore { - node: observation.node, - session: observation.session, - score, - eligibility, - } - } - - fn eligibility(&self, now_ms: i64, observation: PlacementObservation) -> PlacementEligibility { - if !self.fresh(now_ms, observation) { - return if observation.authenticated { - PlacementEligibility::Stale - } else { - PlacementEligibility::Unauthenticated - }; - } - if observation.draining { - return PlacementEligibility::Draining; - } - if observation.pressure >= PlacementPressure::Critical { - return PlacementEligibility::CriticalPressure; - } - if observation.memory_capacity_bytes == 0 || observation.free_memory_bytes == 0 { - return PlacementEligibility::NoMemoryHeadroom; - } - if observation.disk_capacity_bytes == 0 || observation.free_disk_bytes == 0 { - return PlacementEligibility::NoDiskHeadroom; - } - if observation.max_active_cells == 0 - || observation.active_cells >= observation.max_active_cells - { - return PlacementEligibility::NoCellCapacity; - } - if observation.job_capacity == 0 || observation.running_jobs >= observation.job_capacity { - return PlacementEligibility::NoJobCapacity; - } - PlacementEligibility::Eligible - } - - /// Whether one member is authenticated and inside the freshness window. - fn fresh(&self, now_ms: i64, observation: PlacementObservation) -> bool { - observation.authenticated - && observation.observed_at_ms >= 0 - && observation.observed_at_ms <= now_ms - && now_ms.saturating_sub(observation.observed_at_ms) <= self.max_observation_age_ms - } - - /// Whether one member's sample can still describe the fleet after the - /// caller's previous movement batch. - fn sampled_after(&self, now_ms: i64, observation: PlacementObservation, since_ms: i64) -> bool { - self.fresh(now_ms, observation) && observation.observed_at_ms > since_ms - } -} - -fn ratio(numerator: u64, denominator: u64) -> u128 { - if denominator == 0 { - 0 - } else { - u128::from(numerator.min(denominator)) * SCORE_SCALE / u128::from(denominator) - } -} - -/// Orders two members by owned Cells per unit of weight without division. -/// The node id, then the session, breaks ties so one snapshot elects exactly -/// one donor. -fn denser(left: &PlacementObservation, right: &PlacementObservation) -> Ordering { - (u128::from(left.active_cells) * u128::from(right.max_active_cells)) - .cmp(&(u128::from(right.active_cells) * u128::from(left.max_active_cells))) - .then_with(|| right.node.as_bytes().cmp(left.node.as_bytes())) - .then_with(|| right.session.as_bytes().cmp(left.session.as_bytes())) -} - -/// Cells one member can still take before it reaches its deadband below its -/// weighted target. A validated snapshot keeps every member at or below its -/// declared Cell capacity, so the target itself already bounds the slots. -fn room(observation: &PlacementObservation, target: u128) -> u128 { - // A fractional Cell cannot be withheld: rounding up makes a target of - // one unreachable and strands concentrated ownership after scale-out. - let deadband = target * BALANCE_DEADBAND_PERCENT / 100; - target - .saturating_sub(deadband) - .saturating_sub(u128::from(observation.active_cells)) -} - -fn placement_hash(cell: CellId, node: NodeId, session: SessionId) -> u128 { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.cell.placement.v1\0"); - hasher.update(cell.as_bytes()); - hasher.update(node.as_bytes()); - hasher.update(session.as_bytes()); - let digest = hasher.finalize(); - let mut bytes = [0; 16]; - bytes.copy_from_slice(&digest.as_bytes()[..16]); - u128::from_be_bytes(bytes) -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/fleet/placement/tests.rs b/crates/crab-cell-runtime/src/fleet/placement/tests.rs deleted file mode 100644 index 50922668f..000000000 --- a/crates/crab-cell-runtime/src/fleet/placement/tests.rs +++ /dev/null @@ -1,432 +0,0 @@ -use super::*; - -fn observation(byte: u8) -> PlacementObservation { - PlacementObservation { - node: NodeId::from_bytes([byte; 16]), - session: SessionId::from_bytes([byte + 1; 16]), - observed_at_ms: 100, - memory_capacity_bytes: 1_000, - free_memory_bytes: 700, - disk_capacity_bytes: 1_000, - free_disk_bytes: 700, - active_cells: 1, - max_active_cells: 10, - running_jobs: 1, - job_capacity: 10, - publication_backlog: 0, - hydration_backlog: 0, - primitive_backlog: 0, - pressure: PlacementPressure::Normal, - draining: false, - authenticated: true, - current_owner: false, - } -} - -#[test] -fn ranking_is_permutation_invariant_and_total() { - let planner = PlacementPlanner::default(); - let left = planner - .rank( - CellId::from_bytes([9; 32]), - 100, - &[observation(1), observation(2)], - ) - .unwrap(); - let right = planner - .rank( - CellId::from_bytes([9; 32]), - 100, - &[observation(2), observation(1)], - ) - .unwrap(); - assert_eq!(left, right); - assert!(left[0].score > 0); -} - -#[test] -fn stale_and_full_nodes_are_not_candidates() { - let planner = PlacementPlanner::default(); - let mut stale = observation(1); - stale.observed_at_ms = -1; - let mut full = observation(2); - full.active_cells = full.max_active_cells; - let scores = planner - .rank(CellId::from_bytes([9; 32]), 100, &[stale, full]) - .unwrap(); - assert!( - scores - .iter() - .all(|score| score.eligibility != PlacementEligibility::Eligible) - ); - assert!( - planner - .choose(CellId::from_bytes([9; 32]), 100, &[stale, full]) - .unwrap() - .is_none() - ); -} - -#[test] -fn worsening_headroom_cannot_improve_score() { - let planner = PlacementPlanner::default(); - let healthy = observation(1); - let mut worse = healthy; - worse.free_memory_bytes = 100; - worse.free_disk_bytes = 100; - let healthy_score = planner - .choose(CellId::from_bytes([9; 32]), 100, &[healthy]) - .unwrap() - .unwrap() - .score; - let worse_score = planner - .choose(CellId::from_bytes([9; 32]), 100, &[worse]) - .unwrap() - .unwrap() - .score; - assert!(worse_score < healthy_score); -} - -#[test] -fn measured_backlog_reduces_placement_score() { - let planner = PlacementPlanner::default(); - let healthy = observation(1); - let mut busy = healthy; - busy.publication_backlog = 3; - busy.hydration_backlog = 2; - busy.primitive_backlog = 1; - let healthy_score = planner - .choose(CellId::from_bytes([9; 32]), 100, &[healthy]) - .unwrap() - .unwrap() - .score; - let busy_score = planner - .choose(CellId::from_bytes([9; 32]), 100, &[busy]) - .unwrap() - .unwrap() - .score; - assert!(busy_score < healthy_score); -} - -fn demand(cell: u8, source: SessionId) -> CellTransferDemand { - CellTransferDemand { - cell: CellId::from_bytes([cell; 32]), - source, - generation: 1, - memory_bytes: 400, - disk_bytes: 400, - job_credits: 1, - resident_since_ms: 0, - last_moved_at_ms: None, - stable_observations: 2, - settled: true, - } -} - -#[test] -fn transfer_projection_prevents_receiver_overcommit() { - let planner = PlacementPlanner::default(); - let mut donor = observation(1); - donor.draining = true; - let receiver = observation(2); - let first = demand(1, donor.session); - let second = demand(2, donor.session); - let left = planner - .plan_transfers(100, &[donor, receiver], &[first, second], None) - .unwrap(); - let right = planner - .plan_transfers(100, &[receiver, donor], &[second, first], None) - .unwrap(); - assert_eq!(left, right); - assert_eq!(left.len(), 1); - assert_eq!(left[0].cell, first.cell); -} - -#[test] -fn transfer_requires_settlement_residence_and_fresh_destination() { - let planner = PlacementPlanner::default(); - let donor = observation(1); - let mut receiver = observation(2); - receiver.free_memory_bytes = 1_000; - receiver.free_disk_bytes = 1_000; - let mut candidate = demand(1, donor.session); - assert!( - planner - .plan_transfers(100, &[donor, receiver], &[candidate], None) - .unwrap() - .is_empty() - ); - candidate.settled = false; - assert!( - planner - .plan_transfers(100_000, &[donor, receiver], &[candidate], None) - .unwrap() - .is_empty() - ); - candidate.settled = true; - receiver.observed_at_ms = 100_000; - assert_eq!( - planner - .plan_transfers(100_000, &[donor, receiver], &[candidate], None) - .unwrap() - .len(), - 0 - ); - let mut donor = donor; - donor.draining = true; - receiver.observed_at_ms = 100; - assert!( - planner - .plan_transfers(100_000, &[donor, receiver], &[candidate], None) - .unwrap() - .is_empty() - ); -} - -#[test] -fn duplicate_nodes_and_cells_fail_closed() { - let planner = PlacementPlanner::default(); - let node = observation(1); - let candidate = demand(1, node.session); - assert!( - planner - .plan_transfers(100, &[node, node], &[candidate], None) - .is_err() - ); - assert!( - planner - .plan_transfers(100, &[node], &[candidate, candidate], None) - .is_err() - ); -} - -#[test] -fn normal_transfer_requires_stable_gain_and_cooldown() { - let planner = PlacementPlanner::default(); - let mut donor = observation(1); - donor.observed_at_ms = 100_000; - donor.free_memory_bytes = 100; - donor.free_disk_bytes = 100; - let mut receiver = observation(2); - receiver.observed_at_ms = 100_000; - receiver.free_memory_bytes = 1_000; - receiver.free_disk_bytes = 1_000; - let mut candidate = demand(1, donor.session); - candidate.memory_bytes = 100; - candidate.disk_bytes = 100; - assert_eq!( - planner - .plan_transfers(100_000, &[donor, receiver], &[candidate], None) - .unwrap() - .len(), - 1 - ); - candidate.last_moved_at_ms = Some(99_999); - assert!( - planner - .plan_transfers(100_000, &[donor, receiver], &[candidate], None) - .unwrap() - .is_empty() - ); - candidate.last_moved_at_ms = None; - candidate.stable_observations = 1; - assert!( - planner - .plan_transfers(100_000, &[donor, receiver], &[candidate], None) - .unwrap() - .is_empty() - ); -} - -#[test] -fn transfer_budget_counts_projected_restore_bytes() { - let planner = PlacementPlanner::default(); - let mut donor = observation(1); - donor.draining = true; - let mut receiver = observation(2); - receiver.disk_capacity_bytes = 12 * 1024 * 1024 * 1024; - receiver.free_disk_bytes = receiver.disk_capacity_bytes; - let mut first = demand(1, donor.session); - first.disk_bytes = 5 * 1024 * 1024 * 1024; - let mut second = demand(2, donor.session); - second.disk_bytes = first.disk_bytes; - assert_eq!( - planner - .plan_transfers(100, &[donor, receiver], &[first, second], None) - .unwrap() - .len(), - 1 - ); -} - -fn owned(byte: u8, cells: u32, slots: u32) -> PlacementObservation { - let mut observation = observation(byte); - observation.observed_at_ms = 100_000; - observation.active_cells = cells; - observation.max_active_cells = slots; - observation -} - -#[test] -fn balance_elects_one_weighted_donor_and_orders_receivers() { - let planner = PlacementPlanner::default(); - let dense = owned(1, 20, 10); - let wide = owned(2, 5, 20); - let small = owned(3, 5, 10); - let left = planner - .fleet_balance(100_000, &[dense, wide, small], 0) - .unwrap() - .unwrap(); - let right = planner - .fleet_balance(100_000, &[small, dense, wide], 0) - .unwrap() - .unwrap(); - assert_eq!(left, right); - assert_eq!(left.donor, dense.session); - // Twelve Cells over target, thirteen Cells of receiver room, and a - // two-Cell batch: the batch is the binding limit. - assert_eq!(left.surplus, 2); - assert_eq!(left.receivers, vec![wide.session, small.session]); - assert!(left.receivers.contains(&wide.session)); - assert!(!left.receivers.contains(&dense.session)); -} - -#[test] -fn balance_breaks_a_density_tie_by_node_identity() { - let planner = PlacementPlanner::default(); - let left = owned(1, 10, 10); - let right = owned(2, 10, 10); - let idle = owned(3, 0, 10); - let balance = planner - .fleet_balance(100_000, &[left, right, idle], 0) - .unwrap() - .unwrap(); - assert_eq!(balance.donor, left.session); - assert_eq!(balance.receivers, vec![idle.session]); - let balance = planner - .fleet_balance(100_000, &[right, left, idle], 0) - .unwrap() - .unwrap(); - assert_eq!(balance.donor, left.session); -} - -#[test] -fn balance_deadband_bounds_the_donation() { - let planner = PlacementPlanner::default(); - let donor = owned(1, 103, 200); - let roomy = owned(2, 97, 200); - // Each member targets 100 Cells. The receiver's two-Cell margin leaves - // one donation slot, even though the donor could fill the two-Cell batch. - let balance = planner - .fleet_balance(100_000, &[donor, roomy], 0) - .unwrap() - .unwrap(); - assert_eq!(balance.surplus, 1); - assert_eq!(balance.receivers, vec![roomy.session]); -} - -#[test] -fn balance_view_fails_closed_on_mixed_or_unusable_samples() { - let planner = PlacementPlanner::default(); - let donor = owned(1, 4, 10); - let receiver = owned(2, 0, 10); - assert!(planner.fleet_balance(100_000, &[], 0).unwrap().is_none()); - assert!( - planner - .fleet_balance(100_000, &[donor, receiver], 100_000) - .unwrap() - .is_none() - ); - let mut stale = receiver; - stale.observed_at_ms = 100_000 - MAX_OBSERVATION_AGE_MS - 1; - assert!( - planner - .fleet_balance(100_000, &[donor, stale], 0) - .unwrap() - .is_none() - ); - let mut forged = receiver; - forged.authenticated = false; - assert!( - planner - .fleet_balance(100_000, &[donor, forged], 0) - .unwrap() - .is_none() - ); - assert!(planner.fleet_balance(100_000, &[donor, donor], 0).is_err()); - let mut draining = receiver; - draining.draining = true; - assert!( - planner - .fleet_balance(100_000, &[donor, draining], 0) - .unwrap() - .is_none() - ); - let mut shedding = receiver; - shedding.pressure = PlacementPressure::Shedding; - assert!( - planner - .fleet_balance(100_000, &[donor, shedding], 0) - .unwrap() - .is_none() - ); - let balanced = owned(3, 2, 10); - let equal = owned(4, 2, 10); - assert!( - planner - .fleet_balance(100_000, &[balanced, equal], 0) - .unwrap() - .is_none() - ); -} - -#[test] -fn balance_moves_an_idle_cell_without_headroom_gain() { - let planner = PlacementPlanner::default(); - let donor = owned(1, 3, 10); - let receiver = owned(2, 0, 10); - let candidate = demand(1, donor.session); - // Equal headroom ratios leave no material gain, so the transfer score - // gate refuses the move on its own. - assert!( - planner - .plan_transfers(100_000, &[donor, receiver], &[candidate], None) - .unwrap() - .is_empty() - ); - let balance = planner - .fleet_balance(100_000, &[donor, receiver], 0) - .unwrap() - .unwrap(); - let intents = planner - .plan_transfers(100_000, &[donor, receiver], &[candidate], Some(&balance)) - .unwrap(); - assert_eq!(intents.len(), 1); - assert_eq!(intents[0].destination, receiver.session); - assert_eq!(intents[0].cell, candidate.cell); -} - -#[test] -fn balance_donation_requires_the_elected_donor() { - let planner = PlacementPlanner::default(); - let dense = owned(1, 4, 10); - let first = owned(2, 0, 10); - let second = owned(3, 0, 10); - let balance = planner - .fleet_balance(100_000, &[dense, first, second], 0) - .unwrap() - .unwrap(); - assert_eq!(balance.donor, dense.session); - let candidate = demand(1, first.session); - assert!( - planner - .plan_transfers( - 100_000, - &[dense, first, second], - &[candidate], - Some(&balance), - ) - .unwrap() - .is_empty() - ); -} diff --git a/crates/crab-cell-runtime/src/fleet/pressure.rs b/crates/crab-cell-runtime/src/fleet/pressure.rs deleted file mode 100644 index 8c41b9ff6..000000000 --- a/crates/crab-cell-runtime/src/fleet/pressure.rs +++ /dev/null @@ -1,227 +0,0 @@ -//! Pressure classification with hysteresis. -use crate::{Error, Result}; - -const MAX_PERMILLE: u16 = 1_000; - -/// Node pressure state used by placement and paced drain policy. -#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] -pub enum PressureState { - /// No pressure: the node accepts new ownership. - Normal, - /// Soft pressure, or a stale sample: the node paces optional work. - Constrained, - /// Hard pressure on one dimension: the node sheds load. - Shedding, - /// Marker the classifier sets while a sample falls below the exit - /// threshold; it returns to [`PressureState::Normal`] inside the same - /// observation. - Recovering, - /// Critical pressure on both memory and disk: no new ownership. - Critical, -} - -/// Bounded measured utilization sample. Values are permille, not floats. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct PressureSample { - /// Logical time the sample was taken. - pub at_ms: i64, - /// Memory utilization in permille. - pub memory_used_permille: u16, - /// Scratch disk utilization in permille. - pub disk_used_permille: u16, - /// Job-slot utilization in permille. - pub jobs_used_permille: u16, - /// Whether the sample missed its freshness window. - pub stale: bool, -} - -impl PressureSample { - fn validate(self) -> Result<()> { - if self.at_ms < 0 - || self.memory_used_permille > MAX_PERMILLE - || self.disk_used_permille > MAX_PERMILLE - || self.jobs_used_permille > MAX_PERMILLE - { - return Err(Error::Control("pressure sample is invalid")); - } - Ok(()) - } - - fn elevated(self, threshold: u16) -> bool { - self.memory_used_permille >= threshold - || self.disk_used_permille >= threshold - || self.jobs_used_permille >= threshold - } - - fn recovered(self, threshold: u16) -> bool { - self.memory_used_permille <= threshold - && self.disk_used_permille <= threshold - && self.jobs_used_permille <= threshold - } -} - -/// Hysteretic classifier requiring a sustained sample before changing state. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct PressureClassifier { - enter_permille: u16, - exit_permille: u16, - dwell_ms: i64, - state: PressureState, - pending: Option<(PressureState, i64)>, - last_at_ms: Option, -} - -impl PressureClassifier { - /// Creates a classifier with separate enter/exit thresholds. - pub fn new(enter_permille: u16, exit_permille: u16, dwell_ms: i64) -> Result { - if enter_permille > MAX_PERMILLE || exit_permille >= enter_permille || dwell_ms <= 0 { - return Err(Error::Control("pressure thresholds are invalid")); - } - Ok(Self { - enter_permille, - exit_permille, - dwell_ms, - state: PressureState::Normal, - pending: None, - last_at_ms: None, - }) - } - - /// Returns the current hysteretic state. - #[must_use] - pub const fn state(self) -> PressureState { - self.state - } - - /// Advances the classifier; stale samples cannot declare recovery. - pub fn observe(&mut self, sample: PressureSample) -> Result { - sample.validate()?; - if self.last_at_ms.is_some_and(|last| sample.at_ms < last) { - return Err(Error::Control("pressure sample time regressed")); - } - self.last_at_ms = Some(sample.at_ms); - if sample.stale { - self.pending = None; - if self.state == PressureState::Normal { - self.state = PressureState::Constrained; - } - return Ok(self.state); - } - - let desired = if sample.elevated(self.enter_permille) { - if sample.memory_used_permille >= self.enter_permille - && sample.disk_used_permille >= self.enter_permille - { - PressureState::Critical - } else { - PressureState::Shedding - } - } else if sample.recovered(self.exit_permille) { - PressureState::Normal - } else { - PressureState::Constrained - }; - - if desired == self.state { - self.pending = None; - return Ok(self.state); - } - let Some((pending, started)) = self.pending else { - self.pending = Some((desired, sample.at_ms)); - return Ok(self.state); - }; - if pending != desired { - self.pending = Some((desired, sample.at_ms)); - return Ok(self.state); - } - if sample.at_ms.saturating_sub(started) >= self.dwell_ms { - self.state = if desired == PressureState::Normal { - PressureState::Recovering - } else { - desired - }; - self.pending = None; - if self.state == PressureState::Recovering { - self.state = PressureState::Normal; - } - } - Ok(self.state) - } -} - -/// One bounded movement operation admitted by a node-local drain controller. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum MovementKind { - /// Stop admitting work for one Cell. - Quiesce, - /// Release one Cell's ownership. - Release, - /// Admit one incoming Cell. - Receive, -} - -/// Synchronous concurrency/rate budget for pressure movement. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct MovementBudget { - limit: u32, - used: u32, - interval_ms: i64, - window_started_ms: i64, - completed_in_window: u32, -} - -impl MovementBudget { - /// Creates a budget of `limit` movements per `interval_ms` window. - pub fn new(limit: u32, interval_ms: i64) -> Result { - if limit == 0 || interval_ms <= 0 { - return Err(Error::Capacity("movement budget")); - } - Ok(Self { - limit, - used: 0, - interval_ms, - window_started_ms: 0, - completed_in_window: 0, - }) - } - - /// Reserves one movement, failing when the window is spent or time ran - /// backwards. - pub fn try_start(&mut self, now_ms: i64) -> Result { - if now_ms < self.window_started_ms { - return Err(Error::Control("movement time regressed")); - } - if now_ms.saturating_sub(self.window_started_ms) >= self.interval_ms { - self.window_started_ms = now_ms; - self.completed_in_window = 0; - } - if self.used >= self.limit || self.completed_in_window >= self.limit { - return Err(Error::Capacity("movement budget")); - } - self.used += 1; - Ok(MovementPermit { completed: false }) - } - - /// Finishes a reservation: frees its slot and counts it in the window. - pub fn complete(&mut self, permit: &mut MovementPermit) { - if permit.completed { - return; - } - permit.completed = true; - self.used = self.used.saturating_sub(1); - self.completed_in_window = self.completed_in_window.saturating_add(1); - } - - /// Returns the reservations that have not completed yet. - #[must_use] - pub const fn in_flight(self) -> u32 { - self.used - } -} - -/// Reservation returned by [`MovementBudget::try_start`] and released by -/// [`MovementBudget::complete`]. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct MovementPermit { - completed: bool, -} diff --git a/crates/crab-cell-runtime/src/fleet/resource.rs b/crates/crab-cell-runtime/src/fleet/resource.rs deleted file mode 100644 index 758b2c97a..000000000 --- a/crates/crab-cell-runtime/src/fleet/resource.rs +++ /dev/null @@ -1,804 +0,0 @@ -//! Node resource ledger: memory, disk, and job admission. -use std::sync::{Arc, Mutex, Weak}; - -use crate::{Error, Result}; - -pub(crate) const ACTIVE_CELL_NATIVE_BYTES: usize = 64 * 1024; -/// Persistent database, WAL, SHM and capture descriptors reserved per active Cell. -pub const ACTIVE_CELL_FILE_DESCRIPTORS: usize = 8; -// Conservatively cover the entire 8 MiB shared page cache plus 4 MiB for -// SQLite, fetch/decode buffers and view metadata. Do not assume another view -// or writer pays for the cache; measured sharing may reduce this charge later. -pub(crate) const READ_REPLICA_NATIVE_BYTES: usize = 12 * 1024 * 1024; -pub(crate) const READ_REPLICA_FILE_DESCRIPTORS: usize = 4; -pub(crate) const HYDRATION_JOB_CAPACITY: usize = 2; - -/// Bounded resources owned by one runtime admission token. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct ResourceCost { - active_cells: usize, - resident_bytes: usize, - file_descriptors: usize, - retained_bytes: usize, - disk_bytes: u64, - worker_jobs: usize, - primitive_jobs: usize, - hydration_jobs: usize, - io_slots: usize, - blocking_jobs: usize, - recovery_jobs: usize, - dirty_jobs: usize, - scratch_units: usize, -} - -impl ResourceCost { - /// Cost of one active Cell with no work in flight. - #[must_use] - pub const fn active_cell() -> Self { - Self { - active_cells: 1, - resident_bytes: ACTIVE_CELL_NATIVE_BYTES, - file_descriptors: ACTIVE_CELL_FILE_DESCRIPTORS, - ..Self::zero() - } - } - - /// Empty cost, used as the base for [`ResourceCost::active_cell`]. - #[must_use] - pub const fn zero() -> Self { - Self { - active_cells: 0, - resident_bytes: 0, - file_descriptors: 0, - retained_bytes: 0, - disk_bytes: 0, - worker_jobs: 0, - primitive_jobs: 0, - hydration_jobs: 0, - io_slots: 0, - blocking_jobs: 0, - recovery_jobs: 0, - dirty_jobs: 0, - scratch_units: 0, - } - } - - /// Returns the Cells this cost covers. - #[must_use] - pub const fn active_cells(self) -> usize { - self.active_cells - } - - /// Returns the resident memory in bytes. - #[must_use] - pub const fn resident_bytes(self) -> usize { - self.resident_bytes - } - - /// Returns the open file descriptors. - #[must_use] - pub const fn file_descriptors(self) -> usize { - self.file_descriptors - } - - /// Returns the bytes retained beyond the active set. - #[must_use] - pub const fn retained_bytes(self) -> usize { - self.retained_bytes - } - - /// Returns the scratch disk in bytes. - #[must_use] - pub const fn disk_bytes(self) -> u64 { - self.disk_bytes - } - - /// Returns the SQL worker jobs. - #[must_use] - pub const fn worker_jobs(self) -> usize { - self.worker_jobs - } - - /// Returns the primitive maintenance jobs. - #[must_use] - pub const fn primitive_jobs(self) -> usize { - self.primitive_jobs - } - - /// Returns the hydration jobs. - #[must_use] - pub const fn hydration_jobs(self) -> usize { - self.hydration_jobs - } - - /// Returns the object-store I/O slots. - #[must_use] - pub const fn io_slots(self) -> usize { - self.io_slots - } - - /// Returns the blocking activity jobs. - #[must_use] - pub const fn blocking_jobs(self) -> usize { - self.blocking_jobs - } - - /// Returns the recovery jobs. - #[must_use] - pub const fn recovery_jobs(self) -> usize { - self.recovery_jobs - } - - /// Returns the dirty, not-yet-published jobs. - #[must_use] - pub const fn dirty_jobs(self) -> usize { - self.dirty_jobs - } - - /// Returns the scratch units. - #[must_use] - pub const fn scratch_units(self) -> usize { - self.scratch_units - } - - /// Sets the retained bytes. - #[must_use] - pub const fn with_retained_bytes(mut self, bytes: usize) -> Self { - self.retained_bytes = bytes; - self - } - - /// Sets the resident bytes. - #[must_use] - pub const fn with_resident_bytes(mut self, bytes: usize) -> Self { - self.resident_bytes = bytes; - self - } - - /// Sets the open file descriptors. - #[must_use] - pub const fn with_file_descriptors(mut self, descriptors: usize) -> Self { - self.file_descriptors = descriptors; - self - } - - /// Sets the Cell count. - #[must_use] - pub const fn with_active_cells(mut self, cells: usize) -> Self { - self.active_cells = cells; - self - } - - /// Sets the scratch disk in bytes. - #[must_use] - pub const fn with_disk_bytes(mut self, bytes: u64) -> Self { - self.disk_bytes = bytes; - self - } - - /// Sets the SQL worker jobs. - #[must_use] - pub const fn with_worker_jobs(mut self, jobs: usize) -> Self { - self.worker_jobs = jobs; - self - } - - /// Sets the primitive maintenance jobs. - #[must_use] - pub const fn with_primitive_jobs(mut self, jobs: usize) -> Self { - self.primitive_jobs = jobs; - self - } - - /// Sets the hydration jobs. - #[must_use] - pub const fn with_hydration_jobs(mut self, jobs: usize) -> Self { - self.hydration_jobs = jobs; - self - } - - /// Sets the object-store I/O slots. - #[must_use] - pub const fn with_io_slots(mut self, slots: usize) -> Self { - self.io_slots = slots; - self - } - - /// Sets the blocking activity jobs. - #[must_use] - pub const fn with_blocking_jobs(mut self, jobs: usize) -> Self { - self.blocking_jobs = jobs; - self - } - - /// Sets the recovery jobs. - #[must_use] - pub const fn with_recovery_jobs(mut self, jobs: usize) -> Self { - self.recovery_jobs = jobs; - self - } - - /// Sets the dirty, not-yet-published jobs. - #[must_use] - pub const fn with_dirty_jobs(mut self, jobs: usize) -> Self { - self.dirty_jobs = jobs; - self - } - - /// Sets the scratch units. - #[must_use] - pub const fn with_scratch_units(mut self, units: usize) -> Self { - self.scratch_units = units; - self - } - - fn checked_add(self, other: Self) -> Option { - Some(Self { - active_cells: self.active_cells.checked_add(other.active_cells)?, - resident_bytes: self.resident_bytes.checked_add(other.resident_bytes)?, - file_descriptors: self.file_descriptors.checked_add(other.file_descriptors)?, - retained_bytes: self.retained_bytes.checked_add(other.retained_bytes)?, - disk_bytes: self.disk_bytes.checked_add(other.disk_bytes)?, - worker_jobs: self.worker_jobs.checked_add(other.worker_jobs)?, - primitive_jobs: self.primitive_jobs.checked_add(other.primitive_jobs)?, - hydration_jobs: self.hydration_jobs.checked_add(other.hydration_jobs)?, - io_slots: self.io_slots.checked_add(other.io_slots)?, - blocking_jobs: self.blocking_jobs.checked_add(other.blocking_jobs)?, - recovery_jobs: self.recovery_jobs.checked_add(other.recovery_jobs)?, - dirty_jobs: self.dirty_jobs.checked_add(other.dirty_jobs)?, - scratch_units: self.scratch_units.checked_add(other.scratch_units)?, - }) - } - - fn checked_sub(self, other: Self) -> Option { - Some(Self { - active_cells: self.active_cells.checked_sub(other.active_cells)?, - resident_bytes: self.resident_bytes.checked_sub(other.resident_bytes)?, - file_descriptors: self.file_descriptors.checked_sub(other.file_descriptors)?, - retained_bytes: self.retained_bytes.checked_sub(other.retained_bytes)?, - disk_bytes: self.disk_bytes.checked_sub(other.disk_bytes)?, - worker_jobs: self.worker_jobs.checked_sub(other.worker_jobs)?, - primitive_jobs: self.primitive_jobs.checked_sub(other.primitive_jobs)?, - hydration_jobs: self.hydration_jobs.checked_sub(other.hydration_jobs)?, - io_slots: self.io_slots.checked_sub(other.io_slots)?, - blocking_jobs: self.blocking_jobs.checked_sub(other.blocking_jobs)?, - recovery_jobs: self.recovery_jobs.checked_sub(other.recovery_jobs)?, - dirty_jobs: self.dirty_jobs.checked_sub(other.dirty_jobs)?, - scratch_units: self.scratch_units.checked_sub(other.scratch_units)?, - }) - } - - fn fits_within(self, limit: Self) -> bool { - self.active_cells <= limit.active_cells - && self.resident_bytes <= limit.resident_bytes - && self.file_descriptors <= limit.file_descriptors - && self.retained_bytes <= limit.retained_bytes - && self.disk_bytes <= limit.disk_bytes - && self.worker_jobs <= limit.worker_jobs - && self.primitive_jobs <= limit.primitive_jobs - && self.hydration_jobs <= limit.hydration_jobs - && self.io_slots <= limit.io_slots - && self.blocking_jobs <= limit.blocking_jobs - && self.recovery_jobs <= limit.recovery_jobs - && self.dirty_jobs <= limit.dirty_jobs - && self.scratch_units <= limit.scratch_units - } -} - -/// Point-in-time usage and limits for one resource ledger. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct ResourceSnapshot { - /// Cost currently reserved. - pub used: ResourceCost, - /// Cost the ledger admits. - pub limit: ResourceCost, -} - -#[derive(Clone)] -pub(crate) struct ResourceLedger { - // Reservations are short synchronous critical sections; no ledger lock is - // held while a worker, filesystem, or provider operation runs. - state: Arc, -} - -pub(crate) struct LedgerState { - snapshot: Mutex, - released: tokio::sync::Notify, -} - -impl ResourceLedger { - pub(crate) fn new(limit: ResourceCost) -> Self { - Self { - state: Arc::new(LedgerState { - snapshot: Mutex::new(ResourceSnapshot { - used: ResourceCost::zero(), - limit, - }), - released: tokio::sync::Notify::new(), - }), - } - } - - pub(crate) fn weak(&self) -> Weak { - Arc::downgrade(&self.state) - } - - pub(crate) fn try_reserve(&self, cost: ResourceCost) -> Result { - let mut state = self - .state - .snapshot - .lock() - .map_err(|_| Error::Control("resource ledger lock poisoned"))?; - let next = state - .used - .checked_add(cost) - .ok_or(Error::Capacity("resource ledger arithmetic"))?; - if !next.fits_within(state.limit) { - return Err(Error::Capacity("resource ledger")); - } - state.used = next; - Ok(ResourceReservation { - ledger: self.clone(), - cost, - }) - } - - pub(crate) async fn reserve(&self, cost: ResourceCost) -> Result { - if !cost.fits_within(self.snapshot()?.limit) { - return Err(Error::Capacity("resource ledger")); - } - loop { - // Register before checking capacity so a release between the - // failed attempt and await cannot leave the waiter asleep. - let released = self.state.released.notified(); - match self.try_reserve(cost) { - Ok(reservation) => return Ok(reservation), - Err(Error::Capacity(_)) => released.await, - Err(error) => return Err(error), - } - } - } - - pub(crate) fn snapshot(&self) -> Result { - self.state - .snapshot - .lock() - .map(|state| *state) - .map_err(|_| Error::Control("resource ledger lock poisoned")) - } - - pub(crate) fn set_retained_limit(&self, bytes: usize) -> Result<()> { - let mut state = self - .state - .snapshot - .lock() - .map_err(|_| Error::Control("resource ledger lock poisoned"))?; - if state.used.retained_bytes() > bytes { - return Err(Error::Capacity("resource ledger retained bytes")); - } - state.limit = state.limit.with_retained_bytes(bytes); - Ok(()) - } - - pub(crate) fn set_resident_limit(&self, bytes: usize) -> Result<()> { - let mut state = self - .state - .snapshot - .lock() - .map_err(|_| Error::Control("resource ledger lock poisoned"))?; - if bytes == 0 || state.used.resident_bytes() > bytes { - return Err(Error::Capacity("resource ledger resident bytes")); - } - state.limit = state.limit.with_resident_bytes(bytes); - Ok(()) - } - - pub(crate) fn set_disk_limit(&self, bytes: u64) -> Result<()> { - let mut state = self - .state - .snapshot - .lock() - .map_err(|_| Error::Control("resource ledger lock poisoned"))?; - if state.used.disk_bytes() > bytes { - return Err(Error::Capacity("resource ledger disk bytes")); - } - state.limit = state.limit.with_disk_bytes(bytes); - Ok(()) - } - - pub(crate) fn set_host_limits( - &self, - io_slots: usize, - blocking_jobs: usize, - recovery_jobs: usize, - dirty_jobs: usize, - scratch_units: usize, - ) -> Result<()> { - let mut state = self - .state - .snapshot - .lock() - .map_err(|_| Error::Control("resource ledger lock poisoned"))?; - let limit = state - .limit - .with_io_slots(io_slots) - .with_blocking_jobs(blocking_jobs) - .with_recovery_jobs(recovery_jobs) - .with_dirty_jobs(dirty_jobs) - .with_scratch_units(scratch_units); - if !state.used.fits_within(limit) { - return Err(Error::Capacity("resource ledger host limits")); - } - state.limit = limit; - Ok(()) - } - - pub(crate) fn reconcile_disk(&self, bytes: u64) -> Result<()> { - let mut state = self - .state - .snapshot - .lock() - .map_err(|_| Error::Control("resource ledger lock poisoned"))?; - let used = state.used.with_disk_bytes(bytes); - if !used.fits_within(state.limit) { - return Err(Error::Capacity("resource ledger")); - } - state.used = used; - Ok(()) - } - - fn release(&self, cost: ResourceCost) { - if let Ok(mut state) = self.state.snapshot.lock() - && let Some(used) = state.used.checked_sub(cost) - { - state.used = used; - } - } -} - -/// RAII reservation that returns its exact cost on every exit path. -#[must_use = "dropping the reservation releases its capacity"] -pub(crate) struct ResourceReservation { - ledger: ResourceLedger, - cost: ResourceCost, -} - -impl Drop for ResourceReservation { - fn drop(&mut self) { - self.ledger.release(self.cost); - self.ledger.state.released.notify_waiters(); - } -} - -/// Reports one host that outlived the runtime which owned its ledger. -/// -/// The LTX layer consults these admissions on every disk and slot operation, so -/// a stale host would otherwise log per call; one line names the runtime -/// session that must be matched against the deployment's shut-down sessions. -fn warn_dropped_ledger( - session: crate::identity::SessionId, - warned: &std::sync::atomic::AtomicBool, - operation: &'static str, -) { - if !warned.swap(true, std::sync::atomic::Ordering::Relaxed) { - tracing::warn!( - session = %crate::identity::encode_hex(session.as_bytes()), - operation, - "runtime ledger closed; a host outlived the Cell runtime that owned it" - ); - } -} - -pub(crate) struct LedgerDiskAdmission { - state: Weak, - session: crate::identity::SessionId, - warned: std::sync::atomic::AtomicBool, -} - -impl LedgerDiskAdmission { - /// Binds one disk admission to the ledger of the runtime that installs it. - pub(crate) fn new(session: crate::identity::SessionId, ledger: &ResourceLedger) -> Self { - Self { - state: ledger.weak(), - session, - warned: std::sync::atomic::AtomicBool::new(false), - } - } -} - -impl crab_ltx::DiskBudgetAdmission for LedgerDiskAdmission { - fn reconcile(&self, bytes: u64) -> crab_ltx::Result<()> { - let Some(state) = self.state.upgrade() else { - warn_dropped_ledger(self.session, &self.warned, "disk reconcile"); - return Err(crab_ltx::CrabError::InvalidState("runtime ledger closed")); - }; - ResourceLedger { state } - .reconcile_disk(bytes) - .map_err(|error| crab_ltx::CrabError::Other(Box::new(error)))?; - Ok(()) - } - - fn is_live(&self) -> bool { - self.state.strong_count() != 0 - } -} - -pub(crate) struct LedgerHostResourceAdmission { - state: Weak, - session: crate::identity::SessionId, - warned: std::sync::atomic::AtomicBool, -} - -impl LedgerHostResourceAdmission { - /// Binds one host admission to the ledger of the runtime that installs it. - pub(crate) fn new(session: crate::identity::SessionId, ledger: &ResourceLedger) -> Self { - Self { - state: ledger.weak(), - session, - warned: std::sync::atomic::AtomicBool::new(false), - } - } -} - -struct LedgerHostResourcePermit { - _reservation: ResourceReservation, -} - -impl crab_ltx::HostResourcePermit for LedgerHostResourcePermit {} - -impl crab_ltx::HostResourceAdmission for LedgerHostResourceAdmission { - fn reserve( - &self, - kind: crab_ltx::HostResourceKind, - units: u32, - ) -> crab_ltx::Result> { - let Some(state) = self.state.upgrade() else { - warn_dropped_ledger(self.session, &self.warned, "host resource reserve"); - return Err(crab_ltx::CrabError::InvalidState("runtime ledger closed")); - }; - let units = usize::try_from(units) - .map_err(|_| crab_ltx::CrabError::Limit(crab_ltx::LimitKind::HostResourceUnits))?; - let cost = match kind { - crab_ltx::HostResourceKind::Io => ResourceCost::zero().with_io_slots(units), - crab_ltx::HostResourceKind::BlockingJob => { - ResourceCost::zero().with_blocking_jobs(units) - } - crab_ltx::HostResourceKind::Recovery => ResourceCost::zero().with_recovery_jobs(units), - crab_ltx::HostResourceKind::Dirty => ResourceCost::zero().with_dirty_jobs(units), - crab_ltx::HostResourceKind::Scratch => ResourceCost::zero().with_scratch_units(units), - }; - let reservation = ResourceLedger { state } - .try_reserve(cost) - .map_err(|error| crab_ltx::CrabError::Other(Box::new(error)))?; - Ok(Box::new(LedgerHostResourcePermit { - _reservation: reservation, - })) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crab_ltx::HostResourceAdmission; - - #[tokio::test] - async fn waiting_reservation_reuses_released_capacity_without_overcommit() { - let cost = ResourceCost::zero().with_primitive_jobs(1); - let ledger = ResourceLedger::new(cost); - let held = ledger.try_reserve(cost).unwrap(); - let pending = ledger.reserve(cost); - tokio::pin!(pending); - assert!(futures_util::poll!(&mut pending).is_pending()); - assert_eq!(ledger.snapshot().unwrap().used, cost); - drop(held); - let acquired = tokio::time::timeout(std::time::Duration::from_secs(1), pending) - .await - .unwrap() - .unwrap(); - assert_eq!(ledger.snapshot().unwrap().used, cost); - drop(acquired); - assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); - } - - #[tokio::test] - async fn cancelled_reservation_does_not_retain_capacity() { - let cost = ResourceCost::zero().with_primitive_jobs(1); - let ledger = ResourceLedger::new(cost); - let held = ledger.try_reserve(cost).unwrap(); - { - let pending = ledger.reserve(cost); - tokio::pin!(pending); - assert!(futures_util::poll!(&mut pending).is_pending()); - } - drop(held); - let acquired = ledger.try_reserve(cost).unwrap(); - assert_eq!(ledger.snapshot().unwrap().used, cost); - drop(acquired); - } - - #[test] - fn reservations_are_bounded_and_return_to_baseline() { - let ledger = ResourceLedger::new( - ResourceCost::active_cell() - .with_resident_bytes(ACTIVE_CELL_NATIVE_BYTES) - .with_retained_bytes(64) - .with_worker_jobs(1) - .with_hydration_jobs(1), - ); - let reservation = ledger - .try_reserve(ResourceCost::active_cell().with_retained_bytes(32)) - .unwrap(); - assert!(ledger.try_reserve(ResourceCost::active_cell()).is_err()); - assert_eq!(ledger.snapshot().unwrap().used.active_cells(), 1); - let job = ledger - .try_reserve(ResourceCost::zero().with_worker_jobs(1)) - .unwrap(); - assert!( - ledger - .try_reserve(ResourceCost::zero().with_worker_jobs(1)) - .is_err() - ); - let hydration = ledger - .try_reserve(ResourceCost::zero().with_hydration_jobs(1)) - .unwrap(); - assert!( - ledger - .try_reserve(ResourceCost::zero().with_hydration_jobs(1)) - .is_err() - ); - drop(hydration); - drop(job); - drop(reservation); - assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); - } - - #[test] - fn checked_costs_reject_underflow_and_overflow() { - assert!( - ResourceCost::active_cell() - .checked_sub(ResourceCost::active_cell().with_worker_jobs(1)) - .is_none() - ); - assert!( - ResourceCost { - active_cells: usize::MAX, - ..ResourceCost::zero() - } - .checked_add(ResourceCost::active_cell()) - .is_none() - ); - } - - #[test] - fn active_cell_cost_tracks_file_descriptors() { - let cost = ResourceCost::active_cell(); - assert_eq!(cost.file_descriptors(), ACTIVE_CELL_FILE_DESCRIPTORS); - assert_eq!(cost.with_file_descriptors(0).file_descriptors(), 0); - } - - #[test] - fn concurrent_reservations_release_to_the_same_baseline() { - let ledger = ResourceLedger::new( - ResourceCost::active_cell() - .with_active_cells(4) - .with_resident_bytes(4 * ACTIVE_CELL_NATIVE_BYTES) - .with_file_descriptors(4 * ACTIVE_CELL_FILE_DESCRIPTORS), - ); - let barrier = std::sync::Arc::new(std::sync::Barrier::new(16)); - let successful = std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)); - std::thread::scope(|scope| { - for _ in 0..16 { - let ledger = ledger.clone(); - let barrier = std::sync::Arc::clone(&barrier); - let successful = std::sync::Arc::clone(&successful); - scope.spawn(move || { - let reservation = ledger.try_reserve(ResourceCost::active_cell()).ok(); - if reservation.is_some() { - successful.fetch_add(1, std::sync::atomic::Ordering::Relaxed); - } - barrier.wait(); - drop(reservation); - }); - } - }); - assert_eq!(successful.load(std::sync::atomic::Ordering::Relaxed), 4); - assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); - } - - #[test] - fn ltx_disk_reservations_share_the_runtime_ledger() { - let ledger = ResourceLedger::new(ResourceCost::zero().with_disk_bytes(10)); - let budget = crab_ltx::DiskBudget::new(20); - let existing = budget.try_reserve(1).unwrap(); - budget - .install_admission(Arc::new(LedgerDiskAdmission::new( - crate::identity::SessionId::from_bytes([7; 16]), - &ledger, - ))) - .unwrap(); - assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 1); - drop(existing); - assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 0); - - let reservation = budget.try_reserve(4).unwrap(); - assert_eq!(budget.used(), 4); - assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 4); - reservation.resize(7).unwrap(); - assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 7); - assert!(budget.try_reserve(4).is_err()); - assert_eq!(budget.used(), 7); - assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 7); - assert!(reservation.try_grow(4).is_err()); - assert_eq!(reservation.bytes(), 7); - assert!(reservation.resize(11).is_err()); - assert_eq!(reservation.bytes(), 7); - reservation.resize(2).unwrap(); - assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 2); - drop(reservation); - assert_eq!(budget.used(), 0); - assert_eq!(ledger.snapshot().unwrap().used.disk_bytes(), 0); - } - - #[test] - fn ltx_host_admissions_share_one_runtime_ledger() { - let ledger = ResourceLedger::new( - ResourceCost::zero() - .with_io_slots(1) - .with_blocking_jobs(1) - .with_recovery_jobs(1) - .with_dirty_jobs(1) - .with_scratch_units(2), - ); - let admission = LedgerHostResourceAdmission::new( - crate::identity::SessionId::from_bytes([8; 16]), - &ledger, - ); - let io = admission - .reserve(crab_ltx::HostResourceKind::Io, 1) - .unwrap(); - assert!( - admission - .reserve(crab_ltx::HostResourceKind::Io, 1) - .is_err() - ); - let scratch = admission - .reserve(crab_ltx::HostResourceKind::Scratch, 2) - .unwrap(); - assert!( - admission - .reserve(crab_ltx::HostResourceKind::Scratch, 1) - .is_err() - ); - assert_eq!(ledger.snapshot().unwrap().used.io_slots(), 1); - assert_eq!(ledger.snapshot().unwrap().used.scratch_units(), 2); - drop(scratch); - drop(io); - assert_eq!(ledger.snapshot().unwrap().used, ResourceCost::zero()); - } - - #[test] - fn admissions_fail_closed_once_their_runtime_ledger_is_gone() { - let session = crate::identity::SessionId::from_bytes([9; 16]); - let ledger = ResourceLedger::new(ResourceCost::zero().with_disk_bytes(10).with_io_slots(1)); - let disk = LedgerDiskAdmission::new(session, &ledger); - let host = LedgerHostResourceAdmission::new(session, &ledger); - drop(ledger); - - let closed = match crab_ltx::DiskBudgetAdmission::reconcile(&disk, 1) { - Ok(()) => panic!("a closed ledger reconciled disk bytes"), - Err(error) => error, - }; - assert!( - closed.to_string().contains("runtime ledger closed"), - "{closed}" - ); - let closed = match host.reserve(crab_ltx::HostResourceKind::Io, 1) { - Ok(_) => panic!("a closed ledger admitted host slots"), - Err(error) => error, - }; - assert!( - closed.to_string().contains("runtime ledger closed"), - "{closed}" - ); - } -} diff --git a/crates/crab-cell-runtime/src/fleet/scheduler.rs b/crates/crab-cell-runtime/src/fleet/scheduler.rs deleted file mode 100644 index 4e494158e..000000000 --- a/crates/crab-cell-runtime/src/fleet/scheduler.rs +++ /dev/null @@ -1,814 +0,0 @@ -//! Scheduler Tick: due timers, primitive maintenance classes, and dead letters. -use std::collections::{HashMap, HashSet, VecDeque}; - -use crab_ltx::rusqlite::Transaction; - -use crate::cell::catalog::{CatalogProof, CatalogShardScan, CellCatalog}; -use crate::control::authority::{CellAuthority, VersionedControl}; -use crate::identity::{CellTarget, SessionId}; -use crate::node::NodeAdvertisement; -use crate::primitives::blob::blob_cleanup_expired; -use crate::primitives::cron::CronTarget; -use crate::primitives::cron::cron_fire_due_bounded; -use crate::primitives::effects::EffectBatch; -use crate::primitives::effects::{ - effect_cleanup_terminal_bounded, effect_expire_ready_bounded, effect_reclaim_expired_bounded, - inbox_cleanup_expired_bounded, -}; -use crate::primitives::kv::kv_cleanup_expired_bounded; -use crate::primitives::queue::QueueDeadLetterTarget; -use crate::primitives::queue::{ - MAX_ATTEMPTS, QueueDeadLetterWriter, queue_cleanup_expired_bounded, queue_expire_ready_bounded, - queue_expire_ready_bounded_with_dead_letter, queue_reclaim_expired_bounded, - queue_reclaim_expired_bounded_with_dead_letter, -}; -use crate::primitives::workflow::WorkflowDefinition; -use crate::primitives::workflow::{ - workflow_cleanup_terminal_bounded, workflow_fail_one_expired_activity, - workflow_fire_one_due_timer, workflow_reclaim_expired_bounded, -}; -use crate::primitives::{blob, cron, kv, queue, workflow}; -use crate::{Error, Result}; - -const WORKFLOW_RETENTION_MS: i64 = 30 * 24 * 60 * 60 * 1000; -const MAX_TICK_ITEMS: usize = 128; -const CONTROL_BATCH: usize = 32; -/// Unconditional classes: expired requests, inbox, terminal effects, due -/// effects, and expired effect leases. -const BASE_MAINTENANCE_CLASSES: usize = 5; - -#[derive(Clone, Copy)] -struct ProgressObservation { - progress: u64, - changed_at_ms: i64, -} - -/// Tracks advertised scanner progress and excludes sessions that stop advancing. -#[derive(Default)] -pub struct SchedulerFleet { - observations: HashMap, -} - -impl SchedulerFleet { - /// Returns sessions eligible for rendezvous assignment at this observation. - pub fn eligible_sessions( - &mut self, - nodes: &[NodeAdvertisement], - now_ms: i64, - stale_after_ms: i64, - ) -> Result> { - if now_ms < 0 || stale_after_ms <= 0 { - return Err(Error::Control("scheduler liveness interval is invalid")); - } - let live = nodes - .iter() - .map(NodeAdvertisement::session) - .collect::>(); - self.observations - .retain(|session, _| live.contains(session)); - - let mut eligible = Vec::with_capacity(nodes.len()); - for node in nodes { - let observation = - self.observations - .entry(node.session()) - .or_insert(ProgressObservation { - progress: node.progress(), - changed_at_ms: now_ms, - }); - if node.progress() < observation.progress { - return Err(Error::Control("scheduler progress regressed")); - } - if node.progress() > observation.progress { - *observation = ProgressObservation { - progress: node.progress(), - changed_at_ms: now_ms, - }; - } - let capacity = node.capacity(); - if now_ms.saturating_sub(observation.changed_at_ms) < stale_after_ms - && capacity.free_memory_bytes != 0 - && capacity.free_disk_bytes != 0 - && capacity.job_credits != 0 - { - eligible.push(node.session()); - } - } - Ok(eligible) - } -} - -/// One due catalog entry and its exact observed control token. -pub struct DueCell { - catalog: CatalogProof, - control: VersionedControl, -} - -impl DueCell { - /// Returns the catalog proof the scan pinned for this Cell. - #[must_use] - pub const fn catalog(&self) -> &CatalogProof { - &self.catalog - } - - /// Returns the control record and token the scan observed. - #[must_use] - pub const fn control(&self) -> &VersionedControl { - &self.control - } -} - -/// Revision-pinned catalog scan with at most 32 control reads per step. -pub struct DueCellScan { - catalog: CatalogShardScan, - authority: CellAuthority, - pending: VecDeque, -} - -impl DueCellScan { - /// Pins one catalog shard head before any control records are inspected. - pub async fn new(catalog: &CellCatalog, authority: CellAuthority, shard: u8) -> Result { - Ok(Self { - catalog: catalog.scan_shard(shard).await?, - authority, - pending: VecDeque::new(), - }) - } - - /// Returns the catalog revision this scan is pinned to. - #[must_use] - pub const fn revision(&self) -> u64 { - self.catalog.revision() - } - - /// Advances by at most 32 entries; an empty page still represents progress. - pub async fn next_batch(&mut self, now_ms: i64) -> Result>> { - self.next_batch_bounded(now_ms, CONTROL_BATCH).await - } - - /// Advances by at most `limit` entries without discarding unvisited proofs. - pub async fn next_batch_bounded( - &mut self, - now_ms: i64, - limit: usize, - ) -> Result>> { - if now_ms < 0 { - return Err(Error::Command("negative scheduler scan time")); - } - if limit == 0 { - return Err(Error::Command("scheduler scan limit is zero")); - } - if self.pending.is_empty() { - let Some(page) = self.catalog.next_page().await? else { - return Ok(None); - }; - self.pending.extend(page.entries().iter().cloned()); - } - let count = self.pending.len().min(CONTROL_BATCH).min(limit); - let mut due = Vec::new(); - for _ in 0..count { - let proof = self - .pending - .pop_front() - .ok_or(Error::Catalog("scheduler scan queue underflow"))?; - let Some(control) = self.authority.load(proof.entry().cell()).await? else { - continue; - }; - if control.value().cell != proof.entry().cell() { - return Err(Error::Control("catalog control changed Cell")); - } - if control.value().root.is_some() - && control - .value() - .next_due_ms - .is_some_and(|deadline| deadline <= now_ms) - && control.value().state != crate::control::ControlState::Tombstoned - { - due.push(DueCell { - catalog: proof, - control, - }); - } - } - Ok(Some(due)) - } -} - -/// Selects one preferred live scanner for a catalog shard by rendezvous score. -pub fn preferred_scanner(shard: u8, nodes: &[SessionId]) -> Result> { - let mut unique = HashSet::with_capacity(nodes.len()); - let mut winner = None; - for node in nodes { - if !unique.insert(*node.as_bytes()) { - return Err(Error::Control("duplicate scheduler node session")); - } - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.scheduler-rendezvous.v1\0"); - hasher.update(&[shard]); - hasher.update(node.as_bytes()); - let score = *hasher.finalize().as_bytes(); - if winner - .as_ref() - .is_none_or(|(best, _): &([u8; 32], SessionId)| score > *best) - { - winner = Some((score, *node)); - } - } - Ok(winner.map(|(_, node)| node)) -} - -/// Bounded durable work performed by one serialized scheduler Tick. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct SchedulerTickOutcome { - /// Due maintenance items the tick advanced. - pub processed: u32, -} - -/// Rechecks and advances at most 128 due maintenance items in one transaction. -/// -/// This runs the same maintenance classes as the module Tick command, including -/// cron firing and queue dead-lettering: an embedding caller that omits the two -/// targets changes what the Tick does, so they stay explicit here instead of -/// defaulting to "no cron, no dead letter". -pub fn scheduler_tick( - transaction: &Transaction<'_>, - source: &CellTarget, - logical_time_ms: i64, - workflow_definitions: &[&'static dyn WorkflowDefinition], - queue_dead_letter: Option, - cron_targets: &[CronTarget], -) -> Result { - let command_sequence = transaction.query_row( - "SELECT commit_sequence + 1 FROM sys_meta WHERE singleton = 1 AND commit_sequence < 9223372036854775807", - [], - |row| row.get::<_, u64>(0), - )?; - scheduler_tick_at( - transaction, - source, - command_sequence, - logical_time_ms, - workflow_definitions, - queue_dead_letter, - cron_targets, - ) -} - -pub(crate) fn scheduler_tick_at( - transaction: &Transaction<'_>, - source: &CellTarget, - command_sequence: u64, - logical_time_ms: i64, - workflow_definitions: &[&'static dyn WorkflowDefinition], - queue_dead_letter: Option, - cron_targets: &[CronTarget], -) -> Result { - if logical_time_ms < 0 { - return Err(Error::Command("negative scheduler logical time")); - } - let tables = installed_tables(transaction)?; - let mut effects = EffectBatch::new(transaction, source, command_sequence, logical_time_ms)?; - let mut budget = MaintenanceBudget::new(reserved_classes(&tables))?; - - budget.run(|limit| { - transaction - .execute( - "DELETE FROM sys_requests WHERE request_id IN (SELECT request_id FROM sys_requests INDEXED BY sys_requests_expiry WHERE retain_until_ms <= ?1 ORDER BY retain_until_ms, request_id LIMIT ?2)", - (logical_time_ms, limit as i64), - ) - .map_err(Into::into) - })?; - budget.run(|limit| inbox_cleanup_expired_bounded(transaction, logical_time_ms, limit))?; - budget.run(|limit| effect_cleanup_terminal_bounded(transaction, logical_time_ms, limit))?; - budget.run(|limit| effect_expire_ready_bounded(transaction, logical_time_ms, limit))?; - budget.run(|limit| effect_reclaim_expired_bounded(transaction, logical_time_ms, limit))?; - - if tables.contains(kv::KV_TABLE) { - budget.run(|limit| kv_cleanup_expired_bounded(transaction, logical_time_ms, limit))?; - } - if tables.contains(blob::BLOB_TABLE) { - budget.run(|limit| blob_cleanup_expired(transaction, logical_time_ms, limit))?; - } - if tables.contains(cron::CRON_TABLE) { - budget.run(|limit| { - cron_fire_due_bounded( - transaction, - &mut effects, - source, - logical_time_ms, - cron_targets, - limit, - ) - })?; - } - if tables.contains(queue::QUEUE_TABLE) { - budget.run(|limit| queue_cleanup_expired_bounded(transaction, logical_time_ms, limit))?; - budget.run(|limit| { - if let Some(target) = queue_dead_letter { - let mut dead_letter = QueueDeadLetterWriter::new(target, &mut effects); - queue_expire_ready_bounded_with_dead_letter( - transaction, - logical_time_ms, - limit, - Some(&mut dead_letter), - ) - } else { - queue_expire_ready_bounded(transaction, logical_time_ms, limit) - } - })?; - budget.run(|limit| { - if let Some(target) = queue_dead_letter { - let mut dead_letter = QueueDeadLetterWriter::new(target, &mut effects); - queue_reclaim_expired_bounded_with_dead_letter( - transaction, - logical_time_ms, - limit, - Some(&mut dead_letter), - ) - } else { - queue_reclaim_expired_bounded(transaction, logical_time_ms, limit) - } - })?; - } - if tables.contains(workflow::WORKFLOW_TABLE) { - budget - .run(|limit| workflow_cleanup_terminal_bounded(transaction, logical_time_ms, limit))?; - budget.run(|limit| { - repeat(limit, || { - workflow_fail_one_expired_activity( - transaction, - &mut effects, - source, - logical_time_ms, - workflow_definitions, - ) - }) - })?; - budget - .run(|limit| workflow_reclaim_expired_bounded(transaction, logical_time_ms, limit))?; - budget.run(|limit| { - repeat(limit, || { - workflow_fire_one_due_timer( - transaction, - &mut effects, - source, - logical_time_ms, - workflow_definitions, - ) - }) - })?; - } - - // Cleanup classes run only once so rows terminalized above remain observable - // until the next Tick. Other classes can safely consume unused capacity. - budget.fill(|limit| { - transaction - .execute( - "DELETE FROM sys_requests WHERE request_id IN (SELECT request_id FROM sys_requests INDEXED BY sys_requests_expiry WHERE retain_until_ms <= ?1 ORDER BY retain_until_ms, request_id LIMIT ?2)", - (logical_time_ms, limit as i64), - ) - .map_err(Into::into) - })?; - budget.fill(|limit| inbox_cleanup_expired_bounded(transaction, logical_time_ms, limit))?; - budget.fill(|limit| effect_expire_ready_bounded(transaction, logical_time_ms, limit))?; - budget.fill(|limit| effect_reclaim_expired_bounded(transaction, logical_time_ms, limit))?; - if tables.contains(kv::KV_TABLE) { - budget.fill(|limit| kv_cleanup_expired_bounded(transaction, logical_time_ms, limit))?; - } - if tables.contains(blob::BLOB_TABLE) { - budget.fill(|limit| blob_cleanup_expired(transaction, logical_time_ms, limit))?; - } - if tables.contains(cron::CRON_TABLE) { - budget.fill(|limit| { - cron_fire_due_bounded( - transaction, - &mut effects, - source, - logical_time_ms, - cron_targets, - limit, - ) - })?; - } - if tables.contains(queue::QUEUE_TABLE) { - budget.fill(|limit| { - if let Some(target) = queue_dead_letter { - let mut dead_letter = QueueDeadLetterWriter::new(target, &mut effects); - queue_expire_ready_bounded_with_dead_letter( - transaction, - logical_time_ms, - limit, - Some(&mut dead_letter), - ) - } else { - queue_expire_ready_bounded(transaction, logical_time_ms, limit) - } - })?; - budget.fill(|limit| { - if let Some(target) = queue_dead_letter { - let mut dead_letter = QueueDeadLetterWriter::new(target, &mut effects); - queue_reclaim_expired_bounded_with_dead_letter( - transaction, - logical_time_ms, - limit, - Some(&mut dead_letter), - ) - } else { - queue_reclaim_expired_bounded(transaction, logical_time_ms, limit) - } - })?; - } - if tables.contains(workflow::WORKFLOW_TABLE) { - budget.fill(|limit| { - repeat(limit, || { - workflow_fail_one_expired_activity( - transaction, - &mut effects, - source, - logical_time_ms, - workflow_definitions, - ) - }) - })?; - budget - .fill(|limit| workflow_reclaim_expired_bounded(transaction, logical_time_ms, limit))?; - budget.fill(|limit| { - repeat(limit, || { - workflow_fire_one_due_timer( - transaction, - &mut effects, - source, - logical_time_ms, - workflow_definitions, - ) - }) - })?; - } - budget.finish()?; - Ok(SchedulerTickOutcome { - processed: u32::try_from(budget.processed()) - .map_err(|_| Error::Command("scheduler Tick count overflow"))?, - }) -} - -struct MaintenanceBudget { - remaining: usize, - remaining_classes: usize, - fair_share: usize, -} - -impl MaintenanceBudget { - fn new(classes: usize) -> Result { - if classes == 0 { - return Err(Error::Command("scheduler has no maintenance classes")); - } - Ok(Self { - remaining: MAX_TICK_ITEMS, - remaining_classes: classes, - fair_share: MAX_TICK_ITEMS / classes, - }) - } - - fn run(&mut self, operation: impl FnOnce(usize) -> Result) -> Result<()> { - self.remaining_classes = self - .remaining_classes - .checked_sub(1) - .ok_or(Error::Command("scheduler maintenance class mismatch"))?; - // Unused work flows forward, but every later class retains one share. - let reserved = self.fair_share * self.remaining_classes; - let limit = self.remaining.saturating_sub(reserved); - let processed = operation(limit)?; - if processed > limit { - return Err(Error::Command( - "scheduler maintenance class exceeded its budget", - )); - } - self.remaining = self - .remaining - .checked_sub(processed) - .ok_or(Error::Command("scheduler Tick exceeded 128 items"))?; - Ok(()) - } - - fn fill(&mut self, operation: impl FnOnce(usize) -> Result) -> Result<()> { - if self.remaining == 0 { - return Ok(()); - } - let limit = self.remaining; - let processed = operation(limit)?; - if processed > limit { - return Err(Error::Command( - "scheduler maintenance class exceeded its budget", - )); - } - self.remaining = self - .remaining - .checked_sub(processed) - .ok_or(Error::Command("scheduler Tick exceeded 128 items"))?; - Ok(()) - } - - fn processed(&self) -> usize { - MAX_TICK_ITEMS - self.remaining - } - - /// Confirms every reserved class ran. - /// - /// A section that stops running would otherwise keep its reserved share and - /// silently lower the Tick's usable work, hiding the missing maintenance. - fn finish(&self) -> Result<()> { - if self.remaining_classes != 0 { - return Err(Error::Command("scheduler maintenance class mismatch")); - } - Ok(()) - } -} - -fn repeat(limit: usize, mut operation: impl FnMut() -> Result) -> Result { - let mut processed = 0; - while processed < limit && operation()? { - processed += 1; - } - Ok(processed) -} - -/// Computes the earliest durable work or retention deadline after a command. -/// -/// Optional primitive schemas may be absent. Every returned deadline is clamped -/// to the command's logical time so stale work remains immediately discoverable. -pub fn scheduler_next_due_ms( - transaction: &Transaction<'_>, - logical_time_ms: i64, -) -> Result> { - if logical_time_ms < 0 { - return Err(Error::Command("negative scheduler logical time")); - } - let tables = installed_tables(transaction)?; - let mut next = None; - - include_minimum( - transaction, - "SELECT min(retain_until_ms) FROM sys_requests INDEXED BY sys_requests_expiry", - logical_time_ms, - &mut next, - )?; - include_minimum( - transaction, - "SELECT min(retain_until_ms) FROM sys_inbox INDEXED BY sys_inbox_expiry", - logical_time_ms, - &mut next, - )?; - include_minimum( - transaction, - "SELECT min(due_at_ms) FROM sys_effects INDEXED BY sys_effects_due WHERE state = 0", - logical_time_ms, - &mut next, - )?; - include_minimum( - transaction, - "SELECT min(lease_until_ms) FROM sys_effects INDEXED BY sys_effects_leases WHERE state = 1", - logical_time_ms, - &mut next, - )?; - include_minimum( - transaction, - "SELECT min(expires_at_ms) FROM sys_effects INDEXED BY sys_effects_expiry", - logical_time_ms, - &mut next, - )?; - - if tables.contains(kv::KV_TABLE) { - include_minimum( - transaction, - "SELECT min(expires_at_ms) FROM kv_entries INDEXED BY kv_expiry WHERE expires_at_ms IS NOT NULL", - logical_time_ms, - &mut next, - )?; - } - if tables.contains(queue::QUEUE_TABLE) { - let exhausted_ready: bool = transaction.query_row( - "SELECT EXISTS(SELECT 1 FROM queue_messages INDEXED BY queue_attempts WHERE state = 0 AND attempt >= ?1)", - [i64::from(MAX_ATTEMPTS)], - |row| row.get(0), - )?; - if exhausted_ready { - merge_due(logical_time_ms, logical_time_ms, &mut next)?; - } - include_minimum( - transaction, - "SELECT min(lease_until_ms) FROM queue_messages INDEXED BY queue_leases WHERE state = 1", - logical_time_ms, - &mut next, - )?; - include_minimum( - transaction, - "SELECT min(expires_at_ms) FROM queue_messages INDEXED BY queue_retention", - logical_time_ms, - &mut next, - )?; - include_minimum( - transaction, - "SELECT min(retain_until_ms) FROM queue_dedup INDEXED BY queue_dedup_expiry", - logical_time_ms, - &mut next, - )?; - } - if tables.contains(workflow::WORKFLOW_TABLE) { - include_minimum( - transaction, - "SELECT min(due_at_ms) FROM workflow_activities INDEXED BY activities_due WHERE state = 0", - logical_time_ms, - &mut next, - )?; - include_minimum( - transaction, - "SELECT min(lease_until_ms) FROM workflow_activities INDEXED BY activities_leases WHERE state = 1", - logical_time_ms, - &mut next, - )?; - include_minimum( - transaction, - "SELECT min(expires_at_ms) FROM workflow_activities INDEXED BY activities_expiry WHERE state IN (0, 1)", - logical_time_ms, - &mut next, - )?; - include_minimum( - transaction, - "SELECT min(due_at_ms) FROM workflow_timers INDEXED BY timers_due WHERE state = 0", - logical_time_ms, - &mut next, - )?; - let completed = minimum( - transaction, - "SELECT min(completed_at_ms) FROM workflow_runs INDEXED BY workflow_retention WHERE status BETWEEN 1 AND 3 AND completed_at_ms IS NOT NULL", - )?; - if let Some(completed_at_ms) = completed { - let retention = completed_at_ms - .checked_add(WORKFLOW_RETENTION_MS) - .ok_or(Error::Command("workflow retention deadline overflow"))?; - merge_due(retention, logical_time_ms, &mut next)?; - } - } - if tables.contains(blob::BLOB_TABLE) { - include_minimum( - transaction, - "SELECT min(expires_at_ms) FROM blob_uploads INDEXED BY blob_upload_expiry WHERE NOT EXISTS (SELECT 1 FROM blob_objects WHERE blob_objects.upload_id = blob_uploads.upload_id)", - logical_time_ms, - &mut next, - )?; - } - if tables.contains(cron::CRON_TABLE) { - include_minimum( - transaction, - "SELECT min(next_due_ms) FROM cron_schedules INDEXED BY cron_due WHERE enabled = 1", - logical_time_ms, - &mut next, - )?; - } - Ok(next) -} - -/// Optional primitive tables the scheduler maintains, with the Tick classes -/// each one consumes. -/// -/// The capability probe, the class budget, and the maintenance guards all read -/// this list, so a renamed table cannot silently drop a primitive's -/// maintenance. A class count must equal the number of `MaintenanceBudget::run` -/// calls its section performs, which `MaintenanceBudget::finish` enforces. -const PRIMITIVE_TABLES: [(&str, usize); 5] = [ - (kv::KV_TABLE, 1), - (queue::QUEUE_TABLE, 3), - (workflow::WORKFLOW_TABLE, 4), - (blob::BLOB_TABLE, 1), - (cron::CRON_TABLE, 1), -]; - -/// Probes the optional primitive tables the Cell schema currently installs. -fn installed_tables(transaction: &Transaction<'_>) -> Result> { - let mut statement = - transaction.prepare("SELECT name FROM sqlite_schema WHERE type = 'table'")?; - let mut rows = statement.query([])?; - let mut installed = HashSet::new(); - while let Some(row) = rows.next()? { - let name: String = row.get(0)?; - if let Some((table, _)) = PRIMITIVE_TABLES.iter().find(|(table, _)| *table == name) { - installed.insert(*table); - } - } - Ok(installed) -} - -/// Classes one Tick reserves for the installed primitives. -fn reserved_classes(tables: &HashSet<&'static str>) -> usize { - BASE_MAINTENANCE_CLASSES - + PRIMITIVE_TABLES - .iter() - .filter(|(table, _)| tables.contains(table)) - .map(|(_, classes)| classes) - .sum::() -} - -fn include_minimum( - transaction: &Transaction<'_>, - query: &'static str, - logical_time_ms: i64, - next: &mut Option, -) -> Result<()> { - if let Some(value) = minimum(transaction, query)? { - merge_due(value, logical_time_ms, next)?; - } - Ok(()) -} - -fn minimum(transaction: &Transaction<'_>, query: &'static str) -> Result> { - Ok(transaction.query_row(query, [], |row| row.get::<_, Option>(0))?) -} - -fn merge_due(value: i64, logical_time_ms: i64, next: &mut Option) -> Result<()> { - if value < 0 { - return Err(Error::Command("negative stored scheduler deadline")); - } - let value = value.max(logical_time_ms); - *next = Some(next.map_or(value, |current| current.min(value))); - Ok(()) -} - -#[cfg(test)] -mod tests { - use crab_ltx::rusqlite::Connection; - - use super::*; - - fn connection() -> Connection { - let connection = Connection::open_in_memory().unwrap(); - connection - .execute_batch(include_str!("../migrations/runtime.sql")) - .unwrap(); - connection - } - - const PRIMITIVE_SCHEMAS: [&str; 5] = [ - include_str!("../migrations/kv.sql"), - include_str!("../migrations/queue.sql"), - include_str!("../migrations/workflow.sql"), - include_str!("../migrations/blob.sql"), - include_str!("../migrations/cron.sql"), - ]; - - #[test] - fn declared_primitive_tables_are_installed_by_their_migrations() { - let mut seen = HashSet::new(); - for (table, _) in PRIMITIVE_TABLES { - let declaration = format!("CREATE TABLE {table}"); - assert!( - seen.insert(table), - "primitive table {table} is declared twice" - ); - assert!( - PRIMITIVE_SCHEMAS - .iter() - .any(|migration| migration.contains(&declaration)), - "no migration declares {declaration}" - ); - } - } - - #[test] - fn probe_tracks_every_installed_primitive_schema() { - let mut partial_connection = connection(); - partial_connection - .execute_batch(include_str!("../migrations/kv.sql")) - .unwrap(); - let transaction = partial_connection.transaction().unwrap(); - let partial = installed_tables(&transaction).unwrap(); - drop(transaction); - assert!(partial.contains(kv::KV_TABLE)); - assert!(!partial.contains(queue::QUEUE_TABLE)); - assert_eq!(reserved_classes(&partial), BASE_MAINTENANCE_CLASSES + 1); - - let mut complete_connection = connection(); - for schema in PRIMITIVE_SCHEMAS { - complete_connection - .execute_batch(schema) - .expect("primitive schema installs"); - } - let transaction = complete_connection.transaction().unwrap(); - let complete = installed_tables(&transaction).unwrap(); - assert_eq!(complete.len(), PRIMITIVE_TABLES.len()); - for (table, _) in PRIMITIVE_TABLES { - assert!(complete.contains(table), "{table} is not reported"); - } - assert_eq!(reserved_classes(&complete), BASE_MAINTENANCE_CLASSES + 10); - } - - #[test] - fn budget_finish_rejects_a_reserved_class_that_never_ran() { - let mut unused = MaintenanceBudget::new(2).unwrap(); - unused.run(|_| Ok(0)).unwrap(); - assert!(unused.finish().is_err()); - - let mut exhausted = MaintenanceBudget::new(1).unwrap(); - exhausted.run(|_| Ok(0)).unwrap(); - exhausted.finish().unwrap(); - } -} diff --git a/crates/crab-cell-runtime/src/fleet/telemetry.rs b/crates/crab-cell-runtime/src/fleet/telemetry.rs deleted file mode 100644 index 2f21a7945..000000000 --- a/crates/crab-cell-runtime/src/fleet/telemetry.rs +++ /dev/null @@ -1,449 +0,0 @@ -//! Bounded operational telemetry emitted by the runtime. -use std::{sync::Arc, time::Duration}; - -use crate::fleet::pressure::PressureState; -use crate::node::log::DurabilitySource; - -/// Evidence used for one returned durable command or effect outcome. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum CommandResponseSource { - /// An already durable outcome was replayed without another commit. - Recorded, - /// Follower durability proved the new commit. - Fleet, - /// Object publication proved the new commit. - Object, -} - -/// Outcome of an actor-owned resident route lookup. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ResidentRouteOutcome { - /// The route was resident and usable. - Hit, - /// No resident route exists for the Cell. - Miss, - /// A resident route exists but refused the request. - Refused, -} - -/// Outcome of one commit's attempt to use the node's enrolled follower lane. -/// -/// A commit that cannot use its lane still succeeds through object coverage, so -/// `Unavailable` and `Rejected` are the only signals that a node intended fleet -/// durability and silently fell back. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum DurabilitySubmissionOutcome { - /// An enrolled lane accepted the captured commit for shipping. - Fleet, - /// This host installs no node-log durability provider at all. - Unsupported, - /// A provider exists, but no lane is enrolled yet. - Unavailable, - /// The enrolled lane refused or fenced the submission. - Rejected, -} - -/// Kind of one registered primitive call observed at the execution boundary. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum PrimitiveOperationKind { - /// A registered command. - Command, - /// A registered query. - Query, -} - -/// Kind of one catalog object read. -/// -/// Catalog heads and immutable pages are read on the routing and due-scan hot -/// paths, and they bypass the LTX origin counters, so this bounded pair is the -/// only production signal for the metadata plane's object-store calls. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum CatalogReadKind { - /// One shard-head observation. - Head, - /// One immutable page observation. - Page, -} - -/// One phase of acquiring and activating a Cell on this node. -/// -/// A cold route pays ownership, root open, and local restore before the Cell -/// can answer; a warm route pays none of them. Timing the phases apart turns a -/// tail-latency report into a statement about which path to fix. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ActivationPhase { - /// The conditional ownership transition that claims the Cell. - Ownership, - /// Continuing a local database that already holds the observed root. - Resume, - /// Verifying the immutable root graph through the origin. - RootOpen, - /// Materializing the verified root into local disk. - Restore, - /// Opening the local database and publishing the serving control. - Activate, -} - -/// Terminal outcome of one registered primitive call. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum PrimitiveOperationOutcome { - /// The handler committed a result the caller consumes as success. - Success, - /// The handler committed a rejection the caller consumes as the result. - Rejected, - /// The operation failed before a committed result existed. - Failed, -} - -impl From<&crate::Result> for PrimitiveOperationOutcome { - fn from(result: &crate::Result) -> Self { - match result { - Ok(crate::cell::executor::HandlerOutcome::Success(_)) => Self::Success, - Ok(crate::cell::executor::HandlerOutcome::Rejected(_)) => Self::Rejected, - Err(_) => Self::Failed, - } - } -} - -/// Bounded operational events emitted by the Cell durability runtime. -/// -/// Implementations must keep labels finite and must not block the Cell actor. -pub trait CellTelemetry: Send + Sync { - /// Records one registered primitive call by owning module and outcome. - fn primitive_operation( - &self, - _module: &'static str, - _kind: PrimitiveOperationKind, - _outcome: PrimitiveOperationOutcome, - _elapsed: Duration, - ) { - } - - /// Records one completed fleet or object durability proof. - fn durability_proof(&self, _source: DurabilitySource, _waited: Duration) {} - - /// Records one durable command or effect outcome sent to its runtime caller. - /// - /// Elapsed time starts at admitted enqueue; confirmation is the final SQL - /// worker wait after proof, or zero for a recorded result. Transport, queries, - /// migrations, failed results, and abandoned receivers are excluded. - fn command_response( - &self, - _source: CommandResponseSource, - _elapsed: Duration, - _confirmation: Duration, - ) { - } - - /// Records how one commit's node-log submission resolved. - fn durability_submission(&self, _outcome: DurabilitySubmissionOutcome) {} - - /// Records the immutable objects and bytes one preparation attempt uploaded. - /// - /// The count covers every object a Cell root needs — segment bodies, - /// indexes, directory nodes, root documents, segment pages, bundle bodies, - /// and compaction outputs — so an operator can size object-store cost per - /// command instead of inferring it from the database size. Failed attempts - /// record the objects they did upload. - fn publication_cost(&self, _objects: u64, _bytes: u64) {} - - /// Records bytes sent to follower append lanes and whether every lane acknowledged them. - fn node_log_append(&self, _acknowledged: bool, _bytes: u64) {} - - /// Records a bounded-cardinality resident route result. - fn resident_route(&self, _outcome: ResidentRouteOutcome) {} - - /// Records one finite LTX phase outcome. - fn ltx_phase(&self, _phase: crab_ltx::LtxPhase, _elapsed: Duration, _succeeded: bool) {} - - /// Records one catalog object read and its outcome. - fn catalog_read(&self, _kind: CatalogReadKind, _elapsed: Duration, _succeeded: bool) {} - - /// Records one control-record read and its outcome. - /// - /// One control record is read per cold route, per due-scan cell, and per - /// recovery step, so this bounded counter is what makes the metadata - /// plane's dominant cost visible. - fn control_read(&self, _elapsed: Duration, _succeeded: bool) {} - - /// Records the duration of one Cell activation phase. - fn activation_phase(&self, _phase: ActivationPhase, _elapsed: Duration) {} - - /// Records one logical read attributed to a bounded residency class. - fn ltx_logical_read(&self, _origin: crab_ltx::LtxReadOrigin) {} - - /// Records one provider attempt and the bytes returned before its outcome. - fn ltx_origin_request( - &self, - _origin: crab_ltx::LtxReadOrigin, - _outcome: crab_ltx::LtxRequestOutcome, - _bytes: u64, - ) { - } - - /// Aggregates one fixed-size capture ledger without dynamic labels. - fn ltx_capture(&self, _timing: &crab_ltx::CaptureTiming, _succeeded: bool) {} - - /// Records the node's current hysteretic pressure tier. - /// - /// One node reports one tier at a time, and the classifier only hands out - /// `Normal`, `Constrained`, `Shedding`, or `Critical`, so a sink can render - /// this as a bounded gauge family instead of a growing label set. The tier a - /// node reports is the one that decides whether it sheds settled Cells. - fn pressure_state(&self, _state: PressureState) {} -} - -/// Shared late-bound telemetry sink used by runtime components. -#[derive(Clone, Default)] -pub struct CellTelemetryHandle { - inner: Arc>>, -} - -impl CellTelemetryHandle { - pub(crate) fn install(&self, telemetry: Arc) -> crate::Result<()> { - self.inner - .set(telemetry) - .map_err(|_| crate::Error::Control("Cell telemetry was initialized twice")) - } - - /// Creates a handle bound to one sink. - /// - /// Runtime components that do not install through `CellRuntime` — a - /// standalone catalog, for example — use this to share the node's sink. - #[must_use] - pub fn from_sink(telemetry: Arc) -> Self { - let handle = Self::default(); - // The lock is empty at construction, so this set cannot lose a race. - let _ = handle.inner.set(telemetry); - handle - } - - pub(crate) fn durability_proof(&self, source: DurabilitySource, waited: Duration) { - if let Some(telemetry) = self.inner.get() { - telemetry.durability_proof(source, waited); - } - } - - pub(crate) fn command_response( - &self, - source: CommandResponseSource, - elapsed: Duration, - confirmation: Duration, - ) { - if let Some(telemetry) = self.inner.get() { - telemetry.command_response(source, elapsed, confirmation); - } - } - - pub(crate) fn catalog_read(&self, kind: CatalogReadKind, elapsed: Duration, succeeded: bool) { - if let Some(telemetry) = self.inner.get() { - telemetry.catalog_read(kind, elapsed, succeeded); - } - } - - pub(crate) fn control_read(&self, elapsed: Duration, succeeded: bool) { - if let Some(telemetry) = self.inner.get() { - telemetry.control_read(elapsed, succeeded); - } - } - - pub(crate) fn activation_phase(&self, phase: ActivationPhase, elapsed: Duration) { - if let Some(telemetry) = self.inner.get() { - telemetry.activation_phase(phase, elapsed); - } - } - - pub(crate) fn primitive_operation( - &self, - module: &'static str, - kind: PrimitiveOperationKind, - outcome: PrimitiveOperationOutcome, - elapsed: Duration, - ) { - if let Some(telemetry) = self.inner.get() { - telemetry.primitive_operation(module, kind, outcome, elapsed); - } - } - - pub(crate) fn durability_submission(&self, outcome: DurabilitySubmissionOutcome) { - if let Some(telemetry) = self.inner.get() { - telemetry.durability_submission(outcome); - } - } - - pub(crate) fn publication_cost(&self, objects: u64, bytes: u64) { - if let Some(telemetry) = self.inner.get() { - telemetry.publication_cost(objects, bytes); - } - } - - pub(crate) fn node_log_append(&self, acknowledged: bool, bytes: u64) { - if let Some(telemetry) = self.inner.get() { - telemetry.node_log_append(acknowledged, bytes); - } - } - - pub(crate) fn pressure_state(&self, state: PressureState) { - if let Some(telemetry) = self.inner.get() { - telemetry.pressure_state(state); - } - } - - pub(crate) fn resident_route(&self, outcome: ResidentRouteOutcome) { - if let Some(telemetry) = self.inner.get() { - telemetry.resident_route(outcome); - if outcome == ResidentRouteOutcome::Hit { - telemetry.ltx_logical_read(crab_ltx::LtxReadOrigin::Resident); - } - } - } - - fn ltx_phase(&self, phase: crab_ltx::LtxPhase, elapsed: Duration, succeeded: bool) { - if let Some(telemetry) = self.inner.get() { - telemetry.ltx_phase(phase, elapsed, succeeded); - } - } - - fn ltx_logical_read(&self, origin: crab_ltx::LtxReadOrigin) { - if let Some(telemetry) = self.inner.get() { - telemetry.ltx_logical_read(origin); - } - } - - fn ltx_origin_request( - &self, - origin: crab_ltx::LtxReadOrigin, - outcome: crab_ltx::LtxRequestOutcome, - bytes: u64, - ) { - if let Some(telemetry) = self.inner.get() { - telemetry.ltx_origin_request(origin, outcome, bytes); - } - } - - fn ltx_capture(&self, timing: &crab_ltx::CaptureTiming, succeeded: bool) { - if let Some(telemetry) = self.inner.get() { - telemetry.ltx_capture(timing, succeeded); - } - } -} - -impl crab_ltx::LtxTelemetry for CellTelemetryHandle { - fn phase(&self, phase: crab_ltx::LtxPhase, elapsed: Duration, succeeded: bool) { - self.ltx_phase(phase, elapsed, succeeded); - } - - fn logical_read(&self, origin: crab_ltx::LtxReadOrigin) { - self.ltx_logical_read(origin); - } - - fn origin_request( - &self, - origin: crab_ltx::LtxReadOrigin, - outcome: crab_ltx::LtxRequestOutcome, - bytes: u64, - ) { - self.ltx_origin_request(origin, outcome, bytes); - } - - fn capture(&self, timing: &crab_ltx::CaptureTiming, succeeded: bool) { - tracing::debug!( - target: "crab_cell_runtime::action", - event = "cell_capture_completed", - capture_ns = timing.total_nanos, - encode_ns = timing.encode_nanos, - write_ns = timing.local_write_nanos, - fsync_ns = timing.fsync_nanos, - parent_sync_ns = timing.parent_sync_nanos, - checkpoint_ns = timing.checkpoint_nanos, - wal_read_bytes = timing.wal_read_bytes, - ltx_bytes = timing.ltx_bytes, - succeeded, - ); - self.ltx_capture(timing, succeeded); - } -} - -#[cfg(test)] -mod tests { - use super::*; - use std::sync::Mutex; - - #[derive(Default)] - struct RecordingTelemetry { - phases: Mutex>, - logical_reads: Mutex>, - requests: Mutex>, - submissions: Mutex>, - } - - impl CellTelemetry for RecordingTelemetry { - fn durability_submission(&self, outcome: DurabilitySubmissionOutcome) { - self.submissions.lock().unwrap().push(outcome); - } - - fn ltx_phase(&self, phase: crab_ltx::LtxPhase, _: Duration, succeeded: bool) { - self.phases.lock().unwrap().push((phase, succeeded)); - } - - fn ltx_logical_read(&self, origin: crab_ltx::LtxReadOrigin) { - self.logical_reads.lock().unwrap().push(origin); - } - - fn ltx_origin_request( - &self, - origin: crab_ltx::LtxReadOrigin, - outcome: crab_ltx::LtxRequestOutcome, - bytes: u64, - ) { - self.requests.lock().unwrap().push((origin, outcome, bytes)); - } - } - - #[test] - fn ltx_bridge_preserves_only_finite_runtime_dimensions() { - let handle = CellTelemetryHandle::default(); - let recording = Arc::new(RecordingTelemetry::default()); - handle.install(recording.clone()).unwrap(); - - crab_ltx::LtxTelemetry::phase( - &handle, - crab_ltx::LtxPhase::Directory, - Duration::from_millis(2), - true, - ); - crab_ltx::LtxTelemetry::origin_request( - &handle, - crab_ltx::LtxReadOrigin::Hydrating, - crab_ltx::LtxRequestOutcome::Failed, - 4_096, - ); - handle.resident_route(ResidentRouteOutcome::Hit); - handle.durability_submission(DurabilitySubmissionOutcome::Unavailable); - handle.durability_submission(DurabilitySubmissionOutcome::Fleet); - - assert_eq!( - *recording.phases.lock().unwrap(), - vec![(crab_ltx::LtxPhase::Directory, true)] - ); - assert_eq!( - *recording.logical_reads.lock().unwrap(), - vec![crab_ltx::LtxReadOrigin::Resident] - ); - assert_eq!( - *recording.requests.lock().unwrap(), - vec![( - crab_ltx::LtxReadOrigin::Hydrating, - crab_ltx::LtxRequestOutcome::Failed, - 4_096, - )] - ); - assert_eq!( - *recording.submissions.lock().unwrap(), - vec![ - DurabilitySubmissionOutcome::Unavailable, - DurabilitySubmissionOutcome::Fleet, - ] - ); - } -} diff --git a/crates/crab-cell-runtime/src/follower.rs b/crates/crab-cell-runtime/src/follower.rs deleted file mode 100644 index 00bdcdf40..000000000 --- a/crates/crab-cell-runtime/src/follower.rs +++ /dev/null @@ -1,473 +0,0 @@ -//! Follower lanes: per-leader record streams that back fleet durability proofs. -use std::collections::{BTreeMap, HashMap, btree_map::Entry}; -use std::io::{Read as _, Seek as _, SeekFrom, Write as _}; -use std::path::{Path, PathBuf}; -#[cfg(test)] -use std::sync::atomic::{AtomicUsize, Ordering}; -use std::sync::{Arc, Mutex}; - -use bytes::Bytes; - -use crate::identity::SessionId; -use crate::{Error, Result}; - -mod directory; -mod records; - -use directory::*; -use records::*; - -const RECORD_MAGIC: &[u8; 4] = b"CFR1"; -const RECORD_HEADER_BYTES: usize = 52; -const ROTATE_BYTES: u64 = 64 << 20; -const MAX_APPEND_FRAMES: usize = 64; -const MAX_TAIL_PAGE_BYTES: usize = 1 << 20; -const MAX_TAIL_PAGE_FRAMES: usize = 4096; -const MAX_RETIRED_LANES: usize = 1_024; -const FOLLOWER_QUARANTINE: &str = "followers-quarantine"; -const INDEX_BYTES_PER_RECORD: u64 = 128; -const MAX_FOLLOWER_INDEX_BYTES: u64 = 256 << 20; - -#[cfg(test)] -type ScanCounter = Arc; -#[cfg(not(test))] -#[derive(Clone, Copy)] -struct ScanCounter; - -#[cfg(test)] -fn new_scan_counter() -> ScanCounter { - Arc::new(AtomicUsize::new(0)) -} - -#[cfg(not(test))] -const fn new_scan_counter() -> ScanCounter { - ScanCounter -} - -#[cfg(test)] -fn clone_scan_counter(counter: &ScanCounter) -> ScanCounter { - Arc::clone(counter) -} - -#[cfg(not(test))] -const fn clone_scan_counter(counter: &ScanCounter) -> ScanCounter { - *counter -} - -#[cfg(test)] -fn count_scan(counter: &ScanCounter) { - counter.fetch_add(1, Ordering::Relaxed); -} - -#[cfg(not(test))] -const fn count_scan(_: &ScanCounter) {} - -#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] -struct Lane { - leader: SessionId, - epoch: u64, -} - -type LaneState = Arc>>; -type LaneMap = Arc>>; - -/// Durable contiguous range retained by one follower lane. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct FollowerReceipt { - /// First sequence the lane retained before the append. - pub base_sequence: u64, - /// Highest sequence the lane has made durable. - pub durable_through: u64, -} - -/// One bounded page from a sealed follower lane. -pub struct FollowerTailPage { - /// Frames in sequence order. - pub frames: Vec, - /// Sequence to continue from when the page filled its bound. - pub next_sequence: Option, -} - -/// Exact retired follower lane eligible for authority-checked collection. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct RetiredFollowerLane { - leader: SessionId, - epoch: u64, - covered_through: u64, - retired_at_ms: i64, -} - -impl RetiredFollowerLane { - /// Returns the leader whose lane was retired. - #[must_use] - pub const fn leader(&self) -> SessionId { - self.leader - } - - /// Returns the node-log epoch the lane belonged to. - #[must_use] - pub const fn epoch(&self) -> u64 { - self.epoch - } - - /// Returns the highest sequence object storage covered. - #[must_use] - pub const fn covered_through(&self) -> u64 { - self.covered_through - } - - /// Returns the logical time the lane was retired. - #[must_use] - pub const fn retired_at_ms(&self) -> i64 { - self.retired_at_ms - } -} - -/// Local SSD store for checksum-verified follower fragments. -/// -/// The bytes are durability obligations until the leader's contiguous object -/// watermark permits whole-chunk deletion. They are never cache-evicted. -#[derive(Clone)] -pub struct FollowerStore { - root: PathBuf, - limits: crab_ltx::Limits, - lanes: LaneMap, - disk: crab_ltx::DiskBudget, - retained: Arc>, - index_used: Arc>, - quarantined_entries: usize, - scan_counter: ScanCounter, -} - -impl FollowerStore { - /// Opens the `followers` namespace beneath a durable node data directory. - pub fn open( - root: PathBuf, - limits: crab_ltx::Limits, - disk: crab_ltx::DiskBudget, - ) -> Result { - let existed = root.exists(); - std::fs::create_dir_all(&root).map_err(crab_ltx::CrabError::from)?; - if !existed { - let parent = root - .parent() - .ok_or(Error::Node("follower root has no parent"))?; - sync_directory(parent).map_err(crab_ltx::CrabError::from)?; - } - sync_directory(&root).map_err(crab_ltx::CrabError::from)?; - scrub_followers(&root, limits)?; - let retained = disk.try_reserve(follower_bytes(&root)?)?; - let quarantined_entries = quarantine_entry_count(&root)?; - Ok(Self { - root, - limits, - lanes: Arc::new(Mutex::new(HashMap::new())), - disk, - retained: Arc::new(Mutex::new(retained)), - index_used: Arc::new(Mutex::new(0)), - quarantined_entries, - scan_counter: new_scan_counter(), - }) - } - - #[cfg(test)] - pub(crate) fn scan_count(&self) -> usize { - self.scan_counter.load(Ordering::Relaxed) - } - - /// Returns the bytes the store currently retains. - #[must_use] - pub fn retained_bytes(&self) -> u64 { - match self.retained.lock() { - Ok(retained) => retained.bytes(), - Err(poisoned) => poisoned.into_inner().bytes(), - } - } - - /// Returns the bytes the store may still write. - #[must_use] - pub fn available_bytes(&self) -> u64 { - self.disk.available() - } - - /// Reports diagnostic entries isolated by this or an earlier startup scrub. - #[must_use] - pub const fn quarantined_entries(&self) -> usize { - self.quarantined_entries - } - - /// Appends one ordered batch and acknowledges only after `sync_data`. - pub async fn append( - &self, - leader: SessionId, - epoch: u64, - frames: Vec, - covered_through: u64, - ) -> Result { - let encoded_bytes = frames - .iter() - .try_fold(0_u64, |total, frame| total.checked_add(frame.len() as u64)); - if frames.is_empty() - || frames.len() > MAX_APPEND_FRAMES - || encoded_bytes - .is_none_or(|bytes| bytes > self.limits.max_capture_bytes.saturating_add(64 * 240)) - { - return Err(Error::Node("invalid follower append batch")); - } - let lane = Lane { leader, epoch }; - let lock = self.lane_lock(lane)?; - let root = self.root.clone(); - let limits = self.limits; - let retained = Arc::clone(&self.retained); - let index_used = Arc::clone(&self.index_used); - let scan_counter = clone_scan_counter(&self.scan_counter); - let growth = encoded_bytes - .and_then(|bytes| bytes.checked_add((frames.len() * RECORD_HEADER_BYTES) as u64)) - .ok_or(Error::Node("follower append byte count overflow"))?; - tokio::task::spawn_blocking(move || { - let retained = retained - .lock() - .map_err(|_| Error::Node("follower disk reservation lock poisoned"))?; - retained.try_grow(growth)?; - let mut state = lock - .lock() - .map_err(|_| Error::Node("follower lane lock poisoned"))?; - let result = append_sync( - &root, - lane, - frames, - covered_through, - limits, - &index_used, - &mut state, - &scan_counter, - ); - let resize = - follower_bytes(&root).and_then(|bytes| retained.resize(bytes).map_err(Error::from)); - if result.is_err() { - *state = None; - } - settle_disk_reservation(result, resize) - }) - .await - .map_err(Error::FollowerWorkerJoin)? - } - - /// Seals a lane against future appends and returns its retained range. - pub async fn seal(&self, leader: SessionId, epoch: u64) -> Result { - let lane = Lane { leader, epoch }; - let lock = self.lane_lock(lane)?; - let root = self.root.clone(); - let limits = self.limits; - let retained = Arc::clone(&self.retained); - let index_used = Arc::clone(&self.index_used); - let scan_counter = clone_scan_counter(&self.scan_counter); - tokio::task::spawn_blocking(move || { - let retained = retained - .lock() - .map_err(|_| Error::Node("follower disk reservation lock poisoned"))?; - let mut state = lock - .lock() - .map_err(|_| Error::Node("follower lane lock poisoned"))?; - let directory = lane_directory(&root, lane); - if !directory.join("sealed").exists() && !directory.join("retired").exists() { - retained.try_grow(8)?; - } - let result = seal_sync(&root, lane, limits, &index_used, &mut state, &scan_counter); - let resize = - follower_bytes(&root).and_then(|bytes| retained.resize(bytes).map_err(Error::from)); - if result.is_err() { - *state = None; - } - settle_disk_reservation(result, resize) - }) - .await - .map_err(Error::FollowerWorkerJoin)? - } - - /// Retires one fully object-covered lane and keeps a durable append fence. - pub async fn retire( - &self, - leader: SessionId, - epoch: u64, - covered_through: u64, - ) -> Result { - let lane = Lane { leader, epoch }; - let lock = self.lane_lock(lane)?; - let root = self.root.clone(); - let limits = self.limits; - let retained = Arc::clone(&self.retained); - let scan_counter = clone_scan_counter(&self.scan_counter); - tokio::task::spawn_blocking(move || { - let retained = retained - .lock() - .map_err(|_| Error::Node("follower disk reservation lock poisoned"))?; - let mut state = lock - .lock() - .map_err(|_| Error::Node("follower lane lock poisoned"))?; - if !lane_directory(&root, lane).join("retired").exists() { - retained.try_grow(8)?; - } - let result = retire_sync(&root, lane, covered_through, limits, &scan_counter); - let resize = - follower_bytes(&root).and_then(|bytes| retained.resize(bytes).map_err(Error::from)); - *state = None; - settle_disk_reservation(result, resize) - }) - .await - .map_err(Error::FollowerWorkerJoin)? - } - - /// Reads a sealed, verified tail in node-sequence order. - pub async fn read_tail( - &self, - leader: SessionId, - epoch: u64, - first_sequence: u64, - ) -> Result> { - let lane = Lane { leader, epoch }; - let lock = self.lane_lock(lane)?; - let root = self.root.clone(); - let limits = self.limits; - let index_used = Arc::clone(&self.index_used); - let scan_counter = clone_scan_counter(&self.scan_counter); - tokio::task::spawn_blocking(move || { - let mut state = lock - .lock() - .map_err(|_| Error::Node("follower lane lock poisoned"))?; - let result = read_tail_sync( - &root, - lane, - first_sequence, - limits, - &index_used, - &mut state, - usize::MAX, - usize::MAX, - &scan_counter, - ) - .map(|page| page.frames); - if result.is_err() { - *state = None; - } - result - }) - .await - .map_err(Error::FollowerWorkerJoin)? - } - - /// Reads one network-sized page from a sealed, verified tail. - /// - /// A single frame may exceed the page target and is returned alone because - /// node frames are the independently checksummed transport unit. - pub async fn read_tail_page( - &self, - leader: SessionId, - epoch: u64, - first_sequence: u64, - ) -> Result { - let lane = Lane { leader, epoch }; - let lock = self.lane_lock(lane)?; - let root = self.root.clone(); - let limits = self.limits; - let index_used = Arc::clone(&self.index_used); - let scan_counter = clone_scan_counter(&self.scan_counter); - tokio::task::spawn_blocking(move || { - let mut state = lock - .lock() - .map_err(|_| Error::Node("follower lane lock poisoned"))?; - let result = read_tail_sync( - &root, - lane, - first_sequence, - limits, - &index_used, - &mut state, - MAX_TAIL_PAGE_BYTES, - MAX_TAIL_PAGE_FRAMES, - &scan_counter, - ); - if result.is_err() { - *state = None; - } - result - }) - .await - .map_err(Error::FollowerWorkerJoin)? - } - - /// Lists bounded retired lanes whose marker predates the caller's grace cutoff. - pub async fn retired_lanes( - &self, - retired_before_ms: i64, - limit: usize, - ) -> Result> { - if retired_before_ms < 0 || !(1..=MAX_RETIRED_LANES).contains(&limit) { - return Err(Error::Node("retired follower scan bound is invalid")); - } - let root = self.root.clone(); - tokio::task::spawn_blocking(move || retired_lanes_sync(&root, retired_before_ms, limit)) - .await - .map_err(Error::FollowerWorkerJoin)? - } - - /// Deletes one exact, grace-aged retired lane after external authority proof. - pub async fn remove_retired( - &self, - candidate: RetiredFollowerLane, - retired_before_ms: i64, - ) -> Result { - if candidate.retired_at_ms > retired_before_ms { - return Err(Error::Node("retired follower grace period has not elapsed")); - } - let lane = Lane { - leader: candidate.leader, - epoch: candidate.epoch, - }; - let lock = self.lane_lock(lane)?; - let cleanup_lock = Arc::clone(&lock); - let root = self.root.clone(); - let retained = Arc::clone(&self.retained); - let removed = tokio::task::spawn_blocking(move || { - let retained = retained - .lock() - .map_err(|_| Error::Node("follower disk reservation lock poisoned"))?; - let _lane = lock - .lock() - .map_err(|_| Error::Node("follower lane lock poisoned"))?; - let removed = remove_retired_sync(&root, lane, candidate, retired_before_ms)?; - retained.resize(follower_bytes(&root)?)?; - Ok::(removed) - }) - .await - .map_err(Error::FollowerWorkerJoin)??; - if removed { - let mut lanes = self - .lanes - .lock() - .map_err(|_| Error::Node("follower store lock poisoned"))?; - if lanes - .get(&lane) - .is_some_and(|current| Arc::ptr_eq(current, &cleanup_lock)) - && Arc::strong_count(&cleanup_lock) == 2 - { - lanes.remove(&lane); - } - } - Ok(removed) - } - - fn lane_lock(&self, lane: Lane) -> Result>>> { - let mut lanes = self - .lanes - .lock() - .map_err(|_| Error::Node("follower store lock poisoned"))?; - Ok(lanes - .entry(lane) - .or_insert_with(|| Arc::new(Mutex::new(None))) - .clone()) - } -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/follower/directory.rs b/crates/crab-cell-runtime/src/follower/directory.rs deleted file mode 100644 index 39fc0031f..000000000 --- a/crates/crab-cell-runtime/src/follower/directory.rs +++ /dev/null @@ -1,346 +0,0 @@ -//! Follower directory layout, quarantine, and retired lanes. - -use super::*; -use crate::identity::encode_hex; - -pub(super) fn directory_bytes(path: &Path) -> Result { - let mut total = 0_u64; - let mut pending = vec![path.to_owned()]; - while let Some(directory) = pending.pop() { - for entry in std::fs::read_dir(directory)? { - let entry = entry?; - let file_type = entry.file_type()?; - if file_type.is_dir() { - pending.push(entry.path()); - } else if file_type.is_file() { - total = total - .checked_add(entry.metadata()?.len()) - .ok_or(Error::Node("follower retained byte count overflow"))?; - } else { - return Err(Error::Node("follower storage contains a special file")); - } - } - } - Ok(total) -} - -pub(super) fn follower_bytes(root: &Path) -> Result { - let followers = root.join("followers"); - let retained = if followers.exists() { - directory_bytes(&followers)? - } else { - 0 - }; - let quarantine = root.join(FOLLOWER_QUARANTINE); - let quarantined = if quarantine.exists() { - directory_bytes(&quarantine)? - } else { - 0 - }; - retained - .checked_add(quarantined) - .ok_or(Error::Node("follower retained byte count overflow")) -} - -pub(super) fn scrub_followers(root: &Path, limits: crab_ltx::Limits) -> Result<()> { - let followers = root.join("followers"); - if !followers.exists() { - return Ok(()); - } - if !std::fs::symlink_metadata(&followers)?.file_type().is_dir() { - quarantine_entry(root, &followers)?; - return Ok(()); - } - let mut leaders = directory_entries(&followers)?; - for leader_path in leaders.drain(..) { - let leader = match directory_lane_component(&leader_path, parse_session_directory) { - Ok(leader) => leader, - Err(_) => { - quarantine_entry(root, &leader_path)?; - continue; - } - }; - let mut epochs = directory_entries(&leader_path)?; - for epoch_path in epochs.drain(..) { - let epoch = match directory_lane_component(&epoch_path, parse_epoch_directory) { - Ok(epoch) => epoch, - Err(_) => { - quarantine_entry(root, &epoch_path)?; - continue; - } - }; - let lane = Lane { leader, epoch }; - let prune_temp = epoch_path.join("open.log.tmp"); - if prune_temp.exists() { - // A crashed rewrite leaves the old open lane authoritative; - // discard only the uncommitted temporary before validation. - if std::fs::symlink_metadata(&prune_temp)? - .file_type() - .is_file() - { - std::fs::remove_file(&prune_temp)?; - sync_directory(&epoch_path)?; - } else { - quarantine_entry(root, &epoch_path)?; - continue; - } - } - if validate_stored_lane(root, lane, limits).is_err() { - quarantine_entry(root, &epoch_path)?; - } - } - if leader_path.exists() && std::fs::read_dir(&leader_path)?.next().is_none() { - std::fs::remove_dir(&leader_path)?; - sync_directory(&followers)?; - } - } - Ok(()) -} - -fn directory_entries(directory: &Path) -> Result> { - let mut entries = std::fs::read_dir(directory)? - .map(|entry| entry.map(|entry| entry.path())) - .collect::>>()?; - entries.sort(); - Ok(entries) -} - -fn directory_lane_component(path: &Path, parse: impl FnOnce(&Path) -> Result) -> Result { - if !std::fs::symlink_metadata(path)?.file_type().is_dir() { - return Err(Error::Node("follower lane component is not a directory")); - } - parse(path) -} - -fn validate_stored_lane(root: &Path, lane: Lane, limits: crab_ltx::Limits) -> Result<()> { - let directory = lane_directory(root, lane); - for entry in std::fs::read_dir(&directory)? { - let entry = entry?; - let name = entry - .file_name() - .into_string() - .map_err(|_| Error::Node("follower lane entry is not UTF-8"))?; - let file_type = entry.file_type()?; - let valid = match name.as_str() { - "chunks" => file_type.is_dir(), - "sealed" | "retired" => file_type.is_file(), - _ => false, - }; - if !valid { - return Err(Error::Node("follower lane contains an invalid entry")); - } - } - let records = scan_lane(&directory.join("chunks"), lane, limits)?; - let durable_through = records.keys().next_back().copied().unwrap_or(0); - let sealed = directory.join("sealed"); - if sealed.exists() - && read_watermark(&sealed, "follower seal marker is invalid")? != durable_through - { - return Err(Error::Node("follower seal watermark differs")); - } - let retired = directory.join("retired"); - if retired.exists() - && read_watermark(&retired, "follower retire marker is invalid")? < durable_through - { - return Err(Error::Node("follower lane has uncovered records")); - } - Ok(()) -} - -fn quarantine_entry(root: &Path, source: &Path) -> Result<()> { - let quarantine = ensure_child(root, FOLLOWER_QUARANTINE)?; - let mut index = quarantine_entry_count(root)? as u64; - let destination = loop { - index = index - .checked_add(1) - .ok_or(Error::Node("follower quarantine index overflow"))?; - let candidate = quarantine.join(format!("{index:020}.bad")); - if !candidate.exists() { - break candidate; - } - }; - std::fs::rename(source, destination)?; - let source_parent = source - .parent() - .ok_or(Error::Node("follower quarantine source has no parent"))?; - sync_directory(source_parent)?; - sync_directory(&quarantine)?; - Ok(()) -} - -pub(super) fn quarantine_entry_count(root: &Path) -> Result { - let quarantine = root.join(FOLLOWER_QUARANTINE); - if !quarantine.exists() { - return Ok(0); - } - Ok( - std::fs::read_dir(quarantine)?.try_fold(0_usize, |count, entry| { - entry?; - count - .checked_add(1) - .ok_or_else(|| std::io::Error::other("follower quarantine entry count overflow")) - })?, - ) -} - -pub(super) fn retired_lanes_sync( - root: &Path, - retired_before_ms: i64, - limit: usize, -) -> Result> { - let followers = root.join("followers"); - if !followers.exists() { - return Ok(Vec::new()); - } - let mut leaders = std::fs::read_dir(&followers)? - .map(|entry| entry.map(|entry| entry.path())) - .collect::>>()?; - leaders.sort(); - let mut retired = Vec::new(); - for leader_path in leaders { - let leader = parse_session_directory(&leader_path)?; - let mut epochs = std::fs::read_dir(&leader_path)? - .map(|entry| entry.map(|entry| entry.path())) - .collect::>>()?; - epochs.sort(); - for epoch_path in epochs { - let epoch = parse_epoch_directory(&epoch_path)?; - let marker = epoch_path.join("retired"); - if !marker.exists() { - continue; - } - let retired_at_ms = modified_at_ms(&marker)?; - if retired_at_ms > retired_before_ms { - continue; - } - retired.push(RetiredFollowerLane { - leader, - epoch, - covered_through: read_watermark(&marker, "follower retire marker is invalid")?, - retired_at_ms, - }); - if retired.len() == limit { - return Ok(retired); - } - } - } - Ok(retired) -} - -pub(super) fn remove_retired_sync( - root: &Path, - lane: Lane, - candidate: RetiredFollowerLane, - retired_before_ms: i64, -) -> Result { - validate_lane(lane)?; - let directory = lane_directory(root, lane); - let marker = directory.join("retired"); - if !marker.exists() { - return Ok(false); - } - if read_watermark(&marker, "follower retire marker is invalid")? != candidate.covered_through - || modified_at_ms(&marker)? != candidate.retired_at_ms - || candidate.retired_at_ms > retired_before_ms - { - return Err(Error::Node("retired follower marker changed")); - } - std::fs::remove_dir_all(&directory)?; - let leader = directory - .parent() - .ok_or(Error::Node("follower leader directory is missing"))?; - sync_directory(leader)?; - if std::fs::read_dir(leader)?.next().is_none() { - std::fs::remove_dir(leader)?; - if let Some(followers) = leader.parent() { - sync_directory(followers)?; - } - } - Ok(true) -} - -fn modified_at_ms(path: &Path) -> Result { - let duration = std::fs::metadata(path)? - .modified()? - .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| Error::Node("follower marker time predates Unix epoch"))?; - i64::try_from(duration.as_millis()).map_err(|_| Error::Node("follower marker time exceeds i64")) -} - -fn parse_session_directory(path: &Path) -> Result { - let name = path - .file_name() - .and_then(|name| name.to_str()) - .ok_or(Error::Node("follower leader directory is not UTF-8"))?; - if name.len() != 32 { - return Err(Error::Node("follower leader directory is invalid")); - } - let mut bytes = [0_u8; 16]; - for (index, pair) in name.as_bytes().as_chunks::<2>().0.iter().enumerate() { - let text = std::str::from_utf8(pair) - .map_err(|_| Error::Node("follower leader directory is invalid"))?; - bytes[index] = u8::from_str_radix(text, 16) - .map_err(|_| Error::Node("follower leader directory is invalid"))?; - } - let session = SessionId::from_bytes(bytes); - validate_lane(Lane { - leader: session, - epoch: 1, - })?; - Ok(session) -} - -fn parse_epoch_directory(path: &Path) -> Result { - let epoch = path - .file_name() - .and_then(|name| name.to_str()) - .and_then(|name| name.parse::().ok()) - .filter(|epoch| *epoch != 0) - .ok_or(Error::Node("follower epoch directory is invalid"))?; - Ok(epoch) -} - -pub(super) fn settle_disk_reservation( - result: Result, - resize: Result<()>, -) -> Result { - // Reconcile the shared admission even when the filesystem operation - // failed. Returning early on `result` would retain the preflight growth - // forever and eventually make unrelated Cells fail closed for capacity. - resize?; - result -} - -pub(super) fn lane_directory(root: &Path, lane: Lane) -> PathBuf { - root.join("followers") - .join(encode_hex(lane.leader.as_bytes())) - .join(lane.epoch.to_string()) -} - -pub(super) fn ensure_lane_directories(root: &Path, lane: Lane) -> Result<()> { - let followers = ensure_child(root, "followers")?; - let leader = ensure_child(&followers, &encode_hex(lane.leader.as_bytes()))?; - let epoch = ensure_child(&leader, &lane.epoch.to_string())?; - ensure_child(&epoch, "chunks")?; - Ok(()) -} - -fn ensure_child(parent: &Path, name: &str) -> Result { - let child = parent.join(name); - if !child.exists() { - std::fs::create_dir(&child)?; - sync_directory(parent)?; - } - Ok(child) -} - -pub(super) fn validate_lane(lane: Lane) -> Result<()> { - if lane.leader.as_bytes().iter().all(|byte| *byte == 0) || lane.epoch == 0 { - return Err(Error::Node("invalid follower lane")); - } - Ok(()) -} - -pub(super) fn sync_directory(path: &Path) -> std::io::Result<()> { - std::fs::File::open(path)?.sync_all() -} diff --git a/crates/crab-cell-runtime/src/follower/records.rs b/crates/crab-cell-runtime/src/follower/records.rs deleted file mode 100644 index c25ce0c0f..000000000 --- a/crates/crab-cell-runtime/src/follower/records.rs +++ /dev/null @@ -1,99 +0,0 @@ -//! Follower lane record framing, appends, tails, and scans. -//! -//! A lane is a directory of chunk files; every helper here encodes or -//! decodes those records, reconciles the admission that reserved their -//! bytes, and never lets a torn or mismatched tail authorize a receipt. - -use super::*; - -mod append; -mod scan; - -pub(super) use append::*; -pub(super) use scan::*; - -#[derive(Clone)] -pub(in crate::follower) struct StoredRecord { - sequence: u64, - digest: [u8; 32], - path: Arc, - offset: u64, - length: usize, -} -pub(in crate::follower) struct IndexReservation { - used: Arc>, - bytes: u64, -} -impl IndexReservation { - fn new(used: &Arc>, bytes: u64) -> Result { - let mut current = used - .lock() - .map_err(|_| Error::Node("follower index reservation lock poisoned"))?; - let next = current - .checked_add(bytes) - .ok_or(Error::Capacity("follower lane index"))?; - if next > MAX_FOLLOWER_INDEX_BYTES { - return Err(Error::Capacity("follower lane index")); - } - *current = next; - Ok(Self { - used: Arc::clone(used), - bytes, - }) - } - - fn grow(&mut self, additional: u64) -> Result<()> { - if additional == 0 { - return Ok(()); - } - let mut current = self - .used - .lock() - .map_err(|_| Error::Node("follower index reservation lock poisoned"))?; - let next = current - .checked_add(additional) - .ok_or(Error::Capacity("follower lane index"))?; - if next > MAX_FOLLOWER_INDEX_BYTES { - return Err(Error::Capacity("follower lane index")); - } - *current = next; - self.bytes = self - .bytes - .checked_add(additional) - .ok_or(Error::Capacity("follower lane index"))?; - Ok(()) - } - - fn shrink_to(&mut self, bytes: u64) { - if bytes >= self.bytes { - return; - } - let released = self.bytes - bytes; - if let Ok(mut current) = self.used.lock() { - *current = current.saturating_sub(released); - } - self.bytes = bytes; - } - - fn resize_to(&mut self, bytes: u64) -> Result<()> { - if bytes > self.bytes { - self.grow(bytes - self.bytes) - } else { - self.shrink_to(bytes); - Ok(()) - } - } -} -impl Drop for IndexReservation { - fn drop(&mut self) { - if let Ok(mut current) = self.used.lock() { - *current = current.saturating_sub(self.bytes); - } - } -} -pub(in crate::follower) struct LaneMemory { - records: BTreeMap, - open_first: Option, - open_last: Option, - index: IndexReservation, -} diff --git a/crates/crab-cell-runtime/src/follower/records/append.rs b/crates/crab-cell-runtime/src/follower/records/append.rs deleted file mode 100644 index 63e133ddc..000000000 --- a/crates/crab-cell-runtime/src/follower/records/append.rs +++ /dev/null @@ -1,435 +0,0 @@ -//! Lane appends, seals, retirements, and chunk rewrites. - -use super::*; - -pub(in crate::follower) fn append_sync( - root: &Path, - lane: Lane, - frames: Vec, - covered_through: u64, - limits: crab_ltx::Limits, - index_used: &Arc>, - state: &mut Option, - scan_counter: &ScanCounter, -) -> Result { - validate_lane(lane)?; - let directory = lane_directory(root, lane); - let chunks = directory.join("chunks"); - ensure_lane_directories(root, lane)?; - if directory.join("retired").exists() { - return Err(Error::Node("follower lane is retired")); - } - if directory.join("sealed").exists() { - return Err(Error::Node("follower lane is sealed")); - } - if state.is_none() { - let retained = scan_lane_counted(&chunks, lane, limits, scan_counter)?; - let open_records = scan_chunk(&chunks.join("open.log"), lane, limits, true)?; - *state = Some(lane_memory(retained, &open_records, index_used)?); - } - let pruned_through = prune_covered(&chunks, lane, covered_through, limits)?; - let state = state - .as_mut() - .ok_or(Error::Node("follower lane state did not initialize"))?; - if pruned_through.is_some() { - let records = scan_lane_counted(&chunks, lane, limits, scan_counter)?; - let index_bytes = u64::try_from(records.len()) - .map_err(|_| Error::Capacity("follower lane index"))? - .checked_mul(INDEX_BYTES_PER_RECORD) - .ok_or(Error::Capacity("follower lane index"))?; - state.index.resize_to(index_bytes)?; - state.records = records; - let open_records = scan_chunk(&chunks.join("open.log"), lane, limits, true)?; - state.open_first = open_records.first().map(|record| record.sequence); - state.open_last = open_records.last().map(|record| record.sequence); - } - let mut durable_through = state - .records - .keys() - .next_back() - .copied() - .unwrap_or(covered_through); - if durable_through < covered_through { - durable_through = covered_through; - } - - let open_path = chunks.join("open.log"); - let mut file = open_append(&open_path)?; - let mut pending: Vec = Vec::new(); - let mut pending_digests = HashMap::new(); - let mut open_first = state.open_first; - let mut open_last = state.open_last; - let frame_count = - u64::try_from(frames.len()).map_err(|_| Error::Capacity("follower lane index"))?; - state.index.grow( - frame_count - .checked_mul(INDEX_BYTES_PER_RECORD) - .ok_or(Error::Capacity("follower lane index"))?, - )?; - for encoded in frames { - let frame = crab_ltx::inspect_node_frame(encoded.clone(), limits)?; - let scope = frame.scope(); - if scope.leader_session != *lane.leader.as_bytes() || scope.log_epoch != lane.epoch { - return Err(Error::Node("follower frame changed lane scope")); - } - let sequence = scope.node_sequence; - let digest = frame.digest(); - if let Some(existing) = state.records.get(&sequence) { - if existing.digest != digest { - return Err(Error::Node("conflicting duplicate follower frame")); - } - continue; - } - if let Some(existing) = pending_digests.get(&sequence) - && *existing != digest - { - return Err(Error::Node("conflicting duplicate follower frame")); - } - if pending_digests.contains_key(&sequence) { - continue; - } - // Object publication can advance while this frame is still queued for - // shipping. Its authoritative coverage makes a missing prefix safe to - // skip; the follower must still persist every uncovered suffix frame. - if sequence <= covered_through { - continue; - } - if sequence != durable_through.saturating_add(1) { - return Err(Error::Node("follower append has a sequence gap")); - } - let record_bytes = RECORD_HEADER_BYTES as u64 + encoded.len() as u64; - if file.metadata()?.len() > 0 - && file.metadata()?.len().saturating_add(record_bytes) > ROTATE_BYTES - { - file.sync_data()?; - drop(file); - let destination = rotate_open(&chunks, &open_path, open_first, open_last)?; - relocate_records( - &mut state.records, - open_first.ok_or(Error::Node("follower open range is missing"))?, - open_last.ok_or(Error::Node("follower open range is missing"))?, - &destination, - ); - let destination: Arc = Arc::from(destination.as_path()); - for record in &mut pending { - record.path = Arc::clone(&destination); - } - file = open_append(&open_path)?; - open_first = None; - } - let offset = file.metadata()?.len(); - write_record(&mut file, sequence, digest, &encoded)?; - open_first.get_or_insert(sequence); - open_last = Some(sequence); - durable_through = sequence; - pending_digests.insert(sequence, digest); - pending.push(StoredRecord { - sequence, - digest, - path: Arc::from(open_path.as_path()), - offset: offset - .checked_add(RECORD_HEADER_BYTES as u64) - .ok_or(Error::Node("follower record offset overflow"))?, - length: encoded.len(), - }); - } - if !pending.is_empty() { - file.sync_data()?; - state - .records - .extend(pending.into_iter().map(|record| (record.sequence, record))); - state.open_first = open_first; - state.open_last = open_last; - } - let index_bytes = u64::try_from(state.records.len()) - .map_err(|_| Error::Capacity("follower lane index"))? - .checked_mul(INDEX_BYTES_PER_RECORD) - .ok_or(Error::Capacity("follower lane index"))?; - state.index.shrink_to(index_bytes); - let base_sequence = state - .records - .keys() - .next() - .copied() - .unwrap_or_else(|| durable_through.saturating_add(1)); - Ok(FollowerReceipt { - base_sequence, - durable_through, - }) -} -pub(in crate::follower) fn seal_sync( - root: &Path, - lane: Lane, - limits: crab_ltx::Limits, - index_used: &Arc>, - state: &mut Option, - scan_counter: &ScanCounter, -) -> Result { - validate_lane(lane)?; - let directory = lane_directory(root, lane); - let chunks = directory.join("chunks"); - let retired = directory.join("retired"); - if retired.exists() { - let covered_through = read_watermark(&retired, "follower retire marker is invalid")?; - return Ok(FollowerReceipt { - base_sequence: covered_through.saturating_add(1), - durable_through: covered_through, - }); - } - if state.is_none() { - let retained = scan_lane_counted(&chunks, lane, limits, scan_counter)?; - *state = Some(lane_memory(retained, &[], index_used)?); - } - let state = state - .as_ref() - .ok_or(Error::Node("follower lane state did not initialize"))?; - let durable_through = state.records.keys().next_back().copied().unwrap_or(0); - let base_sequence = state.records.keys().next().copied().unwrap_or(0); - let marker = directory.join("sealed"); - if marker.exists() { - let stored = read_watermark(&marker, "follower seal marker is invalid")?; - if stored != durable_through { - return Err(Error::Node("follower seal watermark differs")); - } - } else { - let mut file = std::fs::OpenOptions::new() - .write(true) - .create_new(true) - .open(&marker)?; - file.write_all(&durable_through.to_le_bytes())?; - file.sync_all()?; - sync_directory(&directory)?; - } - Ok(FollowerReceipt { - base_sequence, - durable_through, - }) -} -pub(in crate::follower) fn retire_sync( - root: &Path, - lane: Lane, - covered_through: u64, - limits: crab_ltx::Limits, - scan_counter: &ScanCounter, -) -> Result { - validate_lane(lane)?; - ensure_lane_directories(root, lane)?; - let directory = lane_directory(root, lane); - let chunks = directory.join("chunks"); - let retained = scan_lane_counted(&chunks, lane, limits, scan_counter)?; - let durable_through = retained.keys().next_back().copied().unwrap_or(0); - if durable_through > covered_through { - return Err(Error::Node("follower lane has uncovered records")); - } - let marker = directory.join("retired"); - if marker.exists() { - if read_watermark(&marker, "follower retire marker is invalid")? != covered_through { - return Err(Error::Node("follower retire watermark differs")); - } - } else { - let mut file = std::fs::OpenOptions::new() - .write(true) - .create_new(true) - .open(&marker)?; - file.write_all(&covered_through.to_le_bytes())?; - file.sync_all()?; - sync_directory(&directory)?; - } - if chunks.exists() { - std::fs::remove_dir_all(&chunks)?; - } - let sealed = directory.join("sealed"); - if sealed.exists() { - std::fs::remove_file(sealed)?; - } - sync_directory(&directory)?; - Ok(FollowerReceipt { - base_sequence: covered_through.saturating_add(1), - durable_through: covered_through, - }) -} -pub(in crate::follower) fn write_record( - file: &mut std::fs::File, - sequence: u64, - digest: [u8; 32], - encoded: &[u8], -) -> Result<()> { - file.write_all(RECORD_MAGIC)?; - file.write_all(&sequence.to_le_bytes())?; - file.write_all(&(encoded.len() as u64).to_le_bytes())?; - file.write_all(&digest)?; - file.write_all(encoded)?; - Ok(()) -} -pub(in crate::follower) fn parse_record_header( - header: &[u8; RECORD_HEADER_BYTES], -) -> Result<(u64, u64, [u8; 32])> { - if &header[..4] != RECORD_MAGIC { - return Err(Error::Node("invalid follower record magic")); - } - let sequence = u64::from_le_bytes( - header[4..12] - .try_into() - .map_err(|_| Error::Node("invalid follower record sequence"))?, - ); - let length = u64::from_le_bytes( - header[12..20] - .try_into() - .map_err(|_| Error::Node("invalid follower record length"))?, - ); - let digest = header[20..] - .try_into() - .map_err(|_| Error::Node("invalid follower record digest"))?; - if sequence == 0 || length == 0 { - return Err(Error::Node("invalid follower record header")); - } - Ok((sequence, length, digest)) -} -pub(in crate::follower) fn open_append(path: &Path) -> Result { - let existed = path.exists(); - let file = std::fs::OpenOptions::new() - .create(true) - .append(true) - .read(true) - .open(path)?; - if !existed { - let parent = path - .parent() - .ok_or(Error::Node("follower chunk has no parent"))?; - sync_directory(parent)?; - } - Ok(file) -} -pub(in crate::follower) fn rotate_open( - chunks: &Path, - open_path: &Path, - first: Option, - last: Option, -) -> Result { - let (Some(first), Some(last)) = (first, last) else { - return Err(Error::Node("cannot rotate an empty follower chunk")); - }; - let destination = chunks.join(format!("{first:020}-{last:020}.log")); - std::fs::rename(open_path, &destination)?; - sync_directory(chunks)?; - Ok(destination) -} -pub(in crate::follower) fn relocate_records( - records: &mut BTreeMap, - first: u64, - last: u64, - destination: &Path, -) { - let path: Arc = Arc::from(destination); - for record in records.range_mut(first..=last).map(|(_, record)| record) { - record.path = Arc::clone(&path); - } -} -pub(in crate::follower) fn prune_covered( - chunks: &Path, - lane: Lane, - covered_through: u64, - limits: crab_ltx::Limits, -) -> Result> { - if !chunks.exists() { - return Ok(None); - } - let mut removed = false; - let mut open_removed = false; - let mut pruned_through = None; - for entry in std::fs::read_dir(chunks)? { - let entry = entry?; - let name = entry.file_name(); - let Some(name) = name.to_str() else { - return Err(Error::Node("follower chunk name is not UTF-8")); - }; - if name == "open.log" { - continue; - } - let Some((_, last)) = parse_chunk_name(name) else { - return Err(Error::Node("invalid follower chunk name")); - }; - if last <= covered_through { - std::fs::remove_file(entry.path())?; - removed = true; - pruned_through = Some(pruned_through.map_or(last, |current: u64| current.max(last))); - } - } - let open_path = chunks.join("open.log"); - if open_path.exists() { - let records = scan_chunk(&open_path, lane, limits, true)?; - let mut retained = Vec::with_capacity(records.len()); - for record in records { - if record.sequence <= covered_through { - removed = true; - open_removed = true; - pruned_through = Some( - pruned_through - .map_or(record.sequence, |current: u64| current.max(record.sequence)), - ); - } else { - retained.push(record); - } - } - if open_removed { - rewrite_open_chunk(&open_path, retained)?; - } - } - if removed { - sync_directory(chunks)?; - } - Ok(pruned_through) -} -pub(in crate::follower) fn rewrite_open_chunk( - path: &Path, - records: Vec, -) -> Result<()> { - let parent = path - .parent() - .ok_or(Error::Node("follower open chunk has no parent"))?; - let temporary = parent.join("open.log.tmp"); - if temporary.exists() { - std::fs::remove_file(&temporary)?; - } - if records.is_empty() { - std::fs::remove_file(path)?; - sync_directory(parent)?; - return Ok(()); - } - - // Covered prefixes are rewritten through a synced temporary and atomically - // renamed so a restart sees either the old contiguous lane or the new one. - let result = (|| { - let mut source = std::fs::File::open(path)?; - let mut output = std::fs::OpenOptions::new() - .write(true) - .create_new(true) - .open(&temporary)?; - for record in records { - source.seek(SeekFrom::Start(record.offset))?; - let mut encoded = vec![0; record.length]; - source.read_exact(&mut encoded)?; - if *blake3::hash(&encoded).as_bytes() != record.digest { - return Err(Error::Node("stored follower record changed during prune")); - } - write_record(&mut output, record.sequence, record.digest, &encoded)?; - } - output.sync_data()?; - drop(output); - std::fs::rename(&temporary, path)?; - sync_directory(parent)?; - Ok(()) - })(); - if result.is_err() { - let _ = std::fs::remove_file(&temporary); - } - result -} -pub(in crate::follower) fn parse_chunk_name(name: &str) -> Option<(u64, u64)> { - let value = name.strip_suffix(".log")?; - let (first, last) = value.split_once('-')?; - if first.len() != 20 || last.len() != 20 { - return None; - } - Some((first.parse().ok()?, last.parse().ok()?)) -} diff --git a/crates/crab-cell-runtime/src/follower/records/scan.rs b/crates/crab-cell-runtime/src/follower/records/scan.rs deleted file mode 100644 index 313583725..000000000 --- a/crates/crab-cell-runtime/src/follower/records/scan.rs +++ /dev/null @@ -1,337 +0,0 @@ -//! Lane tails, indexed records, and chunk scans. - -use super::*; - -#[expect( - clippy::too_many_arguments, - reason = "keeps tail bounds, cache ownership, and scan instrumentation explicit" -)] -pub(in crate::follower) fn read_tail_sync( - root: &Path, - lane: Lane, - first_sequence: u64, - limits: crab_ltx::Limits, - index_used: &Arc>, - state: &mut Option, - max_bytes: usize, - max_frames: usize, - scan_counter: &ScanCounter, -) -> Result { - validate_lane(lane)?; - let directory = lane_directory(root, lane); - let marker = directory.join("sealed"); - if !marker.exists() { - return Err(Error::Node("follower lane is not sealed")); - } - if state.is_none() { - let retained = scan_lane_counted(&directory.join("chunks"), lane, limits, scan_counter)?; - let open_records = scan_chunk(&directory.join("chunks/open.log"), lane, limits, true)?; - let scan_only = retained.clone(); - match lane_memory(retained, &open_records, index_used) { - Ok(memory) => *state = Some(memory), - Err(Error::Capacity(_)) => { - return read_tail_records( - root, - lane, - first_sequence, - limits, - &scan_only, - max_bytes, - max_frames, - ); - } - Err(error) => return Err(error), - } - } - let state = state - .as_ref() - .ok_or(Error::Node("follower lane state did not initialize"))?; - read_tail_records( - root, - lane, - first_sequence, - limits, - &state.records, - max_bytes, - max_frames, - ) -} -pub(in crate::follower) fn read_tail_records( - root: &Path, - lane: Lane, - first_sequence: u64, - limits: crab_ltx::Limits, - records: &BTreeMap, - max_bytes: usize, - max_frames: usize, -) -> Result { - let directory = lane_directory(root, lane); - let marker = directory.join("sealed"); - let durable_through = records.keys().next_back().copied().unwrap_or(0); - if read_watermark(&marker, "follower seal marker is invalid")? != durable_through { - return Err(Error::Node("follower seal watermark differs")); - } - if records.is_empty() || first_sequence < *records.keys().next().unwrap_or(&u64::MAX) { - return Err(Error::Node("requested follower tail is not retained")); - } - if first_sequence > durable_through.saturating_add(1) { - return Err(Error::Node( - "requested follower tail is beyond durable data", - )); - } - let mut frames = Vec::new(); - let mut bytes = 0_usize; - let mut next_sequence = None; - for (sequence, record) in records.range(first_sequence..) { - if frames.len() == max_frames - || (!frames.is_empty() && bytes.saturating_add(record.length) > max_bytes) - { - next_sequence = Some(*sequence); - break; - } - let encoded = read_indexed_record(root, lane, record, limits)?; - bytes = bytes.saturating_add(record.length); - frames.push(encoded); - } - Ok(FollowerTailPage { - frames, - next_sequence, - }) -} -pub(in crate::follower) fn read_indexed_record( - root: &Path, - lane: Lane, - record: &StoredRecord, - limits: crab_ltx::Limits, -) -> Result { - let chunks = lane_directory(root, lane).join("chunks"); - if record.path.parent() != Some(chunks.as_path()) - || !std::fs::symlink_metadata(&record.path)? - .file_type() - .is_file() - { - return Err(Error::Node("indexed follower record path is invalid")); - } - let header_offset = record - .offset - .checked_sub(RECORD_HEADER_BYTES as u64) - .ok_or(Error::Node("indexed follower record offset is invalid"))?; - let mut file = std::fs::File::open(&record.path)?; - file.seek(SeekFrom::Start(header_offset))?; - let mut header = [0_u8; RECORD_HEADER_BYTES]; - file.read_exact(&mut header)?; - let (sequence, length, digest) = parse_record_header(&header)?; - if sequence != record.sequence || length != record.length as u64 || digest != record.digest { - return Err(Error::Node("indexed follower record header changed")); - } - let mut encoded = vec![0; record.length]; - file.read_exact(&mut encoded)?; - if *blake3::hash(&encoded).as_bytes() != record.digest { - return Err(Error::Node("stored follower record changed after index")); - } - let frame = crab_ltx::inspect_node_frame(Bytes::from(encoded.clone()), limits)?; - let scope = frame.scope(); - if scope.node_sequence != record.sequence - || scope.leader_session != *lane.leader.as_bytes() - || scope.log_epoch != lane.epoch - || frame.digest() != record.digest - { - return Err(Error::Node("indexed follower record scope changed")); - } - Ok(Bytes::from(encoded)) -} -pub(in crate::follower) fn lane_memory( - records: BTreeMap, - open_records: &[StoredRecord], - index_used: &Arc>, -) -> Result { - let record_count = - u64::try_from(records.len()).map_err(|_| Error::Capacity("follower lane index"))?; - let bytes = record_count - .checked_mul(INDEX_BYTES_PER_RECORD) - .ok_or(Error::Capacity("follower lane index"))?; - Ok(LaneMemory { - open_first: open_records.first().map(|record| record.sequence), - open_last: open_records.last().map(|record| record.sequence), - index: IndexReservation::new(index_used, bytes)?, - records, - }) -} -pub(in crate::follower) fn scan_lane( - chunks: &Path, - lane: Lane, - limits: crab_ltx::Limits, -) -> Result> { - if !chunks.exists() { - return Ok(BTreeMap::new()); - } - let mut paths = std::fs::read_dir(chunks)? - .map(|entry| entry.map(|entry| entry.path())) - .collect::>>()?; - paths.sort(); - let mut records = BTreeMap::new(); - for path in paths { - let name = path - .file_name() - .and_then(|name| name.to_str()) - .ok_or(Error::Node("follower chunk name is not UTF-8"))?; - let open = name == "open.log"; - let expected = if open { - None - } else { - Some(parse_chunk_name(name).ok_or(Error::Node("invalid follower chunk name"))?) - }; - let chunk = scan_chunk(&path, lane, limits, open)?; - if let Some((first, last)) = expected - && (chunk.first().map(|record| record.sequence) != Some(first) - || chunk.last().map(|record| record.sequence) != Some(last)) - { - return Err(Error::Node("follower chunk name differs from contents")); - } - for record in chunk { - match records.entry(record.sequence) { - Entry::Vacant(entry) => { - entry.insert(record); - } - Entry::Occupied(entry) if entry.get().digest == record.digest => {} - Entry::Occupied(_) => { - return Err(Error::Node("conflicting stored follower frame")); - } - } - } - } - if !records - .keys() - .copied() - .collect::>() - .windows(2) - .all(|pair| pair[0].checked_add(1) == Some(pair[1])) - { - return Err(Error::Node("stored follower lane has a sequence gap")); - } - Ok(records) -} -pub(in crate::follower) fn scan_lane_counted( - chunks: &Path, - lane: Lane, - limits: crab_ltx::Limits, - counter: &ScanCounter, -) -> Result> { - count_scan(counter); - scan_lane(chunks, lane, limits) -} -pub(in crate::follower) fn read_watermark(path: &Path, invalid: &'static str) -> Result { - let bytes = std::fs::read(path)?; - let watermark = bytes - .as_slice() - .try_into() - .map(u64::from_le_bytes) - .map_err(|_| Error::Node(invalid))?; - Ok(watermark) -} -pub(in crate::follower) fn scan_chunk( - path: &Path, - lane: Lane, - limits: crab_ltx::Limits, - truncate_suffix: bool, -) -> Result> { - let mut file = match std::fs::OpenOptions::new() - .read(true) - .write(truncate_suffix) - .open(path) - { - Ok(file) => file, - Err(error) if error.kind() == std::io::ErrorKind::NotFound => return Ok(Vec::new()), - Err(error) => return Err(error.into()), - }; - let mut records = Vec::new(); - let mut valid_bytes = 0_u64; - let path: Arc = Arc::from(path); - loop { - let mut header = [0_u8; RECORD_HEADER_BYTES]; - match file.read_exact(&mut header) { - Ok(()) => {} - Err(error) if error.kind() == std::io::ErrorKind::UnexpectedEof => { - return finish_scan(file, records, valid_bytes, truncate_suffix); - } - Err(error) => return Err(error.into()), - } - let parsed = parse_record_header(&header); - let (sequence, length, digest) = match parsed { - Ok(value) => value, - Err(_error) if truncate_suffix => { - file.set_len(valid_bytes)?; - file.sync_data()?; - return Ok(records); - } - Err(error) => return Err(error), - }; - if length > limits.max_capture_bytes.saturating_add(240) || length > usize::MAX as u64 { - if truncate_suffix { - file.set_len(valid_bytes)?; - file.sync_data()?; - return Ok(records); - } - return Err(Error::Node("stored follower frame exceeds limit")); - } - let mut encoded = vec![0; length as usize]; - if let Err(error) = file.read_exact(&mut encoded) { - if truncate_suffix && error.kind() == std::io::ErrorKind::UnexpectedEof { - file.set_len(valid_bytes)?; - file.sync_data()?; - return Ok(records); - } - return Err(error.into()); - } - let encoded = Bytes::from(encoded); - let frame = match crab_ltx::inspect_node_frame(encoded.clone(), limits) { - Ok(frame) if frame.digest() == digest => frame, - Ok(_) | Err(_) if truncate_suffix => { - file.set_len(valid_bytes)?; - file.sync_data()?; - return Ok(records); - } - Ok(_) => return Err(Error::Node("stored follower record digest differs")), - Err(error) => return Err(error.into()), - }; - let scope = frame.scope(); - if scope.node_sequence != sequence - || scope.leader_session != *lane.leader.as_bytes() - || scope.log_epoch != lane.epoch - { - if truncate_suffix { - file.set_len(valid_bytes)?; - file.sync_data()?; - return Ok(records); - } - return Err(Error::Node("stored follower record scope differs")); - } - // Keep only verified locations, not every body in the lane. Restart, - // seal and paged reads must not allocate the entire retained log. - records.push(StoredRecord { - sequence, - digest, - path: Arc::clone(&path), - offset: valid_bytes + RECORD_HEADER_BYTES as u64, - length: encoded.len(), - }); - valid_bytes = valid_bytes - .checked_add(RECORD_HEADER_BYTES as u64 + length) - .ok_or(Error::Node("follower chunk length overflow"))?; - } -} -pub(in crate::follower) fn finish_scan( - mut file: std::fs::File, - records: Vec, - valid_bytes: u64, - truncate_suffix: bool, -) -> Result> { - if file.seek(SeekFrom::End(0))? != valid_bytes { - if !truncate_suffix { - return Err(Error::Node("sealed follower chunk has a torn suffix")); - } - file.set_len(valid_bytes)?; - file.sync_data()?; - } - Ok(records) -} diff --git a/crates/crab-cell-runtime/src/follower/tests.rs b/crates/crab-cell-runtime/src/follower/tests.rs deleted file mode 100644 index 3f2b6dab9..000000000 --- a/crates/crab-cell-runtime/src/follower/tests.rs +++ /dev/null @@ -1,27 +0,0 @@ -use super::*; -use crab_ltx::{Db, NodeFrameScope, encode_node_frame}; - -fn frame(sequence: u64, segment: &crab_ltx::LocalSegment, limits: crab_ltx::Limits) -> Bytes { - encode_node_frame( - NodeFrameScope { - leader_session: [1; 16], - log_epoch: 2, - node_sequence: sequence, - application: [3; 16], - cell: [4; 32], - incarnation: [5; 16], - cell_epoch: 6, - commit_sequence: sequence, - }, - segment.info().clone(), - Bytes::from(std::fs::read(segment.path()).unwrap()), - limits, - ) - .unwrap() - .encoded() - .clone() -} - -mod append; -mod budget; -mod scan; diff --git a/crates/crab-cell-runtime/src/follower/tests/append.rs b/crates/crab-cell-runtime/src/follower/tests/append.rs deleted file mode 100644 index aef398034..000000000 --- a/crates/crab-cell-runtime/src/follower/tests/append.rs +++ /dev/null @@ -1,300 +0,0 @@ -//! Lane appends, torn-suffix recovery, deduplication, seals, and retirement. - -use super::*; - -#[tokio::test] -async fn append_recovers_torn_suffix_deduplicates_and_seals() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ - INSERT INTO events(body) VALUES ('one')", - ) - }) - .unwrap(); - let first = database.capture().unwrap(); - database - .transaction(|transaction| { - transaction.execute("INSERT INTO events(body) VALUES ('two')", [])?; - Ok(()) - }) - .unwrap(); - let second = database.capture().unwrap(); - let first_frame = frame(1, first.segments.first().unwrap(), limits); - let second_frame = frame(2, second.segments.first().unwrap(), limits); - - let root = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let leader = SessionId::from_bytes([1; 16]); - assert_eq!( - store - .append(leader, 2, vec![first_frame.clone()], 0) - .await - .unwrap() - .durable_through, - 1 - ); - assert_eq!( - store - .append(leader, 2, vec![first_frame.clone()], 0) - .await - .unwrap() - .durable_through, - 1 - ); - - let open = lane_directory(root.path(), Lane { leader, epoch: 2 }) - .join("chunks") - .join("open.log"); - std::fs::OpenOptions::new() - .append(true) - .open(&open) - .unwrap() - .write_all(b"torn") - .unwrap(); - drop(store); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - assert_eq!( - std::fs::metadata(&open).unwrap().len(), - (RECORD_HEADER_BYTES + first_frame.len()) as u64 - ); - assert_eq!( - store - .append(leader, 2, vec![second_frame.clone()], 0) - .await - .unwrap() - .durable_through, - 2 - ); - assert_eq!(store.seal(leader, 2).await.unwrap().base_sequence, 1); - assert_eq!(store.read_tail(leader, 2, 1).await.unwrap().len(), 2); - let page = store.read_tail_page(leader, 2, 1).await.unwrap(); - assert_eq!(page.frames.len(), 2); - assert_eq!(page.next_sequence, None); - assert!( - store - .append(leader, 2, vec![second_frame], 0) - .await - .is_err() - ); - std::fs::write( - lane_directory(root.path(), Lane { leader, epoch: 2 }).join("sealed"), - b"torn", - ) - .unwrap(); - assert!(store.read_tail(leader, 2, 1).await.is_err()); - database.close().unwrap(); -} -#[tokio::test] -async fn append_skips_an_object_covered_queued_prefix() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ - INSERT INTO events(body) VALUES ('one')", - ) - }) - .unwrap(); - let first = database.capture().unwrap(); - database - .transaction(|transaction| { - transaction.execute("INSERT INTO events(body) VALUES ('two')", [])?; - Ok(()) - }) - .unwrap(); - let second = database.capture().unwrap(); - let first_frame = frame(1, first.segments.first().unwrap(), limits); - let second_frame = frame(2, second.segments.first().unwrap(), limits); - - for covered in [1, 2] { - let root = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let leader = SessionId::from_bytes([1; 16]); - let receipt = store - .append( - leader, - 2, - vec![first_frame.clone(), second_frame.clone()], - covered, - ) - .await - .unwrap(); - assert_eq!(receipt.base_sequence, covered + 1); - assert_eq!(receipt.durable_through, 2); - let member = crate::identity::NodeId::from_bytes([2; 16]); - let recovery = crate::node::log_recovery::NodeLogRecovery::new( - Arc::new(crate::node::log_transport::LocalFollowerTransport::new( - member, store, - )), - crate::identity::NodeId::from_bytes([1; 16]), - leader, - 2, - vec![member], - covered, - true, - limits, - ) - .unwrap(); - let sealed = recovery.ensure_sealed().await.unwrap(); - assert_eq!(sealed.frames.len(), (2 - covered) as usize); - } - database.close().unwrap(); -} -#[tokio::test] -async fn covered_prefix_is_pruned_from_open_lane_before_restart() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ - INSERT INTO events(body) VALUES ('one')", - ) - }) - .unwrap(); - let capture = database.capture().unwrap(); - let frames = [1, 2, 3, 4, 6] - .into_iter() - .map(|sequence| frame(sequence, capture.segments.first().unwrap(), limits)) - .collect::>(); - - let root = tempfile::TempDir::new().unwrap(); - let leader = SessionId::from_bytes([1; 16]); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - store - .append(leader, 2, frames[..4].to_vec(), 0) - .await - .unwrap(); - let receipt = store - .append(leader, 2, vec![frames[4].clone()], 5) - .await - .unwrap(); - assert_eq!(receipt.base_sequence, 6); - assert_eq!(receipt.durable_through, 6); - - drop(store); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - assert_eq!(store.seal(leader, 2).await.unwrap().base_sequence, 6); - assert_eq!(store.read_tail(leader, 2, 6).await.unwrap().len(), 1); - database.close().unwrap(); -} -#[tokio::test] -async fn append_reserves_capacity_before_writing_and_seal_accounts_for_marker() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let capture = database.capture().unwrap(); - let encoded = frame(1, capture.segments.first().unwrap(), limits); - let required = RECORD_HEADER_BYTES as u64 + encoded.len() as u64; - let leader = SessionId::from_bytes([1; 16]); - - let rejected_root = tempfile::TempDir::new().unwrap(); - let rejected = FollowerStore::open( - rejected_root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(required - 1), - ) - .unwrap(); - assert!( - rejected - .append(leader, 2, vec![encoded.clone()], 0) - .await - .is_err() - ); - assert_eq!(directory_bytes(rejected_root.path()).unwrap(), 0); - - let root = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(required + 8), - ) - .unwrap(); - store.append(leader, 2, vec![encoded], 0).await.unwrap(); - assert_eq!(store.retained_bytes(), required); - store.seal(leader, 2).await.unwrap(); - assert_eq!(store.retained_bytes(), required + 8); - assert_eq!(store.available_bytes(), 0); - database.close().unwrap(); -} -#[tokio::test] -async fn retire_requires_full_coverage_and_persists_an_append_fence() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let capture = database.capture().unwrap(); - let encoded = frame(1, capture.segments.first().unwrap(), limits); - let required = RECORD_HEADER_BYTES as u64 + encoded.len() as u64; - let leader = SessionId::from_bytes([1; 16]); - let root = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(required + 8), - ) - .unwrap(); - store - .append(leader, 2, vec![encoded.clone()], 0) - .await - .unwrap(); - - assert!(store.retire(leader, 2, 0).await.is_err()); - assert_eq!(store.retained_bytes(), required); - assert_eq!( - store.retire(leader, 2, 1).await.unwrap(), - FollowerReceipt { - base_sequence: 2, - durable_through: 1, - } - ); - assert_eq!(store.retained_bytes(), 8); - assert_eq!(store.retire(leader, 2, 1).await.unwrap().durable_through, 1); - assert!(store.retire(leader, 2, 2).await.is_err()); - assert!(store.append(leader, 2, vec![encoded], 1).await.is_err()); - assert_eq!(store.seal(leader, 2).await.unwrap().durable_through, 1); - - drop(store); - let reopened = - FollowerStore::open(root.path().to_owned(), limits, crab_ltx::DiskBudget::new(8)).unwrap(); - assert_eq!(reopened.retained_bytes(), 8); - assert_eq!(reopened.seal(leader, 2).await.unwrap().durable_through, 1); - database.close().unwrap(); -} diff --git a/crates/crab-cell-runtime/src/follower/tests/budget.rs b/crates/crab-cell-runtime/src/follower/tests/budget.rs deleted file mode 100644 index 923f5851b..000000000 --- a/crates/crab-cell-runtime/src/follower/tests/budget.rs +++ /dev/null @@ -1,72 +0,0 @@ -//! Follower byte budgets, admission, and grace-aged lane collection. - -use super::*; - -#[test] -fn open_reserves_only_existing_follower_bytes_and_rejects_an_undersized_budget() { - let root = tempfile::TempDir::new().unwrap(); - std::fs::create_dir(root.path().join("followers")).unwrap(); - std::fs::write(root.path().join("followers/retained.log"), [0_u8; 17]).unwrap(); - std::fs::create_dir(root.path().join("sessions")).unwrap(); - std::fs::write(root.path().join("sessions/unrelated.sqlite"), [0_u8; 64]).unwrap(); - - assert!( - FollowerStore::open( - root.path().to_owned(), - crab_ltx::Limits::default(), - crab_ltx::DiskBudget::new(16), - ) - .is_err() - ); - - let store = FollowerStore::open( - root.path().to_owned(), - crab_ltx::Limits::default(), - crab_ltx::DiskBudget::new(32), - ) - .unwrap(); - assert_eq!(store.retained_bytes(), 17); - assert_eq!(store.available_bytes(), 15); - assert_eq!(store.quarantined_entries(), 1); - assert!(!root.path().join("followers/retained.log").exists()); -} -#[tokio::test] -async fn retired_lane_collection_requires_exact_grace_aged_candidate() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let capture = database.capture().unwrap(); - let encoded = frame(1, capture.segments.first().unwrap(), limits); - let leader = SessionId::from_bytes([1; 16]); - let root = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - store.append(leader, 2, vec![encoded], 0).await.unwrap(); - store.retire(leader, 2, 1).await.unwrap(); - - let candidates = store.retired_lanes(i64::MAX, 1).await.unwrap(); - assert_eq!(candidates.len(), 1); - let candidate = candidates[0]; - assert_eq!(candidate.leader(), leader); - assert_eq!(candidate.epoch(), 2); - assert_eq!(candidate.covered_through(), 1); - assert!( - store - .remove_retired(candidate, candidate.retired_at_ms().saturating_sub(1)) - .await - .is_err() - ); - - assert!(store.remove_retired(candidate, i64::MAX).await.unwrap()); - assert!(!store.remove_retired(candidate, i64::MAX).await.unwrap()); - assert_eq!(store.retained_bytes(), 0); - assert!(store.retired_lanes(i64::MAX, 1).await.unwrap().is_empty()); - database.close().unwrap(); -} diff --git a/crates/crab-cell-runtime/src/follower/tests/scan.rs b/crates/crab-cell-runtime/src/follower/tests/scan.rs deleted file mode 100644 index 0c80426de..000000000 --- a/crates/crab-cell-runtime/src/follower/tests/scan.rs +++ /dev/null @@ -1,396 +0,0 @@ -//! Tail pages, cached and indexed tails, and fail-closed scans. - -use super::*; - -#[tokio::test] -async fn restarted_lane_returns_only_the_requested_large_frame_page() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("source.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE values_(v); INSERT INTO values_ VALUES(randomblob(2097152))", - ) - }) - .unwrap(); - let capture = database.capture().unwrap(); - let root = tempfile::TempDir::new().unwrap(); - let leader = SessionId::from_bytes([1; 16]); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - for sequence in 1..=16 { - store - .append( - leader, - 2, - vec![frame(sequence, &capture.segments[0], limits)], - 0, - ) - .await - .unwrap(); - } - store.seal(leader, 2).await.unwrap(); - drop(store); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - for sequence in [1, 16] { - let page = store.read_tail_page(leader, 2, sequence).await.unwrap(); - assert_eq!(page.frames.len(), 1); - assert_eq!(page.next_sequence, (sequence < 16).then_some(sequence + 1)); - let frame = crab_ltx::inspect_node_frame(page.frames[0].clone(), limits).unwrap(); - assert_eq!(frame.scope().node_sequence, sequence); - } - assert_eq!(store.scan_count(), 1); - database.close().unwrap(); -} -#[tokio::test] -async fn cached_tail_rechecks_mutated_frame_bytes() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("source.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE values_(v); INSERT INTO values_ VALUES(randomblob(2097152))", - ) - }) - .unwrap(); - let capture = database.capture().unwrap(); - let root = tempfile::TempDir::new().unwrap(); - let leader = SessionId::from_bytes([1; 16]); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - for sequence in 1..=2 { - store - .append( - leader, - 2, - vec![frame(sequence, &capture.segments[0], limits)], - 0, - ) - .await - .unwrap(); - } - store.seal(leader, 2).await.unwrap(); - drop(store); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - assert_eq!( - store - .read_tail_page(leader, 2, 1) - .await - .unwrap() - .next_sequence, - Some(2) - ); - let chunks = lane_directory(root.path(), Lane { leader, epoch: 2 }).join("chunks"); - let chunk = std::fs::read_dir(chunks) - .unwrap() - .map(|entry| entry.unwrap().path()) - .find(|path| path.extension().and_then(|value| value.to_str()) == Some("log")) - .unwrap(); - let mut file = std::fs::OpenOptions::new() - .read(true) - .write(true) - .open(chunk) - .unwrap(); - file.seek(SeekFrom::End(-1)).unwrap(); - let mut byte = [0_u8; 1]; - file.read_exact(&mut byte).unwrap(); - file.seek(SeekFrom::End(-1)).unwrap(); - byte[0] ^= 0xff; - file.write_all(&byte).unwrap(); - file.sync_data().unwrap(); - assert!(store.read_tail_page(leader, 2, 2).await.is_err()); - assert_eq!(store.scan_count(), 1); - database.close().unwrap(); -} -#[tokio::test] -async fn cached_tail_fails_when_its_chunk_is_deleted() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("source.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE values_(v); INSERT INTO values_ VALUES(randomblob(2097152))", - ) - }) - .unwrap(); - let capture = database.capture().unwrap(); - let root = tempfile::TempDir::new().unwrap(); - let leader = SessionId::from_bytes([1; 16]); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - for sequence in 1..=2 { - store - .append( - leader, - 2, - vec![frame(sequence, &capture.segments[0], limits)], - 0, - ) - .await - .unwrap(); - } - store.seal(leader, 2).await.unwrap(); - drop(store); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - assert_eq!( - store - .read_tail_page(leader, 2, 1) - .await - .unwrap() - .next_sequence, - Some(2) - ); - let chunks = lane_directory(root.path(), Lane { leader, epoch: 2 }).join("chunks"); - let chunk = std::fs::read_dir(chunks) - .unwrap() - .map(|entry| entry.unwrap().path()) - .find(|path| path.extension().and_then(|value| value.to_str()) == Some("log")) - .unwrap(); - std::fs::remove_file(chunk).unwrap(); - assert!(store.read_tail_page(leader, 2, 2).await.is_err()); - assert_eq!(store.scan_count(), 1); - database.close().unwrap(); -} -#[tokio::test] -async fn indexed_tail_revalidates_headers_and_rebuilds_after_disk_mutation() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("source.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch("CREATE TABLE values_(v); INSERT INTO values_ VALUES (1)") - }) - .unwrap(); - let capture = database.capture().unwrap(); - let encoded = frame(1, capture.segments.first().unwrap(), limits); - let root = tempfile::TempDir::new().unwrap(); - let leader = SessionId::from_bytes([1; 16]); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - store.append(leader, 2, vec![encoded], 0).await.unwrap(); - store.seal(leader, 2).await.unwrap(); - assert_eq!( - store - .read_tail_page(leader, 2, 1) - .await - .unwrap() - .frames - .len(), - 1 - ); - - let open = lane_directory(root.path(), Lane { leader, epoch: 2 }) - .join("chunks") - .join("open.log"); - let mut file = std::fs::OpenOptions::new() - .read(true) - .write(true) - .open(&open) - .unwrap(); - use std::io::{Read as _, Seek as _, SeekFrom, Write as _}; - file.seek(SeekFrom::Start(20)).unwrap(); - let mut original = [0_u8; 1]; - file.read_exact(&mut original).unwrap(); - file.seek(SeekFrom::Start(20)).unwrap(); - file.write_all(&[original[0] ^ 1]).unwrap(); - file.sync_data().unwrap(); - assert!(store.read_tail_page(leader, 2, 1).await.is_err()); - - file.seek(SeekFrom::Start(20)).unwrap(); - file.write_all(&original).unwrap(); - file.sync_data().unwrap(); - assert_eq!( - store - .read_tail_page(leader, 2, 1) - .await - .unwrap() - .frames - .len(), - 1 - ); - database.close().unwrap(); -} -#[tokio::test] -async fn tail_read_falls_back_to_scan_when_index_budget_is_exhausted() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("source.sqlite"), limits).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let capture = database.capture().unwrap(); - let encoded = frame(1, capture.segments.first().unwrap(), limits); - let root = tempfile::TempDir::new().unwrap(); - let leader = SessionId::from_bytes([1; 16]); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - store.append(leader, 2, vec![encoded], 0).await.unwrap(); - store.seal(leader, 2).await.unwrap(); - drop(store); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - *store.index_used.lock().unwrap() = MAX_FOLLOWER_INDEX_BYTES; - - let page = store.read_tail_page(leader, 2, 1).await.unwrap(); - assert_eq!(page.frames.len(), 1); - assert_eq!(page.next_sequence, None); - database.close().unwrap(); -} -#[tokio::test] -async fn closed_chunk_name_must_match_verified_record_range() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let capture = database.capture().unwrap(); - let leader = SessionId::from_bytes([1; 16]); - let root = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - store - .append( - leader, - 2, - vec![frame(1, capture.segments.first().unwrap(), limits)], - 0, - ) - .await - .unwrap(); - let chunks = lane_directory(root.path(), Lane { leader, epoch: 2 }).join("chunks"); - drop(store); - std::fs::rename( - chunks.join("open.log"), - chunks.join("00000000000000000002-00000000000000000002.log"), - ) - .unwrap(); - let reopened = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - assert_eq!(reopened.quarantined_entries(), 1); - assert!(!lane_directory(root.path(), Lane { leader, epoch: 2 }).exists()); - assert!(root.path().join(FOLLOWER_QUARANTINE).exists()); - assert!(reopened.seal(leader, 2).await.is_err()); - database.close().unwrap(); -} -#[tokio::test] -async fn conflicting_duplicate_and_sequence_gap_fail_closed() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let first = database.capture().unwrap(); - database - .transaction(|transaction| { - transaction.execute("INSERT INTO values_ VALUES (1)", [])?; - Ok(()) - }) - .unwrap(); - let second = database.capture().unwrap(); - let root = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let leader = SessionId::from_bytes([1; 16]); - store - .append( - leader, - 2, - vec![frame(1, first.segments.first().unwrap(), limits)], - 0, - ) - .await - .unwrap(); - let retained_before_rejected_append = store.retained_bytes(); - let available_before_rejected_append = store.available_bytes(); - assert!( - store - .append( - leader, - 2, - vec![frame(1, second.segments.first().unwrap(), limits)], - 0, - ) - .await - .is_err() - ); - assert_eq!( - store.retained_bytes(), - retained_before_rejected_append, - "a rejected append must not leak disk admission" - ); - assert_eq!( - store.available_bytes(), - available_before_rejected_append, - "a rejected append must restore shared capacity" - ); - assert!( - store - .append( - leader, - 2, - vec![frame(3, second.segments.first().unwrap(), limits)], - 0, - ) - .await - .is_err() - ); - assert_eq!(store.retained_bytes(), retained_before_rejected_append); - assert_eq!(store.available_bytes(), available_before_rejected_append); - database.close().unwrap(); -} diff --git a/crates/crab-cell-runtime/src/identity.rs b/crates/crab-cell-runtime/src/identity.rs deleted file mode 100644 index b445422f4..000000000 --- a/crates/crab-cell-runtime/src/identity.rs +++ /dev/null @@ -1,258 +0,0 @@ -//! Cell, tenant, application, namespace, session, node, and request identities. -use std::fmt; - -use crate::{Error, Result}; - -macro_rules! fixed_id { - ($name:ident, $len:literal) => { - #[doc = concat!("Opaque fixed-width ", stringify!($name), " identity bytes.")] - #[derive(Clone, Copy, PartialEq, Eq, Hash)] - pub struct $name([u8; $len]); - - impl $name { - /// Wraps the fixed-width identity bytes without validation. - #[must_use] - pub const fn from_bytes(bytes: [u8; $len]) -> Self { - Self(bytes) - } - - /// Returns the raw identity bytes. - #[must_use] - pub const fn as_bytes(&self) -> &[u8; $len] { - &self.0 - } - } - - impl TryFrom<&[u8]> for $name { - type Error = Error; - - fn try_from(value: &[u8]) -> Result { - let bytes = value - .try_into() - .map_err(|_| Error::Identity(concat!(stringify!($name), " length")))?; - Ok(Self(bytes)) - } - } - - impl fmt::Debug for $name { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter.write_str(stringify!($name))?; - formatter.write_str("(")?; - write_hex(formatter, &self.0)?; - formatter.write_str(")") - } - } - }; -} - -fixed_id!(TenantId, 16); -fixed_id!(ApplicationId, 16); -fixed_id!(NamespaceId, 16); -fixed_id!(NodeId, 16); -fixed_id!(SessionId, 16); -fixed_id!(IncarnationId, 16); -fixed_id!(RequestId, 16); -fixed_id!(Digest, 32); -fixed_id!(CellId, 32); - -/// A resolved, authorized Cell location before its content hash is derived. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct CellTarget { - tenant: TenantId, - application: ApplicationId, - namespace: NamespaceId, - partition: Vec, -} - -impl CellTarget { - /// Constructs a target from stable IDs and a bounded partition key. - pub fn new( - tenant: TenantId, - application: ApplicationId, - namespace: NamespaceId, - partition: &[u8], - ) -> Result { - if partition.len() > 1024 { - return Err(Error::Identity("partition exceeds 1024 bytes")); - } - Ok(Self { - tenant, - application, - namespace, - partition: partition.to_vec(), - }) - } - - /// Derives the Cell ID from the target's tenant, application, namespace, - /// and partition. - #[must_use] - pub fn cell_id(&self) -> CellId { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.cell.v1\0"); - hasher.update(self.tenant.as_bytes()); - hasher.update(self.application.as_bytes()); - hasher.update(self.namespace.as_bytes()); - hasher.update(&(self.partition.len() as u32).to_be_bytes()); - hasher.update(&self.partition); - CellId::from_bytes(*hasher.finalize().as_bytes()) - } - - /// Returns the tenant this target belongs to. - #[must_use] - pub const fn tenant(&self) -> TenantId { - self.tenant - } - - /// Returns the application this target belongs to. - #[must_use] - pub const fn application(&self) -> ApplicationId { - self.application - } - - /// Returns the namespace that owns this target's Cell. - #[must_use] - pub const fn namespace(&self) -> NamespaceId { - self.namespace - } - - /// Returns the partition key that selected the Cell within the namespace. - #[must_use] - pub fn partition(&self) -> &[u8] { - &self.partition - } -} - -/// Maps one bounded scope to a fixed power-of-two shard count. -pub fn shard_for_scope(namespace: NamespaceId, scope: &[u8], shard_count: u32) -> Result { - if scope.len() > 1024 { - return Err(Error::Identity("scope exceeds 1024 bytes")); - } - if !(1..=4096).contains(&shard_count) || !shard_count.is_power_of_two() { - return Err(Error::Identity( - "shard count must be a power of two in 1..=4096", - )); - } - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.shard.v1\0"); - hasher.update(namespace.as_bytes()); - hasher.update(&(scope.len() as u32).to_be_bytes()); - hasher.update(scope); - let digest = hasher.finalize(); - let prefix: [u8; 8] = digest.as_bytes()[..8] - .try_into() - .map_err(|_| Error::Identity("BLAKE3 prefix"))?; - Ok((u64::from_be_bytes(prefix) % u64::from(shard_count)) as u32) -} - -/// Encodes a primitive shard as the canonical Cell partition bytes. -#[must_use] -pub const fn partition_for_shard(shard: u32) -> [u8; 4] { - shard.to_be_bytes() -} - -pub(crate) fn write_hex(formatter: &mut fmt::Formatter<'_>, bytes: &[u8]) -> fmt::Result { - for byte in bytes { - write!(formatter, "{byte:02x}")?; - } - Ok(()) -} - -pub(crate) fn encode_hex(bytes: &[u8]) -> String { - const TABLE: &[u8; 16] = b"0123456789abcdef"; - let mut encoded = String::with_capacity(bytes.len() * 2); - for byte in bytes { - encoded.push(TABLE[(byte >> 4) as usize] as char); - encoded.push(TABLE[(byte & 0x0f) as usize] as char); - } - encoded -} - -pub(crate) fn decode_hex(value: &str) -> Result<[u8; N]> { - if value.len() != N * 2 { - return Err(Error::Control("hex field length")); - } - let mut decoded = [0; N]; - for (index, pair) in value.as_bytes().as_chunks::<2>().0.iter().enumerate() { - let high = nibble(pair[0]).ok_or(Error::Control("hex field must be lowercase"))?; - let low = nibble(pair[1]).ok_or(Error::Control("hex field must be lowercase"))?; - decoded[index] = (high << 4) | low; - } - Ok(decoded) -} - -/// Decodes one lowercase hex digit, rejecting everything else. -pub(crate) fn nibble(byte: u8) -> Option { - match byte { - b'0'..=b'9' => Some(byte - b'0'), - b'a'..=b'f' => Some(byte - b'a' + 10), - _ => None, - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn cell_identity_binds_every_field_and_partition_length() { - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NamespaceId::from_bytes([3; 16]), - b"repository-42", - ) - .unwrap(); - assert_eq!( - encode_hex(target.cell_id().as_bytes()), - "62da35085973dc3c2a46bba884a9099899c2039418e901c5836005e66c5fb089" - ); - let other = CellTarget::new( - target.tenant(), - target.application(), - target.namespace(), - b"\0\0\0\x0drepository-42", - ) - .unwrap(); - assert_ne!(target.cell_id(), other.cell_id()); - } - - #[test] - fn cell_target_accepts_a_partition_at_the_1024_byte_limit() { - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NamespaceId::from_bytes([3; 16]), - &[0x7f; 1024], - ) - .unwrap(); - assert_eq!(target.partition().len(), 1024); - } - - #[test] - fn cell_target_rejects_a_partition_past_the_limit() { - assert!(matches!( - CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NamespaceId::from_bytes([3; 16]), - &[0x7f; 1025], - ), - Err(Error::Identity("partition exceeds 1024 bytes")) - )); - } - - #[test] - fn shard_mapping_accepts_a_scope_at_the_1024_byte_limit() { - let namespace = NamespaceId::from_bytes([4; 16]); - assert!(shard_for_scope(namespace, &[0; 1024], 64).is_ok()); - } - - #[test] - fn shard_mapping_rejects_unbounded_or_mutable_topology() { - let namespace = NamespaceId::from_bytes([4; 16]); - assert_eq!(shard_for_scope(namespace, b"scope", 64).unwrap(), 61); - assert!(shard_for_scope(namespace, b"scope", 3).is_err()); - assert!(shard_for_scope(namespace, &[0; 1025], 64).is_err()); - assert_eq!(partition_for_shard(54), [0, 0, 0, 54]); - } -} diff --git a/crates/crab-cell-runtime/src/lib.rs b/crates/crab-cell-runtime/src/lib.rs deleted file mode 100644 index 55beec11f..000000000 --- a/crates/crab-cell-runtime/src/lib.rs +++ /dev/null @@ -1,90 +0,0 @@ -//! Embedded SQLite Cell runtime contracts for Crab services. -//! -//! This crate owns reusable Cell identities, control records, transitions, -//! runtime schema installation and the single-Cell command/pending-publication -//! executor. HTTP, authentication and provider construction remain product -//! concerns of `crab-http-server`. - -#![deny(missing_docs)] -// A panic in a filter process or FUSE path corrupts a worktree, so production -// builds deny unwrap, expect, panic, todo, and unimplemented; test builds keep -// them available. -#![cfg_attr( - not(test), - deny( - clippy::unwrap_used, - clippy::expect_used, - clippy::panic, - clippy::todo, - clippy::unimplemented - ) -)] - -mod coordination; -mod error; -mod retry; - -// Test-only object-store instrumentation. Integration targets and the -// `cell_movement_probe` binary enable the feature; in-crate unit tests do not -// depend on it, so the module never enters a default build. -#[cfg(feature = "test-support")] -pub mod test_support; - -pub mod cell; -pub mod client; -pub mod codec; -pub mod control; -pub mod fleet; -pub mod follower; -pub mod identity; -pub mod ltx; -pub mod node; -pub mod peer; -pub mod primitives; -pub mod publication; -pub mod qualification; -pub mod read_policy; -pub mod recovery; -pub mod registry; - -pub use cell::actor::{CellRuntime, CellRuntimeStats}; - -pub use cell::catalog::CatalogRole; -pub use cell::executor::{MutationIdentity, Resolution}; - -pub use cell::worker::SqlWorkerPool; -pub use client::{ - CellClient, CellReadReplica, Committed, InvocationError, Observed, PendingMutation, - PreparedCommand, Receipt, -}; - -pub use error::{Error, Result}; - -pub use follower::FollowerStore; -pub use identity::{ - ApplicationId, CellId, CellTarget, Digest, NamespaceId, SessionId, TenantId, - partition_for_shard, shard_for_scope, -}; -pub use node::durability::NodeDurabilityConfig; -pub use node::lease::NodeLeaseGuard; - -pub use primitives::blob::{BlobArtifactStore, BlobModule, BlobNamespace}; -pub use primitives::cron::{CronModule, CronNamespace}; -pub use primitives::effects::{EffectModule, EffectSource}; -pub use primitives::kv::{KvModule, KvNamespace}; - -pub use primitives::queue::{QueueModule, QueueNamespace}; -pub use primitives::sql::{SqlCell, SqlModule}; -pub use primitives::workflow::{ - WorkflowActivities, WorkflowActivityModule, WorkflowModule, WorkflowNamespace, -}; - -pub use qualification::{ - QualificationExecution, QualificationOperation, QualificationOperationExecutor, - QualificationProfile, QualificationRunSummary, QualificationWorkload, -}; - -pub use registry::{ - BuildDescriptor, CellModule, Command, MigrationDescriptor, ModuleDescriptor, - NamespaceDescriptor, Query, Registry, RegistryBuilder, -}; diff --git a/crates/crab-cell-runtime/src/ltx.rs b/crates/crab-cell-runtime/src/ltx.rs deleted file mode 100644 index 1ed3b4b78..000000000 --- a/crates/crab-cell-runtime/src/ltx.rs +++ /dev/null @@ -1,9 +0,0 @@ -//! LTX surface the runtime exposes to embedders. -//! -//! `crab-cell-host` and product servers may not depend on `crab-ltx` directly, -//! so the runtime re-exports the exact LTX types its own API uses. - -pub use crab_ltx::{ - CaptureTiming, CellObjectKind, CellReplica, CellStorageLayout, DiskBudget, DiskReservation, - Host, Limits, LtxPhase, LtxReadOrigin, LtxRequestOutcome, ScratchMonitor, -}; diff --git a/crates/crab-cell-runtime/src/migrations/blob.sql b/crates/crab-cell-runtime/src/migrations/blob.sql deleted file mode 100644 index b31af55ad..000000000 --- a/crates/crab-cell-runtime/src/migrations/blob.sql +++ /dev/null @@ -1,37 +0,0 @@ -CREATE TABLE blob_uploads ( - upload_id BLOB PRIMARY KEY CHECK (length(upload_id) = 16), - object_key BLOB NOT NULL CHECK (length(object_key) BETWEEN 1 AND 1024), - request_digest BLOB NOT NULL CHECK (length(request_digest) = 32), - condition INTEGER NOT NULL CHECK (condition BETWEEN 0 AND 2), - expected_etag BLOB CHECK (expected_etag IS NULL OR length(expected_etag) = 32), - content_type TEXT CHECK (content_type IS NULL OR length(content_type) BETWEEN 1 AND 256), - metadata BLOB NOT NULL CHECK (length(metadata) <= 8192), - created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0), - expires_at_ms INTEGER NOT NULL CHECK (expires_at_ms > created_at_ms), - completed INTEGER NOT NULL CHECK (completed IN (0, 1)), - etag BLOB CHECK (etag IS NULL OR length(etag) = 32), - size INTEGER NOT NULL CHECK (size >= 0), - part_count INTEGER NOT NULL CHECK (part_count >= 0) -) STRICT, WITHOUT ROWID; -CREATE INDEX blob_upload_expiry ON blob_uploads(expires_at_ms, upload_id); - -CREATE TABLE blob_parts ( - upload_id BLOB NOT NULL REFERENCES blob_uploads(upload_id) ON DELETE CASCADE, - part_number INTEGER NOT NULL CHECK (part_number BETWEEN 1 AND 4096), - digest BLOB NOT NULL CHECK (length(digest) = 32), - size INTEGER NOT NULL CHECK (size BETWEEN 0 AND 262144), - byte_offset INTEGER CHECK (byte_offset IS NULL OR byte_offset >= 0), - PRIMARY KEY (upload_id, part_number) -) STRICT, WITHOUT ROWID; - -CREATE TABLE blob_objects ( - object_key BLOB PRIMARY KEY CHECK (length(object_key) BETWEEN 1 AND 1024), - upload_id BLOB NOT NULL UNIQUE REFERENCES blob_uploads(upload_id), - etag BLOB NOT NULL CHECK (length(etag) = 32), - size INTEGER NOT NULL CHECK (size >= 0), - part_count INTEGER NOT NULL CHECK (part_count BETWEEN 1 AND 4096), - content_type TEXT, - metadata BLOB NOT NULL, - created_at_ms INTEGER NOT NULL CHECK (created_at_ms >= 0), - updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= created_at_ms) -) STRICT, WITHOUT ROWID; diff --git a/crates/crab-cell-runtime/src/migrations/capacity.sql b/crates/crab-cell-runtime/src/migrations/capacity.sql deleted file mode 100644 index de92e309b..000000000 --- a/crates/crab-cell-runtime/src/migrations/capacity.sql +++ /dev/null @@ -1,9 +0,0 @@ -CREATE TABLE capacity_reservations ( - reservation_key BLOB PRIMARY KEY CHECK (length(reservation_key) BETWEEN 1 AND 128), - pages INTEGER NOT NULL CHECK (pages > 0) -) STRICT, WITHOUT ROWID; -CREATE TABLE capacity_total ( - singleton INTEGER PRIMARY KEY CHECK (singleton = 1), - pages INTEGER NOT NULL CHECK (pages >= 0) -) STRICT; -INSERT INTO capacity_total(singleton, pages) VALUES (1, 0); diff --git a/crates/crab-cell-runtime/src/migrations/cron.sql b/crates/crab-cell-runtime/src/migrations/cron.sql deleted file mode 100644 index 32d7a778f..000000000 --- a/crates/crab-cell-runtime/src/migrations/cron.sql +++ /dev/null @@ -1,13 +0,0 @@ -CREATE TABLE cron_schedules ( - schedule_id BLOB PRIMARY KEY CHECK (length(schedule_id) = 16), - target_index INTEGER NOT NULL CHECK (target_index >= 0), - target_partition BLOB NOT NULL CHECK (length(target_partition) <= 1024), - payload BLOB NOT NULL CHECK (length(payload) <= 262144), - interval_ms INTEGER NOT NULL CHECK (interval_ms BETWEEN 1000 AND 31536000000), - next_due_ms INTEGER NOT NULL CHECK (next_due_ms >= 0), - occurrence INTEGER NOT NULL CHECK (occurrence >= 0), - enabled INTEGER NOT NULL CHECK (enabled IN (0, 1)), - generation INTEGER NOT NULL CHECK (generation >= 1), - updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= 0) -) STRICT, WITHOUT ROWID; -CREATE INDEX cron_due ON cron_schedules(enabled, next_due_ms, schedule_id); diff --git a/crates/crab-cell-runtime/src/migrations/kv.sql b/crates/crab-cell-runtime/src/migrations/kv.sql deleted file mode 100644 index 0333db5f9..000000000 --- a/crates/crab-cell-runtime/src/migrations/kv.sql +++ /dev/null @@ -1,11 +0,0 @@ --- KV schema version 1. One namespace per Cell; scope is part of each key. -CREATE TABLE kv_entries ( - scope BLOB NOT NULL, - key BLOB NOT NULL CHECK (length(key) BETWEEN 1 AND 1024), - version BLOB NOT NULL CHECK (length(version) = 28), - value BLOB NOT NULL CHECK (length(value) <= 4194304), - expires_at_ms INTEGER, - PRIMARY KEY (scope, key) -) STRICT, WITHOUT ROWID; -CREATE INDEX kv_expiry ON kv_entries(expires_at_ms) - WHERE expires_at_ms IS NOT NULL; diff --git a/crates/crab-cell-runtime/src/migrations/queue.sql b/crates/crab-cell-runtime/src/migrations/queue.sql deleted file mode 100644 index fd6e2908d..000000000 --- a/crates/crab-cell-runtime/src/migrations/queue.sql +++ /dev/null @@ -1,78 +0,0 @@ --- Queue schema version 1. state: ready=0, leased=1, acked=2, dead=3. -CREATE TABLE queue_messages ( - message_id BLOB PRIMARY KEY CHECK (length(message_id) = 16), - payload BLOB NOT NULL CHECK (length(payload) <= 262144), - state INTEGER NOT NULL CHECK (state BETWEEN 0 AND 3), - attempt INTEGER NOT NULL CHECK (attempt >= 0), - due_at_ms INTEGER NOT NULL, - expires_at_ms INTEGER NOT NULL, - token BLOB, - lease_until_ms INTEGER, - result_code INTEGER, - dead_letter_effect_id BLOB CHECK (dead_letter_effect_id IS NULL OR length(dead_letter_effect_id) = 32), - CHECK ((state = 1 AND token IS NOT NULL AND length(token) = 16 AND lease_until_ms IS NOT NULL) - OR (state != 1 AND token IS NULL AND lease_until_ms IS NULL)), - CHECK (dead_letter_effect_id IS NULL OR state = 3) -) STRICT, WITHOUT ROWID; -CREATE INDEX queue_ready ON queue_messages(state, due_at_ms, message_id); -CREATE INDEX queue_attempts ON queue_messages(state, attempt); -CREATE INDEX queue_leases ON queue_messages(state, lease_until_ms, message_id); -CREATE INDEX queue_retention ON queue_messages(expires_at_ms); - -CREATE TABLE queue_dedup ( - producer_id BLOB PRIMARY KEY CHECK (length(producer_id) = 16), - payload_digest BLOB NOT NULL CHECK (length(payload_digest) = 32), - message_id BLOB NOT NULL CHECK (length(message_id) = 16), - retain_until_ms INTEGER NOT NULL -) STRICT, WITHOUT ROWID; -CREATE INDEX queue_dedup_expiry ON queue_dedup(retain_until_ms); - -CREATE TABLE queue_control ( - singleton INTEGER PRIMARY KEY CHECK (singleton = 1), - paused INTEGER NOT NULL CHECK (paused IN (0, 1)), - generation INTEGER NOT NULL CHECK (generation >= 0), - updated_at_ms INTEGER NOT NULL CHECK (updated_at_ms >= 0), - ready_count INTEGER NOT NULL CHECK (ready_count >= 0), - leased_count INTEGER NOT NULL CHECK (leased_count >= 0), - acked_count INTEGER NOT NULL CHECK (acked_count >= 0), - dead_count INTEGER NOT NULL CHECK (dead_count >= 0) -) STRICT; -INSERT INTO queue_control( - singleton, paused, generation, updated_at_ms, - ready_count, leased_count, acked_count, dead_count -) -VALUES (1, 0, 0, 0, 0, 0, 0, 0); - -CREATE TRIGGER queue_messages_count_insert -AFTER INSERT ON queue_messages -BEGIN - UPDATE queue_control - SET ready_count = ready_count + (NEW.state = 0), - leased_count = leased_count + (NEW.state = 1), - acked_count = acked_count + (NEW.state = 2), - dead_count = dead_count + (NEW.state = 3) - WHERE singleton = 1; -END; - -CREATE TRIGGER queue_messages_count_delete -AFTER DELETE ON queue_messages -BEGIN - UPDATE queue_control - SET ready_count = ready_count - (OLD.state = 0), - leased_count = leased_count - (OLD.state = 1), - acked_count = acked_count - (OLD.state = 2), - dead_count = dead_count - (OLD.state = 3) - WHERE singleton = 1; -END; - -CREATE TRIGGER queue_messages_count_state_update -AFTER UPDATE OF state ON queue_messages -WHEN OLD.state <> NEW.state -BEGIN - UPDATE queue_control - SET ready_count = ready_count - (OLD.state = 0) + (NEW.state = 0), - leased_count = leased_count - (OLD.state = 1) + (NEW.state = 1), - acked_count = acked_count - (OLD.state = 2) + (NEW.state = 2), - dead_count = dead_count - (OLD.state = 3) + (NEW.state = 3) - WHERE singleton = 1; -END; diff --git a/crates/crab-cell-runtime/src/migrations/runtime.sql b/crates/crab-cell-runtime/src/migrations/runtime.sql deleted file mode 100644 index 8159433a4..000000000 --- a/crates/crab-cell-runtime/src/migrations/runtime.sql +++ /dev/null @@ -1,58 +0,0 @@ --- Runtime schema version 1. Install in every Cell before its primitive schema. -PRAGMA foreign_keys = ON; - -CREATE TABLE sys_meta ( - singleton INTEGER PRIMARY KEY CHECK (singleton = 1), - cell_id BLOB NOT NULL CHECK (length(cell_id) = 32), - incarnation BLOB NOT NULL CHECK (length(incarnation) = 16), - commit_sequence INTEGER NOT NULL CHECK (commit_sequence >= 0), - logical_time_ms INTEGER NOT NULL CHECK (logical_time_ms >= 0), - schema_version INTEGER NOT NULL CHECK (schema_version >= 1) -) STRICT; - -CREATE TABLE sys_requests ( - request_id BLOB PRIMARY KEY CHECK (length(request_id) = 16), - operation_digest BLOB NOT NULL CHECK (length(operation_digest) = 32), - outcome INTEGER NOT NULL CHECK (outcome IN (1, 2)), - result BLOB NOT NULL, - commit_sequence INTEGER NOT NULL CHECK (commit_sequence > 0), - expires_at_ms INTEGER NOT NULL, - retain_until_ms INTEGER NOT NULL CHECK (retain_until_ms >= expires_at_ms) -) STRICT, WITHOUT ROWID; -CREATE INDEX sys_requests_expiry ON sys_requests(retain_until_ms); - -CREATE TABLE sys_inbox ( - effect_id BLOB PRIMARY KEY CHECK (length(effect_id) = 32), - operation_digest BLOB NOT NULL CHECK (length(operation_digest) = 32), - outcome INTEGER NOT NULL CHECK (outcome IN (1, 2)), - result BLOB NOT NULL, - commit_sequence INTEGER NOT NULL, - expires_at_ms INTEGER NOT NULL, - retain_until_ms INTEGER NOT NULL CHECK (retain_until_ms >= expires_at_ms) -) STRICT, WITHOUT ROWID; -CREATE INDEX sys_inbox_expiry ON sys_inbox(retain_until_ms); - -CREATE TABLE sys_effects ( - effect_id BLOB PRIMARY KEY CHECK (length(effect_id) = 32), - destination BLOB NOT NULL CHECK (length(destination) = 32), - operation BLOB NOT NULL, - state INTEGER NOT NULL CHECK (state IN (0, 1, 2, 3)), - attempt INTEGER NOT NULL CHECK (attempt >= 0), - due_at_ms INTEGER NOT NULL, - expires_at_ms INTEGER NOT NULL CHECK (expires_at_ms >= due_at_ms), - token BLOB, - lease_until_ms INTEGER, - created_sequence INTEGER NOT NULL, - result BLOB, - CHECK ((state = 1 AND token IS NOT NULL AND length(token) = 16 AND lease_until_ms IS NOT NULL) - OR (state != 1 AND token IS NULL AND lease_until_ms IS NULL)) -) STRICT, WITHOUT ROWID; -CREATE INDEX sys_effects_due ON sys_effects(state, due_at_ms); -CREATE INDEX sys_effects_leases ON sys_effects(state, lease_until_ms); -CREATE INDEX sys_effects_expiry ON sys_effects(expires_at_ms); - -CREATE TABLE sys_migrations ( - version INTEGER PRIMARY KEY CHECK (version > 0), - digest BLOB NOT NULL CHECK (length(digest) = 32), - applied_sequence INTEGER NOT NULL CHECK (applied_sequence >= 0) -) STRICT; diff --git a/crates/crab-cell-runtime/src/migrations/workflow.sql b/crates/crab-cell-runtime/src/migrations/workflow.sql deleted file mode 100644 index 4b5facc1d..000000000 --- a/crates/crab-cell-runtime/src/migrations/workflow.sql +++ /dev/null @@ -1,82 +0,0 @@ --- Workflow schema version 1. All identity is qualified by run_id. -CREATE TABLE workflow_runs ( - workflow_id BLOB PRIMARY KEY CHECK (length(workflow_id) BETWEEN 1 AND 1024), - run_id BLOB NOT NULL UNIQUE CHECK (length(run_id) = 16), - definition_digest BLOB NOT NULL CHECK (length(definition_digest) = 32), - status INTEGER NOT NULL CHECK (status BETWEEN 0 AND 4), - state BLOB NOT NULL CHECK (length(state) <= 1048576), - event_sequence INTEGER NOT NULL CHECK (event_sequence >= 0), - result BLOB, - completed_at_ms INTEGER -) STRICT, WITHOUT ROWID; - -CREATE TABLE workflow_events ( - run_id BLOB NOT NULL REFERENCES workflow_runs(run_id), - sequence INTEGER NOT NULL CHECK (sequence > 0), - event_id BLOB NOT NULL CHECK (length(event_id) = 32), - event_digest BLOB NOT NULL CHECK (length(event_digest) = 32), - payload BLOB NOT NULL, - PRIMARY KEY (run_id, sequence), - UNIQUE (run_id, event_id) -) STRICT, WITHOUT ROWID; - -CREATE TABLE workflow_control ( - singleton INTEGER PRIMARY KEY CHECK (singleton = 1), - event_count INTEGER NOT NULL CHECK (event_count >= 0) -) STRICT; -INSERT INTO workflow_control(singleton, event_count) VALUES (1, 0); - -CREATE TRIGGER workflow_events_count_insert -AFTER INSERT ON workflow_events -BEGIN - UPDATE workflow_control - SET event_count = event_count + 1 - WHERE singleton = 1; -END; - -CREATE TRIGGER workflow_events_count_delete -AFTER DELETE ON workflow_events -BEGIN - UPDATE workflow_control - SET event_count = event_count - 1 - WHERE singleton = 1; -END; - -CREATE TABLE workflow_activities ( - run_id BLOB NOT NULL REFERENCES workflow_runs(run_id), - activity_id BLOB NOT NULL CHECK (length(activity_id) = 16), - activity_type TEXT NOT NULL, - input BLOB NOT NULL CHECK (length(input) <= 262144), - state INTEGER NOT NULL CHECK (state BETWEEN 0 AND 4), - attempt INTEGER NOT NULL CHECK (attempt >= 0), - due_at_ms INTEGER NOT NULL, - expires_at_ms INTEGER NOT NULL, - token BLOB, - lease_until_ms INTEGER, - completion_token BLOB, - completion_digest BLOB, - result BLOB, - PRIMARY KEY (run_id, activity_id), - CHECK ((state = 1 AND token IS NOT NULL AND length(token) = 16 AND lease_until_ms IS NOT NULL) - OR (state != 1 AND token IS NULL AND lease_until_ms IS NULL)), - CHECK ((completion_token IS NULL AND completion_digest IS NULL) - OR (completion_token IS NOT NULL AND completion_digest IS NOT NULL - AND length(completion_token) = 16 AND length(completion_digest) = 32)) -) STRICT, WITHOUT ROWID; -CREATE INDEX activities_ready - ON workflow_activities(activity_type, state, due_at_ms, run_id, activity_id); -CREATE INDEX activities_due - ON workflow_activities(state, due_at_ms, run_id, activity_id); -CREATE INDEX activities_leases ON workflow_activities(state, lease_until_ms); -CREATE INDEX activities_expiry ON workflow_activities(expires_at_ms); - -CREATE TABLE workflow_timers ( - run_id BLOB NOT NULL REFERENCES workflow_runs(run_id), - timer_id BLOB NOT NULL CHECK (length(timer_id) = 16), - due_at_ms INTEGER NOT NULL, - state INTEGER NOT NULL CHECK (state IN (0, 1, 2)), - PRIMARY KEY (run_id, timer_id) -) STRICT, WITHOUT ROWID; -CREATE INDEX timers_due ON workflow_timers(state, due_at_ms, run_id, timer_id); -CREATE INDEX workflow_retention ON workflow_runs(completed_at_ms) - WHERE status BETWEEN 1 AND 3; diff --git a/crates/crab-cell-runtime/src/node.rs b/crates/crab-cell-runtime/src/node.rs deleted file mode 100644 index 278424a58..000000000 --- a/crates/crab-cell-runtime/src/node.rs +++ /dev/null @@ -1,61 +0,0 @@ -//! Node advertisements, capacity, the node directory, leases, and the durability log. -pub mod durability; -pub mod lease; -pub mod log; -pub mod log_recovery; -pub mod log_shipper; -pub mod log_state; -pub mod log_transport; - -use std::{ - collections::{BTreeSet, HashSet}, - sync::Arc, -}; - -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::{ETag, StorageError, map_object_store_error}; -use ed25519_dalek::{Signature, Signer, SigningKey, Verifier, VerifyingKey}; -use futures_util::StreamExt; -use serde::{Deserialize, Serialize}; -use tokio::sync::RwLock; - -use crate::fleet::placement::{PlacementObservation, PlacementPlanner, PlacementScore}; -use crate::identity::NodeId; -use crate::identity::{Digest, SessionId}; -use crate::node::log::NodeLogRotationBarrier; -use crate::node::log_state::{NodeLogPhase, NodeLogStatus, NodeRecoveryClaim}; -use crate::{Error, Result}; - -const MAX_NODE_BYTES: u64 = 64 * 1024; -const MAX_ENDPOINT_BYTES: usize = 512; -const MAX_FAILURE_DOMAIN_BYTES: usize = 253; -const MAX_MODULES: usize = 128; -const MAX_PEER_VERSIONS: usize = 16; -const MAX_ADVERTISEMENT_LIFETIME_MS: i64 = 15_000; -const MAX_CLOCK_SKEW_MS: i64 = 5 * 60_000; -const STALE_ADVERTISEMENT_RETENTION_MS: i64 = MAX_CLOCK_SKEW_MS + MAX_ADVERTISEMENT_LIFETIME_MS; -const MAX_STALE_COLLECTION_ITEMS: usize = 1_024; -const MAX_LIVE_NODE_RECORDS: usize = 10_000; -const NODE_DIRECTORY_READ_CONCURRENCY: usize = 32; -const RECOVERY_SCAN_CACHE_TTL_MS: i64 = 1_000; -const SIGNING_DOMAIN: &[u8] = b"crab.node.v1\0"; -const PLACEMENT_SIGNING_DOMAIN: &[u8] = b"crab.node-placement.v1\0"; -const NODE_LOG_SELECTION_DOMAIN: &[u8] = b"crab.node-log.member.v1\0"; -const RECOVERY_CANDIDATE_ROTATION_DOMAIN: &[u8] = b"crab.node-recovery-candidate.v1\0"; -const PLACEMENT_SCHEMA_VERSION: u32 = 2; - -/// Current private follower-log wire and persistence protocol. -pub const NODE_LOG_PROTOCOL_VERSION: u32 = 1; -pub use advertisement::{ - FencedNodeSession, NodeAdvertisement, NodeTakeoverProof, SealedNodeLog, - VersionedNodeAdvertisement, -}; -pub use capacity::{NodeCapacity, NodeFailureDomain, NodePlacementCapacity}; -pub use directory::{EnrolledPeerVerifier, NodeDirectory}; -mod advertisement; -mod capacity; -mod directory; - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/node/advertisement.rs b/crates/crab-cell-runtime/src/node/advertisement.rs deleted file mode 100644 index be075aac6..000000000 --- a/crates/crab-cell-runtime/src/node/advertisement.rs +++ /dev/null @@ -1,724 +0,0 @@ -//! Signed node advertisements, leases, proofs, and their wire payloads. -use crate::node::directory::NodeTombstone; - -use super::*; - -mod codec; - -pub(super) use codec::*; - -/// Signed, short-lived identity and capacity advertisement for one node session. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct NodeAdvertisement { - pub(super) node: NodeId, - pub(super) session: SessionId, - pub(super) endpoint: String, - pub(super) fleet: Digest, - pub(super) certificate: Digest, - pub(super) image: Digest, - pub(super) release: Digest, - pub(super) public_key: [u8; 32], - pub(super) generation: u64, - pub(super) progress: u64, - pub(super) issued_at_ms: i64, - pub(super) expires_at_ms: i64, - pub(super) module_digests: Vec, - pub(super) peer_versions: Vec, - pub(super) failure_domain: NodeFailureDomain, - pub(super) capacity: NodeCapacity, - pub(super) log: Option, - pub(super) signature: [u8; 64], - pub(super) placement_version: u32, - pub(super) placement_signature: [u8; 64], - pub(super) placement: Option, -} - -impl NodeAdvertisement { - /// Builds and signs a canonical advertisement with the boot-session key. - #[expect( - clippy::too_many_arguments, - reason = "all signed identity fields stay explicit" - )] - pub fn sign( - node: NodeId, - session: SessionId, - endpoint: String, - fleet: Digest, - certificate: Digest, - image: Digest, - release: Digest, - signing_key: &SigningKey, - progress: u64, - issued_at_ms: i64, - expires_at_ms: i64, - module_digests: Vec, - peer_versions: Vec, - failure_domain: NodeFailureDomain, - capacity: NodeCapacity, - ) -> Result { - let mut advertisement = Self { - node, - session, - endpoint, - fleet, - certificate, - image, - release, - public_key: signing_key.verifying_key().to_bytes(), - generation: 1, - progress, - issued_at_ms, - expires_at_ms, - module_digests, - peer_versions, - failure_domain, - capacity, - log: None, - signature: [0; 64], - placement_version: 0, - placement_signature: [0; 64], - placement: None, - }; - advertisement.validate_shape()?; - advertisement.signature = signing_key.sign(&advertisement.signing_bytes()?).to_bytes(); - Ok(advertisement) - } - - /// Returns the node identity this advertisement describes. - #[must_use] - pub const fn node(&self) -> NodeId { - self.node - } - - /// Returns the boot session that signed the advertisement. - #[must_use] - pub const fn session(&self) -> SessionId { - self.session - } - - /// Returns the peer endpoint the node advertises. - #[must_use] - pub fn endpoint(&self) -> &str { - &self.endpoint - } - - /// Returns the fleet the node belongs to. - #[must_use] - pub const fn fleet(&self) -> Digest { - self.fleet - } - - /// Returns the certificate the node authenticates with. - #[must_use] - pub const fn certificate(&self) -> Digest { - self.certificate - } - - /// Returns the compiled image the node runs. - #[must_use] - pub const fn image(&self) -> Digest { - self.image - } - - /// Returns the release the node runs. - #[must_use] - pub const fn release(&self) -> Digest { - self.release - } - - /// Returns the key that verifies this advertisement's signatures. - pub fn verifying_key(&self) -> Result { - VerifyingKey::from_bytes(&self.public_key).map_err(Error::PeerSignature) - } - - /// Returns the recovery progress watermark the node reports. - #[must_use] - pub const fn progress(&self) -> u64 { - self.progress - } - - /// Returns the generation this advertisement replaced. - #[must_use] - pub const fn generation(&self) -> u64 { - self.generation - } - - /// Returns the logical time the advertisement stops being valid. - #[must_use] - pub const fn expires_at_ms(&self) -> i64 { - self.expires_at_ms - } - - /// Returns the logical time the advertisement was signed. - #[must_use] - pub const fn issued_at_ms(&self) -> i64 { - self.issued_at_ms - } - - /// Returns the module digests the running image declares. - #[must_use] - pub fn module_digests(&self) -> &[Digest] { - &self.module_digests - } - - /// Returns the peer protocol versions the node accepts. - #[must_use] - pub fn peer_versions(&self) -> &[u32] { - &self.peer_versions - } - - /// Returns the failure domain the node reports for placement. - #[must_use] - pub const fn failure_domain(&self) -> &NodeFailureDomain { - &self.failure_domain - } - - /// Returns the signed capacity block, when the node publishes one. - #[must_use] - pub const fn capacity(&self) -> NodeCapacity { - self.capacity - } - - /// Returns the optional signed runtime snapshot used for placement. - #[must_use] - pub const fn placement_capacity(&self) -> Option { - self.placement - } - - /// Replaces the signed runtime snapshot after the caller has measured the - /// node-wide ledger and local disk envelope. - pub fn with_placement_capacity( - mut self, - placement: NodePlacementCapacity, - signing_key: &SigningKey, - ) -> Result { - placement.validated()?; - self.placement = Some(placement); - self.placement_version = PLACEMENT_SCHEMA_VERSION; - self.placement_signature = signing_key - .sign(&self.placement_signing_bytes()?) - .to_bytes(); - Ok(self) - } - - /// Reports whether this advertisement carries an authenticated placement - /// snapshot. Legacy identity-only records remain readable but are never - /// eligible for weighted ownership placement. - #[must_use] - pub fn has_signed_placement(&self) -> bool { - self.placement_version == PLACEMENT_SCHEMA_VERSION - && self.placement.is_some() - && self.placement_signature.iter().any(|byte| *byte != 0) - } - - /// Returns the node-log status, when the node has an enrolled log. - #[must_use] - pub const fn log(&self) -> Option<&NodeLogStatus> { - self.log.as_ref() - } - - pub(super) fn encode(&self) -> Result> { - self.validate_shape()?; - self.verify_signature()?; - let encoded = serde_json::to_vec(&RawAdvertisement::from(self))?; - if encoded.len() as u64 > MAX_NODE_BYTES { - return Err(Error::Node("advertisement exceeds 64 KiB")); - } - Ok(encoded) - } - - pub(super) fn decode_canonical(bytes: &[u8]) -> Result { - if bytes.len() as u64 > MAX_NODE_BYTES { - return Err(Error::Node("advertisement exceeds 64 KiB")); - } - let raw: RawAdvertisement = serde_json::from_slice(bytes)?; - let advertisement = Self::try_from(raw)?; - advertisement.validate_shape()?; - advertisement.verify_signature()?; - if advertisement.encode()?.as_slice() != bytes { - return Err(Error::Node("advertisement JSON is not canonical")); - } - Ok(advertisement) - } - - pub(super) fn validate_at(&self, now_ms: i64) -> Result<()> { - self.validate_shape()?; - if self.issued_at_ms > now_ms.saturating_add(MAX_CLOCK_SKEW_MS) - || self.expires_at_ms <= now_ms - { - return Err(Error::Node("advertisement is not currently valid")); - } - Ok(()) - } - - pub(super) fn validate_shape(&self) -> Result<()> { - if self.node.as_bytes().iter().all(|byte| *byte == 0) - || self.session.as_bytes().iter().all(|byte| *byte == 0) - || self.fleet.as_bytes().iter().all(|byte| *byte == 0) - || self.certificate.as_bytes().iter().all(|byte| *byte == 0) - || self.image.as_bytes().iter().all(|byte| *byte == 0) - || self.release.as_bytes().iter().all(|byte| *byte == 0) - || self.public_key.iter().all(|byte| *byte == 0) - { - return Err(Error::Node("advertisement identity is zero")); - } - if !valid_endpoint(&self.endpoint) { - return Err(Error::Node("advertisement endpoint is invalid")); - } - if self.generation == 0 - || self.progress == 0 - || self.issued_at_ms < 0 - || self.expires_at_ms <= self.issued_at_ms - || self.expires_at_ms.saturating_sub(self.issued_at_ms) > MAX_ADVERTISEMENT_LIFETIME_MS - { - return Err(Error::Node("advertisement progress or time is invalid")); - } - if let Some(log) = &self.log { - log.validate(self.node)?; - if log.phase() != NodeLogPhase::Open || log.recovery().is_some() { - return Err(Error::Node("live node advertisement has a terminal log")); - } - } - self.failure_domain.validate()?; - if self.capacity.follower_free_bytes > self.capacity.free_disk_bytes - || (self.capacity.log_protocol == 0 && self.capacity.follower_free_bytes != 0) - { - return Err(Error::Node("advertisement follower capacity is invalid")); - } - if let Some(placement) = self.placement { - placement.validated()?; - } - if self.module_digests.is_empty() - || self.module_digests.len() > MAX_MODULES - || !self - .module_digests - .windows(2) - .all(|pair| pair[0].as_bytes() < pair[1].as_bytes()) - || self.peer_versions.is_empty() - || self.peer_versions.len() > MAX_PEER_VERSIONS - || !strictly_sorted(&self.peer_versions) - || self.peer_versions.iter().any(|version| *version != 1) - { - return Err(Error::Node("advertisement inventory is invalid")); - } - Ok(()) - } - - pub(super) fn verify_signature(&self) -> Result<()> { - self.verifying_key()? - .verify( - &self.signing_bytes()?, - &Signature::from_bytes(&self.signature), - ) - .map_err(Error::PeerSignature)?; - match self.placement_version { - 0 => { - if self.placement_signature.iter().any(|byte| *byte != 0) { - return Err(Error::Node("legacy placement record carries a signature")); - } - } - PLACEMENT_SCHEMA_VERSION => { - if !self.has_signed_placement() { - return Err(Error::Node("placement signature is missing")); - } - self.verifying_key()? - .verify( - &self.placement_signing_bytes()?, - &Signature::from_bytes(&self.placement_signature), - ) - .map_err(Error::PeerSignature)?; - } - _ => { - // Unknown placement schemas remain readable for identity and - // liveness. They are never eligible for placement until this - // binary understands and verifies their signed fields. - } - } - Ok(()) - } - - pub(super) fn signing_bytes(&self) -> Result> { - let unsigned = serde_json::to_vec(&RawNodeSigningPayload { - identity: RawUnsignedIdentity::from(self), - capacity: RawCapacity::from(self.capacity), - })?; - let mut bytes = Vec::with_capacity(SIGNING_DOMAIN.len() + unsigned.len()); - bytes.extend_from_slice(SIGNING_DOMAIN); - bytes.extend_from_slice(&unsigned); - Ok(bytes) - } - - pub(super) fn placement_signing_bytes(&self) -> Result> { - let mut unsigned = RawAdvertisement::from(self); - unsigned.placement_signature = None; - unsigned.log = None; - // The directory assigns the monotonic heartbeat generation during a - // CAS refresh; the signed lease timestamps/progress remain immutable - // evidence while this server-owned counter is intentionally excluded. - unsigned.lease.generation.clear(); - let unsigned = serde_json::to_vec(&unsigned)?; - let mut bytes = Vec::with_capacity(PLACEMENT_SIGNING_DOMAIN.len() + unsigned.len()); - bytes.extend_from_slice(PLACEMENT_SIGNING_DOMAIN); - bytes.extend_from_slice(&unsigned); - Ok(bytes) - } -} - -/// Exact node advertisement plus the token required for conditional refresh. -#[derive(Clone)] -pub struct VersionedNodeAdvertisement { - pub(super) advertisement: NodeAdvertisement, - pub(super) token: ETag, -} - -/// Proof that the exact predecessor session was atomically fenced after expiry. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct FencedNodeSession { - pub(super) node: NodeId, - pub(super) session: SessionId, - pub(super) claimant: SessionId, - pub(super) claim_generation: u64, - pub(super) claim_expires_at_ms: i64, - pub(super) log: Option, -} - -impl FencedNodeSession { - /// Returns the node whose session was fenced. - #[must_use] - pub const fn node(&self) -> NodeId { - self.node - } - - /// Returns the predecessor session that was fenced. - #[must_use] - pub const fn session(&self) -> SessionId { - self.session - } - - /// Returns the session that fenced it. - #[must_use] - pub const fn claimant(&self) -> SessionId { - self.claimant - } - - /// Returns the claim generation the fence was issued under. - #[must_use] - pub const fn claim_generation(&self) -> u64 { - self.claim_generation - } - - /// Returns the logical time the claim expires. - #[must_use] - pub const fn claim_expires_at_ms(&self) -> i64 { - self.claim_expires_at_ms - } - - /// Returns the node-log status observed while fencing. - #[must_use] - pub const fn log(&self) -> Option<&NodeLogStatus> { - self.log.as_ref() - } - - /// Converts a fence into takeover authority when no fleet proof needs recovery. - pub fn direct_takeover(&self) -> Result { - if self.log.as_ref().is_some_and(NodeLogStatus::active) { - return Err(Error::PendingPublication); - } - Ok(NodeTakeoverProof { - session: self.session, - claimant: self.claimant, - }) - } -} - -/// Proof that predecessor node-log recovery cannot add newer durable Cell state. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct NodeTakeoverProof { - pub(super) session: SessionId, - pub(super) claimant: SessionId, -} - -impl NodeTakeoverProof { - /// Returns the predecessor session this proof covers. - #[must_use] - pub const fn session(&self) -> SessionId { - self.session - } - - /// Returns the session permitted to take over. - #[must_use] - pub const fn claimant(&self) -> SessionId { - self.claimant - } - - pub(crate) fn after_recovery( - fenced: &FencedNodeSession, - sealed: &SealedNodeLog, - ) -> Result { - let recovering = fenced.log.as_ref().ok_or(Error::Fenced)?; - let sealed_log = sealed.log(); - if sealed.session() != fenced.session - || sealed_log.phase() != NodeLogPhase::Sealed - || sealed_log.epoch() != recovering.epoch() - || sealed_log.members() != recovering.members() - || sealed_log.active() != recovering.active() - || sealed_log.tiered_through() != recovering.tiered_through() - { - return Err(Error::Fenced); - } - Ok(Self { - session: fenced.session, - claimant: fenced.claimant, - }) - } -} - -/// Proof that one failed node log has finished pinning every recovered tail. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct SealedNodeLog { - pub(super) session: SessionId, - pub(super) log: NodeLogStatus, -} - -impl SealedNodeLog { - /// Returns the session whose log was sealed. - #[must_use] - pub const fn session(&self) -> SessionId { - self.session - } - - /// Returns the status of the sealed log. - #[must_use] - pub const fn log(&self) -> &NodeLogStatus { - &self.log - } -} - -impl VersionedNodeAdvertisement { - /// Returns the decoded advertisement. - #[must_use] - pub const fn advertisement(&self) -> &NodeAdvertisement { - &self.advertisement - } -} - -pub(super) fn validate_successor( - current: &NodeAdvertisement, - next: &NodeAdvertisement, -) -> Result<()> { - if !same_boot_identity(current, next) - || current.log != next.log - || current.generation.checked_add(1) != Some(next.generation) - || next.progress < current.progress - || next.issued_at_ms <= current.issued_at_ms - || next.expires_at_ms <= current.expires_at_ms - { - return Err(Error::Node( - "advertisement refresh changed boot identity or regressed", - )); - } - Ok(()) -} - -pub(super) fn same_boot_identity(current: &NodeAdvertisement, next: &NodeAdvertisement) -> bool { - // Capacity is a fresh signed heartbeat measurement, so its signature may - // change without allowing the boot identity or signing key to change. - current.node == next.node - && current.session == next.session - && current.endpoint == next.endpoint - && current.fleet == next.fleet - && current.certificate == next.certificate - && current.image == next.image - && current.release == next.release - && current.public_key == next.public_key - && current.module_digests == next.module_digests - && current.peer_versions == next.peer_versions - && current.failure_domain == next.failure_domain -} - -pub(super) fn valid_endpoint(endpoint: &str) -> bool { - if endpoint.len() > MAX_ENDPOINT_BYTES - || !endpoint.is_ascii() - || endpoint - .bytes() - .any(|byte| byte.is_ascii_control() || byte == b' ') - || endpoint.contains(['@', '?', '#']) - { - return false; - } - let Some(authority) = endpoint.strip_prefix("https://") else { - return false; - }; - let authority = authority.strip_suffix('/').unwrap_or(authority); - if authority.is_empty() || authority.contains('/') { - return false; - } - let (host, port) = match authority.strip_prefix('[') { - Some(bracketed) => { - let Some((host, suffix)) = bracketed.split_once(']') else { - return false; - }; - if host.parse::().is_err() { - return false; - } - let port = if suffix.is_empty() { - None - } else { - let Some(port) = suffix.strip_prefix(':') else { - return false; - }; - Some(port) - }; - (None, port) - } - None => match authority.rsplit_once(':') { - Some((host, port)) => (Some(host), Some(port)), - None => (Some(authority), None), - }, - }; - if host.is_some_and(|host| !valid_host(host)) { - return false; - } - port.is_none_or(|port| { - port.parse::() - .is_ok_and(|parsed| parsed != 0 && parsed.to_string() == port) - }) -} - -pub(super) fn valid_host(host: &str) -> bool { - if host.parse::().is_ok() { - return true; - } - !host.is_empty() - && host.len() <= 253 - && host.split('.').all(|label| { - !label.is_empty() - && label.len() <= 63 - && label - .bytes() - .all(|byte| byte.is_ascii_alphanumeric() || byte == b'-') - && label - .as_bytes() - .first() - .is_some_and(u8::is_ascii_alphanumeric) - && label - .as_bytes() - .last() - .is_some_and(u8::is_ascii_alphanumeric) - }) -} - -pub(super) fn strictly_sorted(values: &[T]) -> bool { - values.windows(2).all(|pair| pair[0] < pair[1]) -} - -pub(super) fn encode_log(log: &NodeLogStatus) -> RawNodeLog { - RawNodeLog { - state: match log.phase() { - NodeLogPhase::Open => RawNodeLogPhase::Open, - NodeLogPhase::Recovering => RawNodeLogPhase::Recovering, - NodeLogPhase::Sealed => RawNodeLogPhase::Sealed, - NodeLogPhase::Retired => RawNodeLogPhase::Retired, - }, - epoch: log.epoch().to_string(), - members: log - .members() - .iter() - .map(|member| encode_hex(member.as_bytes())) - .collect(), - active: log.active(), - tiered_through: log.tiered_through().to_string(), - recovery: log.recovery().map(|claim| RawNodeRecoveryClaim { - claimant: encode_hex(claim.claimant().as_bytes()), - generation: claim.generation().to_string(), - expires_at_ms: claim.expires_at_ms().to_string(), - }), - recovery_manifest: log - .recovery_manifest() - .map(|digest| encode_hex(digest.as_bytes())), - } -} - -pub(super) fn decode_log(leader: NodeId, raw: RawNodeLog) -> Result { - let recovery = raw - .recovery - .map(|claim| { - NodeRecoveryClaim::from_parts( - SessionId::from_bytes(decode_hex(&claim.claimant)?), - canonical_u64(&claim.generation)?, - canonical_i64(&claim.expires_at_ms)?, - ) - }) - .transpose()?; - NodeLogStatus::from_parts( - leader, - match raw.state { - RawNodeLogPhase::Open => NodeLogPhase::Open, - RawNodeLogPhase::Recovering => NodeLogPhase::Recovering, - RawNodeLogPhase::Sealed => NodeLogPhase::Sealed, - RawNodeLogPhase::Retired => NodeLogPhase::Retired, - }, - canonical_u64(&raw.epoch)?, - raw.members - .iter() - .map(|member| decode_hex(member).map(NodeId::from_bytes)) - .collect::>>()?, - raw.active, - canonical_u64(&raw.tiered_through)?, - recovery, - raw.recovery_manifest - .map(|digest| decode_hex(&digest).map(Digest::from_bytes)) - .transpose()?, - ) -} - -pub(super) fn canonical_u64(value: &str) -> Result { - let parsed = value - .parse::() - .map_err(|_| Error::Node("invalid unsigned decimal"))?; - if parsed.to_string() != value { - return Err(Error::Node("unsigned decimal is not canonical")); - } - Ok(parsed) -} - -pub(super) fn canonical_i64(value: &str) -> Result { - let parsed = value - .parse::() - .map_err(|_| Error::Node("invalid signed decimal"))?; - if parsed.to_string() != value { - return Err(Error::Node("signed decimal is not canonical")); - } - Ok(parsed) -} - -pub(super) fn encode_hex(bytes: &[u8]) -> String { - pub(super) const TABLE: &[u8; 16] = b"0123456789abcdef"; - let mut encoded = String::with_capacity(bytes.len() * 2); - for byte in bytes { - encoded.push(TABLE[(byte >> 4) as usize] as char); - encoded.push(TABLE[(byte & 0x0f) as usize] as char); - } - encoded -} - -pub(super) fn decode_hex(value: &str) -> Result<[u8; N]> { - if value.len() != N * 2 { - return Err(Error::Node("hex field length is invalid")); - } - let mut decoded = [0; N]; - for (index, pair) in value.as_bytes().as_chunks::<2>().0.iter().enumerate() { - let high = nibble(pair[0]).ok_or(Error::Node("hex field is not lowercase"))?; - let low = nibble(pair[1]).ok_or(Error::Node("hex field is not lowercase"))?; - decoded[index] = (high << 4) | low; - } - Ok(decoded) -} - -pub(super) fn nibble(byte: u8) -> Option { - match byte { - b'0'..=b'9' => Some(byte - b'0'), - b'a'..=b'f' => Some(byte - b'a' + 10), - _ => None, - } -} diff --git a/crates/crab-cell-runtime/src/node/advertisement/codec.rs b/crates/crab-cell-runtime/src/node/advertisement/codec.rs deleted file mode 100644 index 6d97d0f69..000000000 --- a/crates/crab-cell-runtime/src/node/advertisement/codec.rs +++ /dev/null @@ -1,311 +0,0 @@ -//! Strict JSON wire payloads for node advertisements and tombstones. - -use super::*; - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawUnsignedIdentity { - pub(in crate::node) node: String, - pub(in crate::node) session: String, - pub(in crate::node) endpoint: String, - pub(in crate::node) fleet: String, - pub(in crate::node) certificate: String, - pub(in crate::node) image: String, - pub(in crate::node) release: String, - pub(in crate::node) public_key: String, - pub(in crate::node) module_digests: Vec, - pub(in crate::node) peer_versions: Vec, - pub(in crate::node) failure_domain: RawFailureDomain, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawNodeSigningPayload { - pub(in crate::node) identity: RawUnsignedIdentity, - pub(in crate::node) capacity: RawCapacity, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawFailureDomain { - pub(in crate::node) zone: Option, - pub(in crate::node) host: Option, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawNodeTombstoneEnvelope { - pub(in crate::node) tombstone: RawNodeTombstone, -} - -impl From<&NodeTombstone> for RawNodeTombstoneEnvelope { - fn from(value: &NodeTombstone) -> Self { - Self { - tombstone: RawNodeTombstone { - version: 1, - session: encode_hex(value.session.as_bytes()), - node: encode_hex(value.node.as_bytes()), - expires_at_ms: value.expires_at_ms.to_string(), - retired_at_ms: value.retired_at_ms.to_string(), - claimant: value - .claimant - .map(|claimant| encode_hex(claimant.as_bytes())), - claim_generation: value.claim_generation.to_string(), - claim_expires_at_ms: value.claim_expires_at_ms.map(|value| value.to_string()), - log: value.log.as_ref().map(encode_log), - }, - } - } -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawNodeTombstone { - pub(in crate::node) version: u8, - pub(in crate::node) session: String, - pub(in crate::node) node: String, - pub(in crate::node) expires_at_ms: String, - pub(in crate::node) retired_at_ms: String, - pub(in crate::node) claimant: Option, - pub(in crate::node) claim_generation: String, - pub(in crate::node) claim_expires_at_ms: Option, - pub(in crate::node) log: Option, -} - -impl From<&NodeAdvertisement> for RawUnsignedIdentity { - fn from(value: &NodeAdvertisement) -> Self { - Self { - node: encode_hex(value.node.as_bytes()), - session: encode_hex(value.session.as_bytes()), - endpoint: value.endpoint.clone(), - fleet: encode_hex(value.fleet.as_bytes()), - certificate: encode_hex(value.certificate.as_bytes()), - image: encode_hex(value.image.as_bytes()), - release: encode_hex(value.release.as_bytes()), - public_key: encode_hex(&value.public_key), - module_digests: value - .module_digests - .iter() - .map(|digest| encode_hex(digest.as_bytes())) - .collect(), - peer_versions: value.peer_versions.clone(), - failure_domain: RawFailureDomain { - zone: value.failure_domain.zone.clone(), - host: value.failure_domain.host.clone(), - }, - } - } -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawAdvertisement { - pub(in crate::node) version: u8, - pub(in crate::node) identity: RawIdentity, - pub(in crate::node) lease: RawLease, - pub(in crate::node) log: Option, - pub(in crate::node) capacity: RawCapacity, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub(in crate::node) placement: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub(in crate::node) placement_version: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub(in crate::node) placement_signature: Option, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawIdentity { - #[serde(flatten)] - pub(in crate::node) unsigned: RawUnsignedIdentity, - pub(in crate::node) signature: String, -} - -impl From<&NodeAdvertisement> for RawAdvertisement { - fn from(value: &NodeAdvertisement) -> Self { - Self { - version: 1, - identity: RawIdentity { - unsigned: RawUnsignedIdentity::from(value), - signature: encode_hex(&value.signature), - }, - lease: RawLease { - generation: value.generation.to_string(), - progress: value.progress.to_string(), - issued_at_ms: value.issued_at_ms.to_string(), - expires_at_ms: value.expires_at_ms.to_string(), - }, - log: value.log.as_ref().map(encode_log), - capacity: RawCapacity::from(value.capacity), - placement: value.placement.map(|placement| RawPlacementCapacity { - memory_capacity_bytes: placement.memory_capacity_bytes.to_string(), - disk_capacity_bytes: placement.disk_capacity_bytes.to_string(), - active_cells: placement.active_cells, - max_active_cells: placement.max_active_cells, - running_jobs: placement.running_jobs, - job_capacity: placement.job_capacity, - publication_backlog: (value.placement_version >= PLACEMENT_SCHEMA_VERSION) - .then_some(placement.publication_backlog), - hydration_backlog: (value.placement_version >= PLACEMENT_SCHEMA_VERSION) - .then_some(placement.hydration_backlog), - primitive_backlog: (value.placement_version >= PLACEMENT_SCHEMA_VERSION) - .then_some(placement.primitive_backlog), - }), - placement_version: (value.placement_version != 0).then_some(value.placement_version), - placement_signature: value - .placement_signature - .iter() - .any(|byte| *byte != 0) - .then(|| encode_hex(&value.placement_signature)), - } - } -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawLease { - pub(in crate::node) generation: String, - pub(in crate::node) progress: String, - pub(in crate::node) issued_at_ms: String, - pub(in crate::node) expires_at_ms: String, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawCapacity { - pub(in crate::node) free_memory_bytes: String, - pub(in crate::node) free_disk_bytes: String, - pub(in crate::node) follower_free_bytes: String, - pub(in crate::node) follower_retained_bytes: String, - pub(in crate::node) job_credits: u32, - pub(in crate::node) log_protocol: u32, -} - -impl From for RawCapacity { - fn from(value: NodeCapacity) -> Self { - Self { - free_memory_bytes: value.free_memory_bytes.to_string(), - free_disk_bytes: value.free_disk_bytes.to_string(), - follower_free_bytes: value.follower_free_bytes.to_string(), - follower_retained_bytes: value.follower_retained_bytes.to_string(), - job_credits: value.job_credits, - log_protocol: value.log_protocol, - } - } -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawPlacementCapacity { - pub(in crate::node) memory_capacity_bytes: String, - pub(in crate::node) disk_capacity_bytes: String, - pub(in crate::node) active_cells: u32, - pub(in crate::node) max_active_cells: u32, - pub(in crate::node) running_jobs: u32, - pub(in crate::node) job_capacity: u32, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub(in crate::node) publication_backlog: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub(in crate::node) hydration_backlog: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub(in crate::node) primitive_backlog: Option, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawNodeLog { - pub(in crate::node) state: RawNodeLogPhase, - pub(in crate::node) epoch: String, - pub(in crate::node) members: Vec, - pub(in crate::node) active: bool, - pub(in crate::node) tiered_through: String, - pub(in crate::node) recovery: Option, - pub(in crate::node) recovery_manifest: Option, -} - -#[derive(Deserialize, Serialize)] -#[serde(rename_all = "snake_case")] -pub(in crate::node) enum RawNodeLogPhase { - Open, - Recovering, - Sealed, - Retired, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::node) struct RawNodeRecoveryClaim { - pub(in crate::node) claimant: String, - pub(in crate::node) generation: String, - pub(in crate::node) expires_at_ms: String, -} - -impl TryFrom for NodeAdvertisement { - type Error = Error; - - fn try_from(value: RawAdvertisement) -> Result { - if value.version != 1 { - return Err(Error::Node("unsupported advertisement version")); - } - let raw = value.identity.unsigned; - let session = SessionId::from_bytes(decode_hex(&raw.session)?); - let node = NodeId::from_bytes(decode_hex(&raw.node)?); - Ok(Self { - node, - session, - endpoint: raw.endpoint, - fleet: Digest::from_bytes(decode_hex(&raw.fleet)?), - certificate: Digest::from_bytes(decode_hex(&raw.certificate)?), - image: Digest::from_bytes(decode_hex(&raw.image)?), - release: Digest::from_bytes(decode_hex(&raw.release)?), - public_key: decode_hex(&raw.public_key)?, - generation: canonical_u64(&value.lease.generation)?, - progress: canonical_u64(&value.lease.progress)?, - issued_at_ms: canonical_i64(&value.lease.issued_at_ms)?, - expires_at_ms: canonical_i64(&value.lease.expires_at_ms)?, - module_digests: raw - .module_digests - .iter() - .map(|value| decode_hex(value).map(Digest::from_bytes)) - .collect::>>()?, - peer_versions: raw.peer_versions, - failure_domain: NodeFailureDomain::new( - raw.failure_domain.zone, - raw.failure_domain.host, - )?, - capacity: NodeCapacity { - free_memory_bytes: canonical_u64(&value.capacity.free_memory_bytes)?, - free_disk_bytes: canonical_u64(&value.capacity.free_disk_bytes)?, - follower_free_bytes: canonical_u64(&value.capacity.follower_free_bytes)?, - follower_retained_bytes: canonical_u64(&value.capacity.follower_retained_bytes)?, - job_credits: value.capacity.job_credits, - log_protocol: value.capacity.log_protocol, - }, - placement: value - .placement - .map(|placement| { - NodePlacementCapacity { - memory_capacity_bytes: canonical_u64(&placement.memory_capacity_bytes)?, - disk_capacity_bytes: canonical_u64(&placement.disk_capacity_bytes)?, - active_cells: placement.active_cells, - max_active_cells: placement.max_active_cells, - running_jobs: placement.running_jobs, - job_capacity: placement.job_capacity, - publication_backlog: placement.publication_backlog.unwrap_or(0), - hydration_backlog: placement.hydration_backlog.unwrap_or(0), - primitive_backlog: placement.primitive_backlog.unwrap_or(0), - } - .validated() - }) - .transpose()?, - log: value.log.map(|log| decode_log(node, log)).transpose()?, - signature: decode_hex(&value.identity.signature)?, - placement_version: value.placement_version.unwrap_or(0), - placement_signature: value - .placement_signature - .map(|signature| decode_hex(&signature)) - .transpose()? - .unwrap_or([0; 64]), - }) - } -} diff --git a/crates/crab-cell-runtime/src/node/capacity.rs b/crates/crab-cell-runtime/src/node/capacity.rs deleted file mode 100644 index 125534754..000000000 --- a/crates/crab-cell-runtime/src/node/capacity.rs +++ /dev/null @@ -1,108 +0,0 @@ -//! Node capacity and failure-domain records. - -use super::*; - -/// Capacity hints published by one node boot session. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct NodeCapacity { - /// Free memory the node reports. - pub free_memory_bytes: u64, - /// Free scratch disk the node reports. - pub free_disk_bytes: u64, - /// Free space in the node's follower store. - pub follower_free_bytes: u64, - /// Bytes the node's follower store retains. - pub follower_retained_bytes: u64, - /// Job credits the node offers to the fleet. - pub job_credits: u32, - /// Node-log protocol version the node speaks. - pub log_protocol: u32, -} - -/// Signed runtime capacity measurements used by the placement planner. -/// -/// The ordinary capacity hints remain intentionally small and compatible with -/// older node records. This optional block carries the totals and live counts -/// required to compare a node's usable headroom without guessing from host -/// totals on the receiving side. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct NodePlacementCapacity { - /// Total memory the node reports. - pub memory_capacity_bytes: u64, - /// Total scratch disk the node reports. - pub disk_capacity_bytes: u64, - /// Cells the node currently owns. - pub active_cells: u32, - /// Cells the node admits. - pub max_active_cells: u32, - /// Jobs the node is running. - pub running_jobs: u32, - /// Jobs the node admits. - pub job_capacity: u32, - /// Publications waiting to be acknowledged. - pub publication_backlog: u32, - /// Hydrations waiting to run. - pub hydration_backlog: u32, - /// Primitive maintenance items waiting. - pub primitive_backlog: u32, -} - -impl NodePlacementCapacity { - /// Validates and returns a placement snapshot with measured node totals. - pub const fn validated(self) -> Result { - if self.memory_capacity_bytes == 0 - || self.disk_capacity_bytes == 0 - || self.max_active_cells == 0 - || self.active_cells > self.max_active_cells - || self.job_capacity == 0 - || self.running_jobs > self.job_capacity - { - return Err(Error::Node("placement capacity is invalid")); - } - Ok(self) - } -} - -/// Stable topology labels used only to prefer independent follower nodes. -#[derive(Clone, Debug, Default, PartialEq, Eq)] -pub struct NodeFailureDomain { - pub(super) zone: Option, - pub(super) host: Option, -} - -impl NodeFailureDomain { - /// Validates optional zone and host labels advertised for one boot identity. - pub fn new(zone: Option, host: Option) -> Result { - let domain = Self { zone, host }; - domain.validate()?; - Ok(domain) - } - - /// Returns the availability zone, when the node declares one. - #[must_use] - pub fn zone(&self) -> Option<&str> { - self.zone.as_deref() - } - - /// Returns the host, when the node declares one. - #[must_use] - pub fn host(&self) -> Option<&str> { - self.host.as_deref() - } - - pub(super) fn validate(&self) -> Result<()> { - if [self.zone.as_deref(), self.host.as_deref()] - .into_iter() - .flatten() - .any(|value| { - value.is_empty() - || value.len() > MAX_FAILURE_DOMAIN_BYTES - || !value.is_ascii() - || value.bytes().any(|byte| !byte.is_ascii_graphic()) - }) - { - return Err(Error::Node("node failure domain is invalid")); - } - Ok(()) - } -} diff --git a/crates/crab-cell-runtime/src/node/directory.rs b/crates/crab-cell-runtime/src/node/directory.rs deleted file mode 100644 index 513f5a3d3..000000000 --- a/crates/crab-cell-runtime/src/node/directory.rs +++ /dev/null @@ -1,655 +0,0 @@ -//! Node directory records, tombstones, and recovery-candidate windows. -use crate::node::advertisement::RawNodeTombstoneEnvelope; -use crate::node::advertisement::canonical_i64; -use crate::node::advertisement::canonical_u64; -use crate::node::advertisement::decode_hex; -use crate::node::advertisement::decode_log; -use crate::node::advertisement::same_boot_identity; -use crate::node::advertisement::validate_successor; - -use super::*; - -mod advertisement; -mod log; -mod recovery; - -/// Object-store directory for one fleet and compiled release. -#[derive(Clone)] -pub struct NodeDirectory { - pub(super) layout: CellStorageLayout, - pub(super) fleet: Digest, - pub(super) image: Digest, - pub(super) release: Digest, - // Candidate discovery is advisory; claims always reload the authoritative - // record. Sharing this short-lived snapshot keeps cloned schedulers from - // multiplying a full directory scan without changing failover authority. - pub(super) recovery_scan: Arc>>>, - // Reader placement is advisory. Authority, session authentication and - // maintenance continue to read their canonical records directly. - reader_membership: Arc>>, -} - -/// Request verifier bound to one mTLS-authenticated enrollment observation. -pub struct EnrolledPeerVerifier { - advertisement: NodeAdvertisement, - verifier: crate::peer::PeerVerifier, -} - -impl EnrolledPeerVerifier { - /// Verifies the signed request while rechecking the enrollment's lifetime. - pub fn verify( - &self, - request: crate::peer::UnverifiedPeerRequest, - now_ms: i64, - ) -> Result { - self.advertisement.validate_at(now_ms)?; - self.verifier.verify_decoded(request, now_ms) - } -} - -#[derive(Clone, Copy)] -pub(super) enum AdvertisementScan { - LiveRelease, - AdvertisedFleet, -} - -impl NodeDirectory { - /// Binds the directory to one fleet, image, and release scope. - #[must_use] - pub fn new(layout: CellStorageLayout, fleet: Digest, image: Digest, release: Digest) -> Self { - Self { - layout, - fleet, - image, - release, - recovery_scan: Arc::new(RwLock::new(None)), - reader_membership: Arc::new(RwLock::new(None)), - } - } - - /// Returns the fleet this directory is scoped to. - #[must_use] - pub const fn fleet(&self) -> Digest { - self.fleet - } - - /// Strict-creates one signed boot-session advertisement. - pub async fn create( - &self, - advertisement: NodeAdvertisement, - now_ms: i64, - ) -> Result { - self.validate(&advertisement, now_ms)?; - let encoded = advertisement.encode()?; - let path = self.layout.node_path(advertisement.session.as_bytes()); - let result = match self - .layout - .store() - .create_strict_with_etag(&path, Bytes::from(encoded)) - .await - { - Ok(token) => Ok(VersionedNodeAdvertisement { - advertisement, - token, - }), - Err(create_error) => match self.load(advertisement.session, now_ms).await? { - Some(current) if current.advertisement == advertisement => Ok(current), - Some(_) | None => Err(create_error.into()), - }, - }; - self.reader_membership.write().await.take(); - result - } - - /// Loads and verifies one exact, currently valid boot session. - pub async fn load( - &self, - session: SessionId, - now_ms: i64, - ) -> Result> { - let Some((advertisement, token)) = self.load_canonical(session).await? else { - return Ok(None); - }; - self.validate(&advertisement, now_ms)?; - Ok(Some(VersionedNodeAdvertisement { - advertisement, - token, - })) - } - - /// Inspects one signed advertisement without requiring its lease to remain live. - /// - /// This is an operational read only: callers must use [`Self::load`] or - /// [`Self::is_live`] for admission and takeover decisions. - pub async fn inspect_advertisement( - &self, - session: SessionId, - now_ms: i64, - ) -> Result> { - let Some((advertisement, _)) = self.load_canonical(session).await? else { - return Ok(None); - }; - advertisement.validate_shape()?; - advertisement.verify_signature()?; - self.validate_scope(&advertisement)?; - if advertisement.issued_at_ms > now_ms.saturating_add(MAX_CLOCK_SKEW_MS) { - return Err(Error::Node("advertisement issue time is in the future")); - } - Ok(Some(advertisement)) - } - - /// Reports whether an exact canonical session is currently live. - /// - /// Missing and expired sessions return `false`. Malformed, misplaced, or - /// foreign records fail closed instead of being treated as takeover evidence. - pub async fn is_live(&self, session: SessionId, now_ms: i64) -> Result { - let Some((advertisement, _)) = self.load_canonical(session).await? else { - return Ok(false); - }; - self.validate_scope(&advertisement)?; - if advertisement.issued_at_ms > now_ms.saturating_add(MAX_CLOCK_SKEW_MS) { - return Err(Error::Node("advertisement is not currently valid")); - } - Ok(advertisement.expires_at_ms > now_ms) - } - - /// Reports whether an exact session has a permanent canonical tombstone. - /// - /// Missing or advertised sessions return false. This does not grant ownership - /// or permit reuse of the retired identity. - pub async fn is_retired(&self, session: SessionId) -> Result { - let path = self.layout.node_path(session.as_bytes()); - let Some((record, _)) = self.load_record_at(&path).await? else { - return Ok(false); - }; - validate_record_path(&self.layout, record.session(), &path)?; - Ok(matches!(record, NodeRecord::Tombstone(_))) - } - - pub(super) fn validate(&self, advertisement: &NodeAdvertisement, now_ms: i64) -> Result<()> { - advertisement.validate_at(now_ms)?; - advertisement.verify_signature()?; - self.validate_scope(advertisement) - } - - pub(super) async fn update_advertisement( - &self, - observed: &VersionedNodeAdvertisement, - next: NodeAdvertisement, - now_ms: i64, - ) -> Result { - let path = self.layout.node_path(next.session.as_bytes()); - match self - .layout - .store() - .update(&path, Bytes::from(next.encode()?), observed.token.clone()) - .await - { - Ok(token) => Ok(VersionedNodeAdvertisement { - advertisement: next, - token, - }), - Err(update_error) => match self.load(next.session, now_ms).await? { - Some(current) if current.advertisement == next => Ok(current), - Some(_) | None => Err(update_error.into()), - }, - } - } - - pub(super) async fn load_canonical( - &self, - session: SessionId, - ) -> Result> { - let path = self.layout.node_path(session.as_bytes()); - let Some((record, token)) = self.load_record_at(&path).await? else { - return Ok(None); - }; - if record.session() != session { - return Err(Error::Node("advertisement path and session differ")); - } - match record { - NodeRecord::Advertisement(advertisement) => Ok(Some((*advertisement, token))), - NodeRecord::Tombstone(_) => Ok(None), - } - } - - pub(super) async fn load_record_at( - &self, - path: &object_store::path::Path, - ) -> Result> { - let (body, token) = match self - .layout - .store() - .get_with_etag_bounded(path, MAX_NODE_BYTES) - .await - { - Ok(value) => value, - Err(StorageError::NotFound { .. }) => return Ok(None), - Err(error) => return Err(error.into()), - }; - Ok(Some((NodeRecord::decode_canonical(&body)?, token))) - } - - pub(super) fn validate_scope(&self, advertisement: &NodeAdvertisement) -> Result<()> { - if advertisement.fleet != self.fleet - || advertisement.image != self.image - || advertisement.release != self.release - { - return Err(Error::Node("advertisement fleet, image or release differs")); - } - Ok(()) - } -} - -pub(super) struct RecoveryScanSnapshot { - pub(super) observed_at_ms: i64, - pub(super) includes_live_nodes: bool, - pub(super) live_nodes: HashSet, - pub(super) records: Vec, -} - -pub(super) struct RecoveryCandidateRecord { - pub(super) session: SessionId, - pub(super) expires_at_ms: i64, - pub(super) claimant: Option, - pub(super) claim_expires_at_ms: Option, - pub(super) active: bool, - pub(super) phase: NodeLogPhase, - pub(super) members: Vec, -} - -impl RecoveryCandidateRecord { - pub(super) fn eligible_for(&self, claimant: SessionId, now_ms: i64) -> bool { - self.expires_at_ms <= now_ms - && self.active - && matches!(self.phase, NodeLogPhase::Open | NodeLogPhase::Recovering) - && (self.claimant == Some(claimant) - || self - .claim_expires_at_ms - .is_none_or(|expires_at_ms| expires_at_ms <= now_ms)) - } -} - -/// Bounded rotating window over the expired sessions discovered in one scan. -/// -/// Object-store listings are not a durable work queue. Keeping only the first -/// page lets a permanently failing early session starve every later session, -/// so the window rotates its start key while retaining at most `2 * limit` -/// session IDs. -pub(super) struct RecoveryCandidateWindow { - pub(super) start: [u8; 16], - pub(super) limit: usize, - pub(super) after: BTreeSet<[u8; 16]>, - pub(super) before: BTreeSet<[u8; 16]>, -} - -impl RecoveryCandidateWindow { - pub(super) fn new(now_ms: i64, limit: usize) -> Result { - if now_ms < 0 || limit == 0 { - return Err(Error::Node("node recovery candidate window is invalid")); - } - let bucket = u64::try_from(now_ms / 1_000) - .map_err(|_| Error::Node("node recovery candidate rotation overflows"))?; - let mut hasher = blake3::Hasher::new(); - hasher.update(RECOVERY_CANDIDATE_ROTATION_DOMAIN); - hasher.update(&bucket.to_be_bytes()); - let digest = hasher.finalize(); - let mut start = [0_u8; 16]; - start.copy_from_slice(&digest.as_bytes()[..16]); - Ok(Self::with_start(start, limit)) - } - - pub(super) fn with_start(start: [u8; 16], limit: usize) -> Self { - Self { - start, - limit, - after: BTreeSet::new(), - before: BTreeSet::new(), - } - } - - pub(super) fn push(&mut self, session: SessionId) { - let key = *session.as_bytes(); - let window = if key >= self.start { - &mut self.after - } else { - &mut self.before - }; - if !window.insert(key) { - return; - } - if window.len() > self.limit { - let evicted = if key >= self.start { - window.iter().next_back().copied() - } else { - window.iter().next().copied() - }; - if let Some(evicted) = evicted { - window.remove(&evicted); - } - } - } - - pub(super) fn finish(self) -> Vec { - self.after - .into_iter() - .chain(self.before.into_iter().rev()) - .take(self.limit) - .map(SessionId::from_bytes) - .collect() - } -} - -pub(super) fn recovery_executor_eligible(advertisement: &NodeAdvertisement) -> bool { - let capacity = advertisement.capacity(); - let placement_has_headroom = advertisement.placement_capacity().is_none_or(|placement| { - placement.active_cells < placement.max_active_cells - && placement.running_jobs < placement.job_capacity - }); - capacity.log_protocol == NODE_LOG_PROTOCOL_VERSION - && capacity.free_memory_bytes != 0 - && capacity.free_disk_bytes != 0 - && capacity.job_credits != 0 - && placement_has_headroom -} - -pub(super) fn validate_record_path( - layout: &CellStorageLayout, - session: SessionId, - path: &object_store::path::Path, -) -> Result<()> { - if layout.node_path(session.as_bytes()) != *path { - return Err(Error::Node("advertisement path and session differ")); - } - Ok(()) -} - -pub(super) enum NodeRecord { - Advertisement(Box), - Tombstone(Box), -} - -impl NodeRecord { - pub(super) fn decode_canonical(bytes: &[u8]) -> Result { - if let Ok(advertisement) = NodeAdvertisement::decode_canonical(bytes) { - return Ok(Self::Advertisement(Box::new(advertisement))); - } - Ok(Self::Tombstone(Box::new(NodeTombstone::decode_canonical( - bytes, - )?))) - } - - const fn session(&self) -> SessionId { - match self { - Self::Advertisement(advertisement) => advertisement.session, - Self::Tombstone(tombstone) => tombstone.session, - } - } - - pub(super) fn log(&self) -> Option<&NodeLogStatus> { - match self { - Self::Advertisement(advertisement) => advertisement.log.as_ref(), - Self::Tombstone(tombstone) => tombstone.log.as_ref(), - } - } -} - -pub(super) struct NodeTombstone { - pub(super) session: SessionId, - pub(super) node: NodeId, - pub(super) expires_at_ms: i64, - pub(super) retired_at_ms: i64, - pub(super) claimant: Option, - pub(super) claim_generation: u64, - pub(super) claim_expires_at_ms: Option, - pub(super) log: Option, -} - -impl NodeTombstone { - pub(super) fn new( - session: SessionId, - node: NodeId, - expires_at_ms: i64, - retired_at_ms: i64, - claimant: Option, - log: Option, - ) -> Result { - let tombstone = Self { - session, - node, - expires_at_ms, - retired_at_ms, - claimant, - claim_generation: u64::from(claimant.is_some()), - claim_expires_at_ms: claimant.map(|_| { - retired_at_ms.saturating_add(crate::node::log_state::RECOVERY_CLAIM_LIFETIME_MS) - }), - log, - }; - tombstone.validate()?; - Ok(tombstone) - } - - pub(super) fn claim(mut self, claimant: SessionId, now_ms: i64) -> Result { - let generation = match (self.claimant, self.claim_expires_at_ms) { - (Some(current), Some(expires_at_ms)) - if current == claimant && now_ms < expires_at_ms => - { - return Ok(self); - } - (Some(_), Some(expires_at_ms)) if now_ms < expires_at_ms => { - return Err(Error::Node("node recovery is already claimed")); - } - (Some(_), Some(_)) => self - .claim_generation - .checked_add(1) - .ok_or(Error::Node("node recovery claim generation overflow"))?, - (None, None) => 1, - _ => return Err(Error::Node("node recovery claim is invalid")), - }; - self.claimant = Some(claimant); - self.claim_generation = generation; - self.claim_expires_at_ms = Some( - now_ms - .checked_add(crate::node::log_state::RECOVERY_CLAIM_LIFETIME_MS) - .ok_or(Error::Node("node recovery claim time overflow"))?, - ); - if let Some(log) = &self.log { - self.log = Some(log.begin_recovery(self.node, claimant, now_ms)?); - } - self.validate()?; - Ok(self) - } - - pub(super) fn renew(mut self, fenced: &FencedNodeSession, now_ms: i64) -> Result { - if self.session != fenced.session - || self.claimant != Some(fenced.claimant) - || self.claim_generation != fenced.claim_generation - { - return Err(Error::Fenced); - } - let current_expiry = self - .claim_expires_at_ms - .ok_or(Error::Node("node recovery claim expiry is missing"))?; - if now_ms >= current_expiry { - return Err(Error::Fenced); - } - let next_expiry = now_ms - .checked_add(crate::node::log_state::RECOVERY_CLAIM_LIFETIME_MS) - .filter(|expires_at_ms| *expires_at_ms > current_expiry) - .ok_or(Error::Node("node recovery claim expiry did not advance"))?; - self.claim_expires_at_ms = Some(next_expiry); - if let Some(log) = &self.log { - self.log = Some(log.renew_recovery( - self.node, - fenced.claimant, - fenced.claim_generation, - now_ms, - )?); - } - self.validate()?; - Ok(self) - } - - pub(super) fn seal( - mut self, - fenced: &FencedNodeSession, - recovery_manifest: Option, - now_ms: i64, - ) -> Result { - if self.session != fenced.session - || self.claimant != Some(fenced.claimant) - || self.claim_generation != fenced.claim_generation - || self - .claim_expires_at_ms - .is_none_or(|expires_at_ms| expires_at_ms <= now_ms) - { - return Err(Error::Fenced); - } - let log = self - .log - .as_ref() - .ok_or(Error::Node("claimed session has no enrolled node log"))? - .seal_recovery( - self.node, - fenced.claimant, - fenced.claim_generation, - recovery_manifest, - )?; - self.claimant = None; - self.claim_generation = 0; - self.claim_expires_at_ms = None; - self.log = Some(log); - self.validate()?; - Ok(self) - } - - pub(super) fn fenced(&self) -> Result { - let claimant = self - .claimant - .ok_or(Error::Node("node session is not claimed"))?; - Ok(FencedNodeSession { - node: self.node, - session: self.session, - claimant, - claim_generation: self.claim_generation, - claim_expires_at_ms: self - .claim_expires_at_ms - .ok_or(Error::Node("node recovery claim expiry is missing"))?, - log: self.log.clone(), - }) - } - - pub(super) fn encode(&self) -> Result> { - self.validate()?; - let encoded = serde_json::to_vec(&RawNodeTombstoneEnvelope::from(self))?; - if encoded.len() as u64 > MAX_NODE_BYTES { - return Err(Error::Node("node tombstone exceeds 64 KiB")); - } - Ok(encoded) - } - - pub(super) fn decode_canonical(bytes: &[u8]) -> Result { - if bytes.len() as u64 > MAX_NODE_BYTES { - return Err(Error::Node("node tombstone exceeds 64 KiB")); - } - let raw: RawNodeTombstoneEnvelope = serde_json::from_slice(bytes)?; - if raw.tombstone.version != 1 { - return Err(Error::Node("unsupported node tombstone version")); - } - let raw = raw.tombstone; - let session = SessionId::from_bytes(decode_hex(&raw.session)?); - let node = NodeId::from_bytes(decode_hex(&raw.node)?); - let tombstone = Self { - session, - node, - expires_at_ms: canonical_i64(&raw.expires_at_ms)?, - retired_at_ms: canonical_i64(&raw.retired_at_ms)?, - claimant: raw - .claimant - .map(|claimant| decode_hex(&claimant).map(SessionId::from_bytes)) - .transpose()?, - claim_generation: canonical_u64(&raw.claim_generation)?, - claim_expires_at_ms: raw - .claim_expires_at_ms - .as_deref() - .map(canonical_i64) - .transpose()?, - log: raw.log.map(|log| decode_log(node, log)).transpose()?, - }; - tombstone.validate()?; - if tombstone.encode()?.as_slice() != bytes { - return Err(Error::Node("node tombstone JSON is not canonical")); - } - Ok(tombstone) - } - - pub(super) fn validate(&self) -> Result<()> { - if self.session.as_bytes().iter().all(|byte| *byte == 0) - || self.node.as_bytes().iter().all(|byte| *byte == 0) - || self.expires_at_ms < 0 - || self.retired_at_ms < 0 - || self.claimant.is_some_and(|claimant| { - claimant == self.session || claimant.as_bytes().iter().all(|byte| *byte == 0) - }) - || self.claimant.is_some() != self.claim_expires_at_ms.is_some() - || self.claimant.is_some() != (self.claim_generation != 0) - || self - .claim_expires_at_ms - .is_some_and(|expires_at_ms| expires_at_ms <= self.retired_at_ms) - { - return Err(Error::Node("node tombstone is invalid")); - } - if let Some(log) = &self.log { - log.validate(self.node)?; - let log_claim = log.recovery(); - if self.claimant.is_some() != (log.phase() == NodeLogPhase::Recovering) - || log_claim.map(|claim| claim.claimant()) != self.claimant - || log_claim.map(|claim| claim.generation()) - != (self.claim_generation != 0).then_some(self.claim_generation) - || log_claim.map(|claim| claim.expires_at_ms()) != self.claim_expires_at_ms - { - return Err(Error::Node("node tombstone log claim differs")); - } - } - Ok(()) - } -} - -pub(super) fn compare_member_candidate( - left: &(&NodeAdvertisement, [u8; 32]), - right: &(&NodeAdvertisement, [u8; 32]), - context: &[&NodeAdvertisement], -) -> std::cmp::Ordering { - let zone_score = |candidate: &NodeAdvertisement| { - context - .iter() - .filter(|other| { - known_domain_difference( - candidate.failure_domain.zone(), - other.failure_domain.zone(), - ) - }) - .count() - }; - let host_score = |candidate: &NodeAdvertisement| { - context - .iter() - .filter(|other| { - known_domain_difference( - candidate.failure_domain.host(), - other.failure_domain.host(), - ) - }) - .count() - }; - zone_score(left.0) - .cmp(&zone_score(right.0)) - .then_with(|| host_score(left.0).cmp(&host_score(right.0))) - .then_with(|| left.1.cmp(&right.1)) - .then_with(|| right.0.node.as_bytes().cmp(left.0.node.as_bytes())) -} - -pub(super) fn known_domain_difference(left: Option<&str>, right: Option<&str>) -> bool { - matches!((left, right), (Some(left), Some(right)) if left != right) -} diff --git a/crates/crab-cell-runtime/src/node/directory/advertisement.rs b/crates/crab-cell-runtime/src/node/directory/advertisement.rs deleted file mode 100644 index 31e0c0ae4..000000000 --- a/crates/crab-cell-runtime/src/node/directory/advertisement.rs +++ /dev/null @@ -1,655 +0,0 @@ -//! Advertisement scans, liveness, placement inputs, and refreshes. -//! -//! Every entry point here reads or writes the signed advertisement records a -//! node publishes, so they share the scan bounds and fail-closed rules the -//! maintenance loop relies on. - -use super::*; - -const READER_MEMBERSHIP_TTL: std::time::Duration = std::time::Duration::from_secs(1); - -pub(super) struct ReaderMembership { - started: tokio::time::Instant, - observed_at_ms: i64, - nodes: Arc>, -} - -impl ReaderMembership { - fn current(&self, now_ms: i64) -> bool { - self.started.elapsed() < READER_MEMBERSHIP_TTL - && now_ms >= self.observed_at_ms - && now_ms.saturating_sub(self.observed_at_ms) < 1_000 - } -} - -impl NodeDirectory { - /// Selects advisory read-replica destinations from signed live nodes. - /// - /// A destination must still reserve its own resources and verify Cell - /// authority before opening a snapshot; this selection grants no read or - /// ownership capability. Discovery is shared across clones for at most one - /// second; expired advertisements are excluded on every selection. - pub async fn select_readers( - &self, - cell: crate::CellId, - owner: SessionId, - code: Digest, - desired: usize, - now_ms: i64, - limit: usize, - ) -> Result> { - let live = self.reader_membership(now_ms, limit).await?; - // An expired owner still identifies the excluded physical node. - // Selection is advisory and must survive owner death so warm readers - // remain discoverable; query and takeover gates enforce liveness. - // Node identity and failure domain cannot change within a boot session. - // Reuse their signed discovery proof for exclusion; this grants no - // liveness, which the query's final authority gate checks independently. - let owner_advertisement = match live.iter().find(|node| node.session() == owner) { - Some(owner) => Some(owner.clone()), - None => self.inspect_advertisement(owner, now_ms).await?, - }; - let owner_node = if let Some(advertisement) = &owner_advertisement { - advertisement.node() - } else { - let path = self.layout.node_path(owner.as_bytes()); - match self.load_record_at(&path).await? { - Some((NodeRecord::Tombstone(tombstone), _)) => tombstone.node, - _ => return Err(Error::Node("read-replica owner record is missing")), - } - }; - let mut zones = owner_advertisement - .as_ref() - .and_then(|owner| owner.failure_domain().zone()) - .map(str::to_owned) - .into_iter() - .collect::>(); - let mut hosts = owner_advertisement - .as_ref() - .and_then(|owner| owner.failure_domain().host()) - .map(str::to_owned) - .into_iter() - .collect::>(); - let mut eligible = live - .iter() - .filter(|candidate| { - let capacity = candidate.capacity(); - candidate.expires_at_ms() > now_ms - && candidate.node() != owner_node - && candidate.module_digests().contains(&code) - && capacity.free_memory_bytes - >= crate::fleet::resource::READ_REPLICA_NATIVE_BYTES as u64 - && capacity.free_disk_bytes > 0 - && capacity.job_credits > 0 - }) - .cloned() - .collect::>(); - let mut selected = Vec::with_capacity(desired.min(eligible.len())); - while selected.len() < desired && !eligible.is_empty() { - let index = eligible - .iter() - .enumerate() - .max_by_key(|(_, candidate)| { - let zone = candidate.failure_domain().zone(); - let host = candidate.failure_domain().host(); - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab-cell-read-replica-placement-v1"); - hasher.update(cell.as_bytes()); - hasher.update(candidate.node().as_bytes()); - ( - u8::from(zone.is_some_and(|value| !zones.contains(value))), - u8::from(host.is_some_and(|value| !hosts.contains(value))), - *hasher.finalize().as_bytes(), - ) - }) - .map(|(index, _)| index) - .ok_or(Error::Node("read-replica placement lost its candidate"))?; - let candidate = eligible.swap_remove(index); - if let Some(zone) = candidate.failure_domain().zone() { - zones.insert(zone.to_owned()); - } - if let Some(host) = candidate.failure_domain().host() { - hosts.insert(host.to_owned()); - } - selected.push(candidate); - } - Ok(selected) - } - - async fn reader_membership( - &self, - now_ms: i64, - limit: usize, - ) -> Result>> { - if limit == 0 { - return Err(Error::Node("live node limit must be nonzero")); - } - let cached = self.reader_membership.read().await; - if let Some(snapshot) = cached.as_ref().filter(|snapshot| snapshot.current(now_ms)) { - if snapshot - .nodes - .iter() - .filter(|node| node.expires_at_ms() > now_ms) - .count() - > limit - { - return Err(Error::Node("live node directory exceeds its limit")); - } - return Ok(Arc::clone(&snapshot.nodes)); - } - drop(cached); - // One bounded scan serves concurrent requests across Cells. Never use - // an expired observation when its refresh fails or is cancelled. - let mut cached = self.reader_membership.write().await; - if let Some(snapshot) = cached.as_ref().filter(|snapshot| snapshot.current(now_ms)) { - if snapshot - .nodes - .iter() - .filter(|node| node.expires_at_ms() > now_ms) - .count() - > limit - { - return Err(Error::Node("live node directory exceeds its limit")); - } - return Ok(Arc::clone(&snapshot.nodes)); - } - let started = tokio::time::Instant::now(); - let nodes = Arc::new(self.live(now_ms, limit).await?); - *cached = Some(ReaderMembership { - started, - observed_at_ms: now_ms, - nodes: Arc::clone(&nodes), - }); - Ok(nodes) - } - - /// Streams and verifies every currently live boot-session advertisement. - /// - /// Expired records do not count against `limit`; malformed, misplaced, or - /// foreign live records fail closed so maintenance cannot mistake an active - /// incompatible fleet for an offline deployment. - pub async fn live(&self, now_ms: i64, limit: usize) -> Result> { - let advertisements = self - .scan_advertisements(now_ms, limit, AdvertisementScan::LiveRelease) - .await?; - if advertisements.iter().enumerate().any(|(index, left)| { - advertisements[index + 1..] - .iter() - .any(|right| left.node == right.node) - }) { - return Err(Error::Node("multiple live sessions advertise one node")); - } - Ok(advertisements) - } - - /// Chooses a destination using only authenticated, measured placement - /// blocks advertised by the current live fleet. - /// - /// Nodes that have not rolled out the placement block are omitted. They - /// remain usable for ordinary authority routing but cannot become an - /// advisory destination through this method. This path chooses a new - /// owner when no live owner can be preferred, so no candidate receives - /// the owner stickiness bonus merely for handling the request. - pub async fn choose_advertised_placement( - &self, - planner: &PlacementPlanner, - cell: crate::CellId, - now_ms: i64, - limit: usize, - ) -> Result> { - let live = self.live(now_ms, limit).await?; - let observations = live - .iter() - .filter_map(|advertisement| { - PlacementObservation::from_signed_advertisement(advertisement, now_ms, false).ok() - }) - .collect::>(); - planner.choose(cell, now_ms, &observations) - } - - /// Resolves a stable physical node to its one current live boot session. - pub async fn resolve_node( - &self, - node: NodeId, - now_ms: i64, - ) -> Result> { - if node.as_bytes().iter().all(|byte| *byte == 0) { - return Err(Error::Node("node identity is zero")); - } - Ok(self - .live(now_ms, MAX_STALE_COLLECTION_ITEMS) - .await? - .into_iter() - .find(|advertisement| advertisement.node == node)) - } - - /// Selects the exact deterministic follower ensemble from current live capacity. - /// - /// An empty result means this fleet cannot currently satisfy the desired - /// one-follower/two-follower durability shape and must use object proof. - pub async fn select_log_members( - &self, - leader: SessionId, - required_follower_bytes: u64, - now_ms: i64, - limit: usize, - ) -> Result> { - if required_follower_bytes == 0 { - return Err(Error::Node("node-log follower byte requirement is zero")); - } - let live = self.live(now_ms, limit).await?; - let leader = live - .iter() - .find(|candidate| candidate.session == leader) - .ok_or(Error::Node("node-log leader is not live"))?; - let desired = live - .len() - .saturating_sub(1) - .min(crate::node::log_state::MAX_NODE_LOG_MEMBERS); - if desired == 0 { - return Ok(Vec::new()); - } - let mut eligible = live - .iter() - .filter(|candidate| { - candidate.node != leader.node - && candidate.capacity.log_protocol == NODE_LOG_PROTOCOL_VERSION - && candidate.capacity.follower_free_bytes >= required_follower_bytes - && candidate.capacity.free_memory_bytes != 0 - && candidate.capacity.free_disk_bytes != 0 - && candidate.capacity.job_credits != 0 - }) - .map(|candidate| { - let mut hasher = blake3::Hasher::new(); - hasher.update(NODE_LOG_SELECTION_DOMAIN); - hasher.update(leader.session.as_bytes()); - hasher.update(candidate.node.as_bytes()); - (candidate, *hasher.finalize().as_bytes()) - }) - .collect::>(); - if eligible.len() < desired { - return Ok(Vec::new()); - } - let mut selected_advertisements = Vec::with_capacity(desired); - while selected_advertisements.len() < desired { - let context = std::iter::once(leader) - .chain(selected_advertisements.iter().copied()) - .collect::>(); - let selected_index = eligible - .iter() - .enumerate() - .max_by(|(_, left), (_, right)| compare_member_candidate(left, right, &context)) - .map(|(index, _)| index) - .ok_or(Error::Node("node-log follower ensemble is unavailable"))?; - selected_advertisements.push(eligible.swap_remove(selected_index).0); - } - let mut selected = selected_advertisements - .into_iter() - .map(|advertisement| advertisement.node) - .collect::>(); - selected.sort_unstable_by(|left, right| left.as_bytes().cmp(right.as_bytes())); - Ok(selected) - } - - /// Lists every unfenced advertised session in this fleet, including expired records. - /// - /// Graceful withdrawal happens only after writers close. Stale collection first fences - /// the exact record by ETag. Maintenance must not interpret heartbeat expiry alone as drain. - pub async fn advertised_sessions(&self, now_ms: i64, limit: usize) -> Result> { - Ok(self - .scan_advertisements(now_ms, limit, AdvertisementScan::AdvertisedFleet) - .await? - .into_iter() - .map(|advertisement| advertisement.session) - .collect()) - } - - pub(super) async fn scan_advertisements( - &self, - now_ms: i64, - limit: usize, - scan: AdvertisementScan, - ) -> Result> { - if limit == 0 { - return Err(Error::Node(match scan { - AdvertisementScan::LiveRelease => "live node limit must be nonzero", - AdvertisementScan::AdvertisedFleet => { - "advertised node session limit must be nonzero" - } - })); - } - let prefix = self.layout.node_directory_path(); - let stream = self.layout.store().inner().list(Some(&prefix)); - let mut records = stream - .map(|item| { - let prefix = prefix.clone(); - async move { - let meta = - item.map_err(|error| map_object_store_error(error, prefix.as_ref()))?; - let (body, _) = match self - .layout - .store() - .get_with_etag_bounded(&meta.location, MAX_NODE_BYTES) - .await - { - Ok(value) => value, - Err(StorageError::NotFound { .. }) => return Ok(None), - Err(error) => return Err(error.into()), - }; - let NodeRecord::Advertisement(advertisement) = - NodeRecord::decode_canonical(&body)? - else { - return Ok(None); - }; - validate_record_path(&self.layout, advertisement.session, &meta.location)?; - match scan { - AdvertisementScan::LiveRelease => { - if advertisement.expires_at_ms <= now_ms { - return Ok(None); - } - self.validate(&advertisement, now_ms)?; - } - AdvertisementScan::AdvertisedFleet => { - advertisement.validate_shape()?; - advertisement.verify_signature()?; - if advertisement.fleet != self.fleet - || advertisement.issued_at_ms - > now_ms.saturating_add(MAX_CLOCK_SKEW_MS) - { - return Err(Error::Node( - "advertised node fleet or issue time differs", - )); - } - } - } - Ok(Some(*advertisement)) - } - }) - .buffer_unordered(NODE_DIRECTORY_READ_CONCURRENCY); - let mut advertisements = Vec::new(); - while let Some(advertisement) = records.next().await { - let Some(advertisement) = advertisement? else { - continue; - }; - if advertisements.len() == limit { - return Err(Error::Node(match scan { - AdvertisementScan::LiveRelease => "live node directory exceeds its limit", - AdvertisementScan::AdvertisedFleet => { - "advertised node session directory exceeds its limit" - } - })); - } - advertisements.push(advertisement); - } - advertisements - .sort_unstable_by(|left, right| left.session.as_bytes().cmp(right.session.as_bytes())); - Ok(advertisements) - } - - /// Fences a bounded number of advertisements past the clock-skew horizon. - /// - /// Tombstones remain until node-log recovery and every owned Cell complete; - /// generic stale collection cannot prove that retention condition. - pub async fn collect_stale(&self, now_ms: i64, limit: usize) -> Result { - if now_ms < 0 || !(1..=MAX_STALE_COLLECTION_ITEMS).contains(&limit) { - return Err(Error::Node( - "stale node collection limit or time is invalid", - )); - } - let cutoff_ms = now_ms.saturating_sub(STALE_ADVERTISEMENT_RETENTION_MS); - let prefix = self.layout.node_directory_path(); - let mut stream = self.layout.store().inner().list(Some(&prefix)); - let mut removed = 0; - while removed < limit - && let Some(item) = stream.next().await - { - let meta = item.map_err(|error| map_object_store_error(error, prefix.as_ref()))?; - let Some((record, token)) = self.load_record_at(&meta.location).await? else { - continue; - }; - let session = record.session(); - validate_record_path(&self.layout, session, &meta.location)?; - match record { - NodeRecord::Tombstone(_) => {} - NodeRecord::Advertisement(advertisement) - if advertisement.expires_at_ms <= cutoff_ms => - { - let tombstone = NodeTombstone::new( - advertisement.session, - advertisement.node, - advertisement.expires_at_ms, - now_ms, - None, - advertisement.log.clone(), - )?; - let encoded = tombstone.encode()?; - match self - .layout - .store() - .update(&meta.location, Bytes::from(encoded), token) - .await - { - Ok(_) => { - removed += 1; - } - Err(update_error) => match self.load_record_at(&meta.location).await? { - None => return Err(Error::Node("stale node record disappeared")), - Some((NodeRecord::Tombstone(_), _)) => removed += 1, - Some((NodeRecord::Advertisement(_), _)) => { - if !matches!(update_error, StorageError::StateConflict { .. }) { - return Err(update_error.into()); - } - } - }, - } - } - NodeRecord::Advertisement(_) => {} - } - } - Ok(removed) - } - - /// Conditionally withdraws the exact advertisement owned by a shutting-down node. - pub async fn withdraw(&self, observed: &VersionedNodeAdvertisement, now_ms: i64) -> Result<()> { - if now_ms < 0 { - return Err(Error::Node("node withdrawal time is invalid")); - } - self.validate_scope(&observed.advertisement)?; - if observed.advertisement.log.is_some() { - return Err(Error::Node( - "node log must be sealed before session withdrawal", - )); - } - let path = self - .layout - .node_path(observed.advertisement.session.as_bytes()); - let tombstone = NodeTombstone::new( - observed.advertisement.session, - observed.advertisement.node, - observed.advertisement.expires_at_ms, - now_ms, - None, - None, - )?; - let result = match self - .layout - .store() - .update( - &path, - Bytes::from(tombstone.encode()?), - observed.token.clone(), - ) - .await - { - Ok(_) => Ok(()), - Err(update_error) => match self.load_record_at(&path).await? { - None => Ok(()), - Some((NodeRecord::Tombstone(current), _)) if current.claimant.is_none() => Ok(()), - Some((NodeRecord::Tombstone(_), _)) => Err(Error::Fenced), - Some((NodeRecord::Advertisement(current), _)) - if *current == observed.advertisement => - { - Err(update_error.into()) - } - Some((NodeRecord::Advertisement(_), _)) => { - Err(Error::Node("advertisement changed during node withdrawal")) - } - }, - }; - self.reader_membership.write().await.take(); - result - } - - /// Withdraws a drained boot session, reconciling an unobserved heartbeat CAS. - /// - /// The caller must fence local admission, finish runtime/log drain, and stop - /// issuing refreshes first. Only successors of the same signed boot identity - /// may be reconciled; an unsealed log or recovery claim still rejects cleanup. - pub async fn withdraw_after_drain( - &self, - observed: &VersionedNodeAdvertisement, - now_ms: i64, - ) -> Result<()> { - let mut current = observed.clone(); - for _ in 0..4 { - let error = match self.withdraw(¤t, now_ms).await { - Ok(()) => return Ok(()), - Err(error) => error, - }; - let Some((next, token)) = self.load_canonical(observed.advertisement.session).await? - else { - return Err(error); - }; - // Canceling renewal cannot cancel a CAS already at the provider. - // Rebase only on a newer heartbeat from this drained boot; the - // exact withdrawal CAS prevents any later renewal from reviving it. - if !same_boot_identity(&observed.advertisement, &next) - || next.generation <= current.advertisement.generation - || next.issued_at_ms < current.advertisement.issued_at_ms - { - return Err(error); - } - current = VersionedNodeAdvertisement { - advertisement: next, - token, - }; - } - Err(Error::Node( - "node session changed during drained withdrawal", - )) - } - - /// Authenticates one request against its live advertisement and mTLS leaf digest. - pub async fn verify_peer_request( - &self, - input: &[u8], - certificate: Digest, - certificate_public_key: [u8; 32], - now_ms: i64, - ) -> Result { - let request = crate::peer::UnverifiedPeerRequest::decode(input)?; - self.peer_verifier( - request.session(), - certificate, - certificate_public_key, - now_ms, - ) - .await? - .verify(request, now_ms) - } - - /// Loads a live enrollment and binds its verifier to the mTLS identity. - /// - /// The returned verifier rechecks enrollment expiry when verification runs, - /// allowing callers to release CPU admission during this provider read. - pub async fn peer_verifier( - &self, - session: SessionId, - certificate: Digest, - certificate_public_key: [u8; 32], - now_ms: i64, - ) -> Result { - let enrolled = self - .load(session, now_ms) - .await? - .ok_or(Error::PeerAuthorization("peer session is not enrolled"))?; - if enrolled.advertisement.certificate != certificate { - return Err(Error::PeerAuthorization( - "mTLS certificate does not match peer session", - )); - } - if enrolled.advertisement.public_key != certificate_public_key { - return Err(Error::PeerAuthorization( - "mTLS certificate key does not match peer session", - )); - } - let verifier = crate::peer::PeerVerifier::new( - session, - self.release, - enrolled.advertisement.verifying_key()?, - ); - Ok(EnrolledPeerVerifier { - advertisement: enrolled.advertisement, - verifier, - }) - } - - /// Conditionally publishes the next heartbeat for the same boot session. - pub async fn refresh( - &self, - observed: &VersionedNodeAdvertisement, - next: NodeAdvertisement, - now_ms: i64, - ) -> Result { - let mut base = observed.clone(); - for _ in 0..4 { - if !same_boot_identity(&base.advertisement, &next) { - return Err(Error::Node("advertisement refresh changed boot identity")); - } - if next.issued_at_ms <= base.advertisement.issued_at_ms { - if next.progress <= base.advertisement.progress - && next.expires_at_ms <= base.advertisement.expires_at_ms - { - return Ok(base); - } - return Err(Error::Node("advertisement refresh lease regressed")); - } - let mut candidate = next.clone(); - candidate.generation = base - .advertisement - .generation - .checked_add(1) - .ok_or(Error::Node("node session generation overflow"))?; - candidate.log.clone_from(&base.advertisement.log); - self.validate(&candidate, now_ms)?; - validate_successor(&base.advertisement, &candidate)?; - let path = self.layout.node_path(candidate.session.as_bytes()); - match self - .layout - .store() - .update(&path, Bytes::from(candidate.encode()?), base.token.clone()) - .await - { - Ok(token) => { - return Ok(VersionedNodeAdvertisement { - advertisement: candidate, - token, - }); - } - Err(update_error) => match self.load(candidate.session, now_ms).await? { - Some(current) if current.advertisement == candidate => return Ok(current), - Some(current) - if same_boot_identity(&base.advertisement, ¤t.advertisement) - && current.advertisement.issued_at_ms - >= base.advertisement.issued_at_ms - && current.advertisement.progress >= base.advertisement.progress => - { - base = current; - } - Some(_) | None => return Err(update_error.into()), - }, - } - } - Err(Error::Node("node session changed during heartbeat refresh")) - } -} diff --git a/crates/crab-cell-runtime/src/node/directory/log.rs b/crates/crab-cell-runtime/src/node/directory/log.rs deleted file mode 100644 index f697b4d57..000000000 --- a/crates/crab-cell-runtime/src/node/directory/log.rs +++ /dev/null @@ -1,281 +0,0 @@ -//! Log authorization and lifecycle transitions for one node advertisement. -//! -//! A log epoch may only accept appends, retirements, or recovery from the -//! session that owns it, and every lifecycle transition is a CAS on the -//! signed advertisement, so these entry points share one validity window. - -use super::*; - -impl NodeDirectory { - /// Verifies live enrollment and that requested truncation is object-covered. - pub async fn authorize_log_append( - &self, - leader: SessionId, - member: NodeId, - log_epoch: u64, - covered_through: u64, - now_ms: i64, - ) -> Result { - let current = self - .load(leader, now_ms) - .await? - .ok_or(Error::PeerAuthorization("node-log leader is not live"))?; - let log = current - .advertisement - .log - .as_ref() - .ok_or(Error::PeerAuthorization( - "node-log leader has no enrolled log", - ))?; - log.permits_append(current.advertisement.node, member, log_epoch)?; - if covered_through > log.tiered_through() { - return Err(Error::PeerAuthorization( - "node-log append watermark exceeds authority", - )); - } - Ok(log.clone()) - } - - /// Verifies the leader may retire this member's fully object-covered epoch. - pub async fn authorize_log_retire( - &self, - leader: SessionId, - member: NodeId, - log_epoch: u64, - covered_through: u64, - now_ms: i64, - ) -> Result { - let log = self - .authorize_log_append(leader, member, log_epoch, covered_through, now_ms) - .await?; - if log.tiered_through() != covered_through { - return Err(Error::PeerAuthorization( - "node-log retire watermark differs from authority", - )); - } - Ok(log) - } - - /// Reports whether the authoritative session record still names one log epoch. - /// - /// A missing record is corruption rather than collection authority and fails - /// closed. Callers may delete an exact grace-aged retired follower lane only - /// when this returns `false`. - pub async fn log_epoch_referenced(&self, session: SessionId, epoch: u64) -> Result { - if epoch == 0 { - return Err(Error::Node("node-log epoch is zero")); - } - let path = self.layout.node_path(session.as_bytes()); - let Some((record, _)) = self.load_record_at(&path).await? else { - return Err(Error::Node("node session record is missing")); - }; - if record.session() != session { - return Err(Error::Node("node advertisement path and session differ")); - } - if let NodeRecord::Advertisement(advertisement) = &record { - self.validate_scope(advertisement)?; - advertisement.validate_shape()?; - advertisement.verify_signature()?; - } - Ok(record.log().is_some_and(|log| log.epoch() == epoch)) - } - - /// Verifies a live claimant may seal or read this follower's failed-owner lane. - pub async fn authorize_log_recovery( - &self, - leader: SessionId, - claimant: SessionId, - member: NodeId, - log_epoch: u64, - now_ms: i64, - ) -> Result { - self.load(claimant, now_ms) - .await? - .ok_or(Error::PeerAuthorization("node-log recoverer is not live"))?; - let path = self.layout.node_path(leader.as_bytes()); - let Some((NodeRecord::Tombstone(tombstone), _)) = self.load_record_at(&path).await? else { - return Err(Error::PeerAuthorization("node-log leader is not fenced")); - }; - let log = tombstone.log.as_ref().ok_or(Error::PeerAuthorization( - "node-log leader has no recovery log", - ))?; - log.permits_recovery_read(tombstone.node, claimant, member, log_epoch, now_ms)?; - Ok(log.clone()) - } - - /// Selects and CAS-enrolls the complete follower set before any frame is sent. - pub async fn recruit_log( - &self, - observed: &VersionedNodeAdvertisement, - log_epoch: u64, - required_follower_bytes: u64, - live_node_limit: usize, - now_ms: i64, - ) -> Result { - self.try_recruit_log( - observed, - log_epoch, - required_follower_bytes, - live_node_limit, - now_ms, - ) - .await? - .ok_or(Error::Node("node-log follower ensemble is unavailable")) - } - - /// CAS-enrolls followers when a complete ensemble is currently available. - pub async fn try_recruit_log( - &self, - observed: &VersionedNodeAdvertisement, - log_epoch: u64, - required_follower_bytes: u64, - live_node_limit: usize, - now_ms: i64, - ) -> Result> { - self.validate(&observed.advertisement, now_ms)?; - if observed.advertisement.log.is_some() { - return Err(Error::Node("node session already has an enrolled log")); - } - let members = self - .select_log_members( - observed.advertisement.session, - required_follower_bytes, - now_ms, - live_node_limit, - ) - .await?; - if members.is_empty() { - return Ok(None); - } - let mut next = observed.advertisement.clone(); - next.generation = next - .generation - .checked_add(1) - .ok_or(Error::Node("node session generation overflow"))?; - next.log = Some(NodeLogStatus::open(next.node, log_epoch, members)?); - self.update_advertisement(observed, next, now_ms) - .await - .map(Some) - } - - /// CAS-activates the exact enrolled epoch after every member fsyncs its first batch. - pub async fn activate_log( - &self, - observed: &VersionedNodeAdvertisement, - now_ms: i64, - ) -> Result { - self.validate(&observed.advertisement, now_ms)?; - let log = observed - .advertisement - .log - .as_ref() - .ok_or(Error::Node("node session has no enrolled log"))? - .activate(observed.advertisement.node)?; - let mut next = observed.advertisement.clone(); - next.generation = next - .generation - .checked_add(1) - .ok_or(Error::Node("node session generation overflow"))?; - next.log = Some(log); - self.update_advertisement(observed, next, now_ms).await - } - - /// CAS-advances the largest contiguous node sequence covered by object roots. - pub async fn advance_log_coverage( - &self, - observed: &VersionedNodeAdvertisement, - tiered_through: u64, - now_ms: i64, - ) -> Result { - self.validate(&observed.advertisement, now_ms)?; - let log = observed - .advertisement - .log - .as_ref() - .ok_or(Error::Node("node session has no enrolled log"))? - .advance_tiered(observed.advertisement.node, tiered_through)?; - let mut next = observed.advertisement.clone(); - next.generation = next - .generation - .checked_add(1) - .ok_or(Error::Node("node session generation overflow"))?; - next.log = Some(log); - self.update_advertisement(observed, next, now_ms).await - } - - /// CASes a fully object-covered old epoch to a newly selected inactive epoch. - pub async fn rotate_log( - &self, - observed: &VersionedNodeAdvertisement, - barrier: &NodeLogRotationBarrier, - required_follower_bytes: u64, - live_node_limit: usize, - now_ms: i64, - ) -> Result { - self.validate(&observed.advertisement, now_ms)?; - let current = observed - .advertisement - .log - .as_ref() - .ok_or(Error::Node("node session has no enrolled log"))?; - if barrier.leader_session() != observed.advertisement.session - || barrier.log_epoch() != current.epoch() - || barrier.members() != current.members() - || barrier.covered_through() != current.tiered_through() - { - return Err(Error::Node("node-log rotation barrier differs")); - } - let members = self - .select_log_members( - observed.advertisement.session, - required_follower_bytes, - now_ms, - live_node_limit, - ) - .await?; - if members.is_empty() { - return Err(Error::Node("node-log follower ensemble is unavailable")); - } - let next_epoch = current - .epoch() - .checked_add(1) - .ok_or(Error::Node("node-log epoch overflow"))?; - let mut next = observed.advertisement.clone(); - next.generation = next - .generation - .checked_add(1) - .ok_or(Error::Node("node session generation overflow"))?; - next.log = Some(NodeLogStatus::open(next.node, next_epoch, members)?); - self.update_advertisement(observed, next, now_ms).await - } - - /// CAS-clears one fully object-covered log before clean session withdrawal. - pub async fn close_log( - &self, - observed: &VersionedNodeAdvertisement, - barrier: &NodeLogRotationBarrier, - now_ms: i64, - ) -> Result { - self.validate(&observed.advertisement, now_ms)?; - let current = observed - .advertisement - .log - .as_ref() - .ok_or(Error::Node("node session has no enrolled log"))?; - if current.phase() != NodeLogPhase::Open - || barrier.leader_session() != observed.advertisement.session - || barrier.log_epoch() != current.epoch() - || barrier.members() != current.members() - || barrier.covered_through() != current.tiered_through() - { - return Err(Error::Node("node-log close barrier differs")); - } - let mut next = observed.advertisement.clone(); - next.generation = next - .generation - .checked_add(1) - .ok_or(Error::Node("node session generation overflow"))?; - next.log = None; - self.update_advertisement(observed, next, now_ms).await - } -} diff --git a/crates/crab-cell-runtime/src/node/directory/recovery.rs b/crates/crab-cell-runtime/src/node/directory/recovery.rs deleted file mode 100644 index 1960b5462..000000000 --- a/crates/crab-cell-runtime/src/node/directory/recovery.rs +++ /dev/null @@ -1,564 +0,0 @@ -//! Expired-claim fencing, takeover proofs, and recovery candidate windows. -//! -//! Every entry point here runs before a Cell moves: the directory must prove -//! the previous boot session is expired, that a takeover claim is signed by the -//! right node, and that the tail it hands over is object-covered. - -use super::*; - -impl NodeDirectory { - /// Reads live scheduler membership and primes the advisory recovery scan. - /// - /// Each call reads fresh signed records. Recovery claims still reload their - /// claimant and failed session before the fencing CAS. - /// Invalid bounds, incompatible records and storage failures return errors. - pub async fn live_for_recovery( - &self, - now_ms: i64, - limit: usize, - ) -> Result> { - if now_ms < 0 || !(1..=MAX_LIVE_NODE_RECORDS).contains(&limit) { - return Err(Error::Node("node recovery membership bound is invalid")); - } - let mut cached = self.recovery_scan.write().await; - // A scheduler cycle needs a fresh view. Clear the old observation so - // cancellation or failure cannot leave it serving recovery discovery. - cached.take(); - let (snapshot, live) = self.scan_recovery_records(now_ms, true, limit).await?; - *cached = Some(Arc::new(snapshot)); - Ok(live) - } - - /// Fences one expired boot session with an ETag CAS before any Cell takeover. - /// - /// Missing, live, malformed, or foreign records fail closed. The tombstone - /// remains durable so a paused owner cannot revive its old advertisement. - pub async fn claim_expired( - &self, - session: SessionId, - claimant: SessionId, - now_ms: i64, - ) -> Result { - self.claim_expired_inner(session, claimant, now_ms, false, false) - .await - } - - /// Claims an expired session only after the live claimant passes recovery - /// admission immediately before the fencing CAS. - /// - /// Candidate discovery is advisory and may be stale by the time a - /// scheduler reaches this boundary. The generic [`Self::claim_expired`] - /// API intentionally remains usable by low-level recovery tests and - /// callers that provide their own admission policy; production recovery - /// uses this stricter entry point so a drained claimant cannot acquire a - /// claim after its signed capacity has become ineligible. - pub async fn claim_expired_for_recovery( - &self, - session: SessionId, - claimant: SessionId, - now_ms: i64, - ) -> Result { - self.claim_expired_inner(session, claimant, now_ms, false, true) - .await - } - - /// Fences an expired session or reuses its completed takeover authority. - /// - /// An unsealed active log returns `PendingPublication` without writing a - /// claim so the follower recovery scheduler can proceed. - pub async fn claim_expired_for_takeover( - &self, - session: SessionId, - claimant: SessionId, - now_ms: i64, - ) -> Result { - match self - .claim_expired_inner(session, claimant, now_ms, true, false) - .await - { - Ok(fenced) => fenced.direct_takeover(), - // Another request may have fenced the same dead session after our - // routing observation. Re-read durable proof before reporting its - // claim conflict; Cell ownership still requires a separate CAS. - Err(error) => self - .takeover_proof(session, claimant, now_ms) - .await? - .ok_or(error), - } - } - - pub(super) async fn claim_expired_inner( - &self, - session: SessionId, - claimant: SessionId, - now_ms: i64, - reject_active_log: bool, - require_recovery_eligibility: bool, - ) -> Result { - if now_ms < 0 || claimant.as_bytes().iter().all(|byte| *byte == 0) || claimant == session { - return Err(Error::Node("node recovery time is invalid")); - } - let claimant_advertisement = self - .load(claimant, now_ms) - .await? - .ok_or(Error::Node("node recovery claimant is not live"))?; - if require_recovery_eligibility - && !recovery_executor_eligible(claimant_advertisement.advertisement()) - { - return Err(Error::Capacity("node recovery claimant is not eligible")); - } - let path = self.layout.node_path(session.as_bytes()); - let Some((record, token)) = self.load_record_at(&path).await? else { - return Err(Error::Node("expired node session record is missing")); - }; - let tombstone = match record { - NodeRecord::Tombstone(tombstone) => { - if tombstone.session != session { - return Err(Error::Node("node tombstone session differs")); - } - if reject_active_log && tombstone.log.as_ref().is_some_and(NodeLogStatus::active) { - return Err(Error::PendingPublication); - } - if tombstone.claimant == Some(claimant) - && tombstone - .claim_expires_at_ms - .is_some_and(|expires_at_ms| expires_at_ms > now_ms) - { - return tombstone.fenced(); - } - (*tombstone).claim(claimant, now_ms)? - } - NodeRecord::Advertisement(advertisement) => { - self.validate_scope(&advertisement)?; - advertisement.validate_shape()?; - advertisement.verify_signature()?; - if advertisement.session != session || advertisement.expires_at_ms > now_ms { - return Err(Error::Node("node session is not expired")); - } - if reject_active_log - && advertisement - .log - .as_ref() - .is_some_and(NodeLogStatus::active) - { - return Err(Error::PendingPublication); - } - NodeTombstone::new( - session, - advertisement.node, - advertisement.expires_at_ms, - now_ms, - None, - advertisement.log.clone(), - )? - .claim(claimant, now_ms)? - } - }; - let proof = tombstone.fenced()?; - match self - .layout - .store() - .update(&path, Bytes::from(tombstone.encode()?), token) - .await - { - Ok(_) => Ok(proof), - Err(update_error) => match self.load_record_at(&path).await? { - Some((NodeRecord::Tombstone(current), _)) - if current.session == session - && current.claimant == Some(claimant) - && current - .claim_expires_at_ms - .is_some_and(|expires_at_ms| expires_at_ms > now_ms) => - { - current.fenced() - } - Some(_) | None => Err(update_error.into()), - }, - } - } - - /// Loads takeover authority from a permanent fence with no unrecovered active log. - pub async fn takeover_proof( - &self, - session: SessionId, - claimant: SessionId, - now_ms: i64, - ) -> Result> { - if now_ms < 0 || claimant == session { - return Err(Error::Node("node takeover time is invalid")); - } - self.load(claimant, now_ms) - .await? - .ok_or(Error::Node("node takeover claimant is not live"))?; - let path = self.layout.node_path(session.as_bytes()); - let Some((record, _)) = self.load_record_at(&path).await? else { - return Err(Error::Node("node takeover record is missing")); - }; - let NodeRecord::Tombstone(tombstone) = record else { - return Ok(None); - }; - if tombstone.session != session { - return Err(Error::Node("node tombstone session differs")); - } - // The claim serializes active follower-tail recovery, not all Cells - // from one failed node. Absent/inactive logs cannot add acknowledged - // state after fencing, so each live successor can acquire its own Cell. - let ready = match tombstone.log.as_ref() { - Some(log) if matches!(log.phase(), NodeLogPhase::Sealed | NodeLogPhase::Retired) => { - true - } - Some(log) => !log.active(), - None => true, - }; - Ok(ready.then_some(NodeTakeoverProof { session, claimant })) - } - - /// Resolves the deterministic live original-follower successor for a - /// sealed failed session. The returned advertisement is advisory; the - /// destination still rechecks takeover proof and Cell control CAS. - pub async fn preferred_recovery_node( - &self, - session: SessionId, - now_ms: i64, - ) -> Result> { - if now_ms < 0 { - return Err(Error::Node("node recovery time is invalid")); - } - let path = self.layout.node_path(session.as_bytes()); - let Some((record, _)) = self.load_record_at(&path).await? else { - return Err(Error::Node("node recovery record is missing")); - }; - let log = match record { - NodeRecord::Advertisement(advertisement) => { - self.validate_scope(&advertisement)?; - advertisement.validate_shape()?; - advertisement.verify_signature()?; - if advertisement.expires_at_ms > now_ms { - return Ok(None); - } - advertisement.log - } - NodeRecord::Tombstone(tombstone) => tombstone.log, - }; - let Some(log) = log else { - return Ok(None); - }; - if !log.active() || !matches!(log.phase(), NodeLogPhase::Sealed | NodeLogPhase::Retired) { - return Ok(None); - } - let live = self.live(now_ms, MAX_LIVE_NODE_RECORDS).await?; - Ok(log - .members() - .iter() - .filter_map(|member| { - live.iter().find(|advertisement| { - advertisement.node() == *member && recovery_executor_eligible(advertisement) - }) - }) - .min_by(|left, right| left.node().as_bytes().cmp(right.node().as_bytes())) - .cloned()) - } - - /// Lists expired active logs whose claim is available to this live session. - pub async fn recovery_candidates( - &self, - claimant: SessionId, - now_ms: i64, - limit: usize, - ) -> Result> { - self.recovery_candidates_filtered(claimant, None, false, now_ms, limit) - .await - } - - /// Lists expired logs for which this claimant is the deterministic live - /// original-follower successor. Stable NodeIds choose the winner; the - /// current boot SessionId remains the claim authority. - pub async fn recovery_candidates_for_node( - &self, - claimant: SessionId, - claimant_node: NodeId, - now_ms: i64, - limit: usize, - ) -> Result> { - self.recovery_candidates_filtered(claimant, Some(claimant_node), false, now_ms, limit) - .await - } - - /// Lists expired logs only when no enrolled original follower is live. - /// This is the bounded any-node fallback after follower-affine attempts. - pub async fn recovery_candidates_without_live_followers( - &self, - claimant: SessionId, - now_ms: i64, - limit: usize, - ) -> Result> { - self.recovery_candidates_filtered(claimant, None, true, now_ms, limit) - .await - } - - pub(super) async fn recovery_candidates_filtered( - &self, - claimant: SessionId, - claimant_node: Option, - require_no_live_followers: bool, - now_ms: i64, - limit: usize, - ) -> Result> { - if now_ms < 0 || limit == 0 || limit > MAX_STALE_COLLECTION_ITEMS { - return Err(Error::Node("node recovery candidate bound is invalid")); - } - let claimant_advertisement = self - .load(claimant, now_ms) - .await? - .ok_or(Error::Node("node recovery claimant is not live"))?; - if !recovery_executor_eligible(claimant_advertisement.advertisement()) { - return Ok(Vec::new()); - } - if claimant_node.is_some_and(|node| claimant_advertisement.advertisement.node() != node) { - return Err(Error::Node("node recovery claimant identity differs")); - } - let snapshot = self - .recovery_scan_snapshot(now_ms, claimant_node.is_some() || require_no_live_followers) - .await?; - let live_nodes = &snapshot.live_nodes; - let mut candidates = RecoveryCandidateWindow::new(now_ms, limit)?; - for record in &snapshot.records { - if !record.eligible_for(claimant, now_ms) || record.session == claimant { - continue; - } - if let Some(claimant_node) = claimant_node { - let preferred = record - .members - .iter() - .filter(|member| live_nodes.contains(member)) - .min_by(|left, right| left.as_bytes().cmp(right.as_bytes())); - if preferred != Some(&claimant_node) { - continue; - } - } else if require_no_live_followers - && record - .members - .iter() - .any(|member| live_nodes.contains(member)) - { - continue; - } - candidates.push(record.session); - } - Ok(candidates.finish()) - } - - pub(in crate::node) async fn recovery_scan_snapshot( - &self, - now_ms: i64, - include_live_nodes: bool, - ) -> Result> { - if let Some(snapshot) = self.recovery_scan.read().await.as_ref() - && now_ms >= snapshot.observed_at_ms - && now_ms.saturating_sub(snapshot.observed_at_ms) < RECOVERY_SCAN_CACHE_TTL_MS - && (!include_live_nodes || snapshot.includes_live_nodes) - { - return Ok(Arc::clone(snapshot)); - } - - let mut cached = self.recovery_scan.write().await; - if let Some(snapshot) = cached.as_ref() - && now_ms >= snapshot.observed_at_ms - && now_ms.saturating_sub(snapshot.observed_at_ms) < RECOVERY_SCAN_CACHE_TTL_MS - && (!include_live_nodes || snapshot.includes_live_nodes) - { - return Ok(Arc::clone(snapshot)); - } - - let (snapshot, _) = self - .scan_recovery_records(now_ms, include_live_nodes, MAX_LIVE_NODE_RECORDS) - .await?; - let snapshot = Arc::new(snapshot); - *cached = Some(Arc::clone(&snapshot)); - Ok(snapshot) - } - - async fn scan_recovery_records( - &self, - now_ms: i64, - include_live_nodes: bool, - live_node_limit: usize, - ) -> Result<(RecoveryScanSnapshot, Vec)> { - let prefix = self.layout.node_directory_path(); - let stream = self.layout.store().inner().list(Some(&prefix)); - let mut live_nodes = HashSet::new(); - let mut live_sessions = HashSet::new(); - let mut live = Vec::new(); - let mut records = stream - .map(|item| { - let prefix = prefix.clone(); - async move { - let meta = - item.map_err(|error| map_object_store_error(error, prefix.as_ref()))?; - let Some((record, _)) = self.load_record_at(&meta.location).await? else { - return Ok(None); - }; - let session = record.session(); - validate_record_path(&self.layout, session, &meta.location)?; - match record { - NodeRecord::Advertisement(advertisement) => { - self.validate_scope(&advertisement)?; - advertisement.validate_shape()?; - advertisement.verify_signature()?; - if advertisement.issued_at_ms > now_ms.saturating_add(MAX_CLOCK_SKEW_MS) - { - return Err(Error::Node("advertised node issue time differs")); - } - let candidate = - advertisement - .log - .as_ref() - .map(|log| RecoveryCandidateRecord { - session, - expires_at_ms: advertisement.expires_at_ms, - claimant: None, - claim_expires_at_ms: None, - active: log.active(), - phase: log.phase(), - members: log.members().to_vec(), - }); - let live = if include_live_nodes && advertisement.expires_at_ms > now_ms - { - self.validate(&advertisement, now_ms)?; - Some(advertisement) - } else { - None - }; - Ok(Some((candidate, live))) - } - NodeRecord::Tombstone(tombstone) => { - let candidate = - tombstone.log.as_ref().map(|log| RecoveryCandidateRecord { - session, - expires_at_ms: tombstone.expires_at_ms, - claimant: tombstone.claimant, - claim_expires_at_ms: tombstone.claim_expires_at_ms, - active: log.active(), - phase: log.phase(), - members: log.members().to_vec(), - }); - Ok(Some((candidate, None))) - } - } - } - }) - .buffer_unordered(NODE_DIRECTORY_READ_CONCURRENCY); - let mut candidates = Vec::new(); - while let Some(record) = records.next().await { - if let Some((record, advertisement)) = record? { - if let Some(advertisement) = advertisement { - if live.len() == live_node_limit { - return Err(Error::Node("live node directory exceeds its limit")); - } - let node = advertisement.node(); - if !live_sessions.insert(node) { - return Err(Error::Node("multiple live sessions advertise one node")); - } - if recovery_executor_eligible(&advertisement) { - live_nodes.insert(node); - } - live.push(*advertisement); - } - let Some(record) = record else { - continue; - }; - if candidates.len() == MAX_LIVE_NODE_RECORDS { - return Err(Error::Node("node recovery directory exceeds its limit")); - } - candidates.push(record); - } - } - live.sort_unstable_by(|left, right| left.session.as_bytes().cmp(right.session.as_bytes())); - let snapshot = RecoveryScanSnapshot { - observed_at_ms: now_ms, - includes_live_nodes: include_live_nodes, - live_nodes, - records: candidates, - }; - Ok((snapshot, live)) - } - - /// Extends an exact recovery claim while its claimant remains live. - pub async fn refresh_recovery_claim( - &self, - fenced: &FencedNodeSession, - now_ms: i64, - ) -> Result { - self.load(fenced.claimant, now_ms) - .await? - .ok_or(Error::Fenced)?; - let path = self.layout.node_path(fenced.session.as_bytes()); - let Some((NodeRecord::Tombstone(current), token)) = self.load_record_at(&path).await? - else { - return Err(Error::Fenced); - }; - let renewed = (*current).renew(fenced, now_ms)?; - let proof = renewed.fenced()?; - self.layout - .store() - .update(&path, Bytes::from(renewed.encode()?), token) - .await?; - Ok(proof) - } - - /// Seals an exact recovery claim after every affected Cell pins its overlay. - pub(crate) async fn seal_recovery( - &self, - fenced: &FencedNodeSession, - recovery_manifest: Option, - now_ms: i64, - ) -> Result { - let path = self.layout.node_path(fenced.session.as_bytes()); - let Some((NodeRecord::Tombstone(current), token)) = self.load_record_at(&path).await? - else { - return Err(Error::Fenced); - }; - if let Some(log) = ¤t.log - && log.phase() == NodeLogPhase::Sealed - && log.recovery_manifest() == recovery_manifest - { - return Ok(SealedNodeLog { - session: current.session, - log: log.clone(), - }); - } - let sealed = (*current).seal(fenced, recovery_manifest, now_ms)?; - let log = sealed - .log - .clone() - .ok_or(Error::Node("sealed node session lost its log"))?; - match self - .layout - .store() - .update(&path, Bytes::from(sealed.encode()?), token) - .await - { - Ok(_) => Ok(SealedNodeLog { - session: sealed.session, - log, - }), - Err(update_error) => match self.load_record_at(&path).await? { - Some((NodeRecord::Tombstone(current), _)) - if current.session == fenced.session - && current.log.as_ref().is_some_and(|current| { - current.phase() == NodeLogPhase::Sealed - && current.recovery_manifest() == recovery_manifest - }) => - { - Ok(SealedNodeLog { - session: current.session, - log: current - .log - .ok_or(Error::Node("sealed node session lost its log"))?, - }) - } - Some(_) | None => Err(update_error.into()), - }, - } - } -} diff --git a/crates/crab-cell-runtime/src/node/durability.rs b/crates/crab-cell-runtime/src/node/durability.rs deleted file mode 100644 index a9d0ed5d8..000000000 --- a/crates/crab-cell-runtime/src/node/durability.rs +++ /dev/null @@ -1,271 +0,0 @@ -//! Durability gate: commit tickets and fleet or object proofs. -use std::sync::Arc; - -use futures_util::future::BoxFuture; -use tokio::sync::{Mutex, OnceCell}; - -use crate::identity::NodeId; -use crate::identity::SessionId; -use crate::node::lease::NodeLeaseGuard; -use crate::node::log::{ - CommitTicket, DurabilityGate, DurabilityProof, DurabilitySource, NodeLogRotationBarrier, -}; -use crate::node::log_shipper::{NodeLogShipper, NodeLogSubmission}; -use crate::node::log_transport::NodeLogTransport; -use crate::{Error, Result}; - -/// Authoritative node-session mutations required by follower durability. -/// -/// Implementations must serialize these mutations with heartbeat refreshes and -/// reconcile an ambiguous CAS only when the exact session and log epoch match. -pub trait NodeLogAuthority: Send + Sync { - /// Activates one log epoch for this session. - fn activate<'a>(&'a self, log_epoch: u64) -> BoxFuture<'a, Result<()>>; - - /// Advances the tiered coverage watermark for one log epoch. - fn advance_coverage<'a>( - &'a self, - log_epoch: u64, - tiered_through: u64, - ) -> BoxFuture<'a, Result<()>>; - - /// Closes one log epoch at its rotation barrier. - fn close<'a>(&'a self, barrier: &'a NodeLogRotationBarrier) -> BoxFuture<'a, Result<()>>; -} - -/// Provider-neutral inputs for constructing one node-log durability epoch. -/// -/// Providers own enrollment and the authority/transport implementations. The -/// host owns turning these inputs into the runtime durability object so the -/// product boundary cannot accidentally create a second shipping path. -pub struct NodeDurabilityConfig { - session: SessionId, - node: NodeId, - log_epoch: u64, - members: Vec, - transport: Arc, - authority: Arc, - node_lease: NodeLeaseGuard, - limits: crab_ltx::Limits, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, -} - -impl NodeDurabilityConfig { - /// Binds provider enrollment to one exact node session and log epoch. - #[expect( - clippy::too_many_arguments, - reason = "the provider-neutral boundary keeps every enrollment contract explicit" - )] - pub fn new( - session: SessionId, - node: NodeId, - log_epoch: u64, - members: Vec, - transport: Arc, - authority: Arc, - node_lease: NodeLeaseGuard, - limits: crab_ltx::Limits, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, - ) -> Result { - if session.as_bytes().iter().all(|byte| *byte == 0) - || node.as_bytes().iter().all(|byte| *byte == 0) - || log_epoch == 0 - || members.is_empty() - { - return Err(Error::Node("invalid node-log durability configuration")); - } - Ok(Self { - session, - node, - log_epoch, - members, - transport, - authority, - node_lease, - limits, - telemetry, - }) - } - - /// Constructs the runtime-owned durability object for this epoch. - pub fn build(self) -> Result> { - let gate = DurabilityGate::new(self.session, self.node, self.log_epoch, self.members)?; - let shipper = NodeLogShipper::new_with_telemetry( - gate.clone(), - Arc::clone(&self.transport), - self.limits, - self.telemetry, - )?; - Ok(Arc::new(NodeDurability::new( - gate, - shipper, - self.authority, - self.transport, - self.node_lease, - ))) - } -} - -/// One enrolled node-log epoch and its non-forgeable durability proof boundary. -/// -/// The runtime can submit captured cuts immediately after enrollment, but no -/// fleet proof becomes visible until the first ticket is fsynced by every -/// member and the authoritative inactive-to-active CAS succeeds. -pub struct NodeDurability { - gate: DurabilityGate, - shipper: NodeLogShipper, - authority: Arc, - transport: Arc, - node_lease: NodeLeaseGuard, - activated: OnceCell<()>, - object_coverage: Mutex<()>, - shutdown: Mutex<()>, - closed: std::sync::atomic::AtomicBool, -} - -impl NodeDurability { - /// Creates one node-log durability epoch over its gate, shipper, authority, - /// transport, and node lease. - #[must_use] - pub fn new( - gate: DurabilityGate, - shipper: NodeLogShipper, - authority: Arc, - transport: Arc, - node_lease: NodeLeaseGuard, - ) -> Self { - Self { - gate, - shipper, - authority, - transport, - node_lease, - activated: OnceCell::new(), - object_coverage: Mutex::new(()), - shutdown: Mutex::new(()), - closed: std::sync::atomic::AtomicBool::new(false), - } - } - - /// Returns whether shipping stopped or the epoch reached its rotation threshold. - #[must_use] - pub fn needs_rotation(&self, max_issued_frames: u64) -> bool { - self.gate.shipping_scope().is_err() || self.gate.issued_through() >= max_issued_frames - } - - /// Assigns and asynchronously ships one captured commit to every member. - pub async fn submit(&self, submission: NodeLogSubmission) -> Result { - self.node_lease.check()?; - let ticket = tokio::select! { - result = self.shipper.submit(submission) => result?, - () = self.node_lease.wait_fenced() => return Err(Error::Fenced), - }; - self.node_lease.check()?; - Ok(ticket) - } - - /// Returns fleet proof only after follower fsync and authoritative activation. - pub async fn prove_fleet(&self, ticket: CommitTicket) -> Result { - self.node_lease.check()?; - tokio::select! { - result = self.gate.wait_followers(ticket) => result?, - () = self.node_lease.wait_fenced() => return Err(Error::Fenced), - } - self.activated - .get_or_try_init(|| async { - self.node_lease.check()?; - tokio::select! { - result = self.authority.activate(ticket.log_epoch()) => result?, - () = self.node_lease.wait_fenced() => return Err(Error::Fenced), - } - self.node_lease.check()?; - self.gate.activate_fleet()?; - Ok::<(), Error>(()) - }) - .await?; - let proof = tokio::select! { - result = self.gate.prove(ticket) => result?, - () = self.node_lease.wait_fenced() => return Err(Error::Fenced), - }; - self.node_lease.check()?; - if proof.source() != DurabilitySource::Fleet { - return Err(Error::Node( - "fleet proof was superseded by object durability", - )); - } - Ok(proof) - } - - /// Returns the first valid follower or object proof for one ticket. - pub async fn prove(&self, ticket: CommitTicket) -> Result { - self.node_lease.check()?; - let activate = async { - self.gate.wait_followers(ticket).await?; - self.activated - .get_or_try_init(|| async { - self.node_lease.check()?; - tokio::select! { - result = self.authority.activate(ticket.log_epoch()) => result?, - () = self.node_lease.wait_fenced() => return Err(Error::Fenced), - } - self.node_lease.check()?; - self.gate.activate_fleet()?; - Ok::<(), Error>(()) - }) - .await?; - Ok::<(), Error>(()) - }; - tokio::pin!(activate); - let proof = tokio::select! { - proof = self.gate.prove(ticket) => proof?, - activated = &mut activate => { - activated?; - self.gate.prove(ticket).await? - } - () = self.node_lease.wait_fenced() => return Err(Error::Fenced), - }; - self.node_lease.check()?; - Ok(proof) - } - - /// Records an already-published object root and persists its contiguous watermark. - /// - /// Callers must complete the exact Cell root CAS before invoking this method. - pub async fn prove_object(&self, ticket: CommitTicket) -> Result { - // Publishers from different Cells share this watermark. Serialize the - // preview, authority CAS and local confirmation so concurrent completions - // cannot leave the persisted prefix behind the local truncation proof. - let _coverage = self.object_coverage.lock().await; - self.node_lease.check()?; - let tiered_through = self.gate.preview_object(ticket)?; - tokio::select! { - result = self.authority.advance_coverage(ticket.log_epoch(), tiered_through) => result?, - () = self.node_lease.wait_fenced() => return Err(Error::Fenced), - } - self.node_lease.check()?; - self.gate.prove_object(ticket)?; - let proof = self.gate.prove(ticket).await?; - if proof.source() != DurabilitySource::Object { - return Err(Error::Node("object proof lost its durability race")); - } - Ok(proof) - } - - /// Drains accepted frames and permanently closes fleet issuance for this epoch. - pub async fn shutdown(&self) -> Result<()> { - let _shutdown = self.shutdown.lock().await; - if self.closed.load(std::sync::atomic::Ordering::Acquire) { - return Ok(()); - } - self.shipper.shutdown().await?; - let barrier = self.gate.begin_rotation()?; - crate::node::log::retire_node_log(Arc::clone(&self.transport), &barrier).await?; - self.authority.close(&barrier).await?; - self.closed - .store(true, std::sync::atomic::Ordering::Release); - Ok(()) - } -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/node/durability/tests.rs b/crates/crab-cell-runtime/src/node/durability/tests.rs deleted file mode 100644 index bb64a405a..000000000 --- a/crates/crab-cell-runtime/src/node/durability/tests.rs +++ /dev/null @@ -1,408 +0,0 @@ -use std::sync::{Arc, Mutex}; - -use bytes::Bytes; -use tokio::sync::Notify; - -use super::*; -use crate::follower::FollowerReceipt; -use crate::identity::{ApplicationId, CellId, SessionId}; -use crate::identity::{IncarnationId, NodeId}; -use crate::node::log_transport::{ - AppendRequest, NodeLogTransport, RetireRequest, SealRequest, TailRequest, -}; - -#[derive(Default)] -struct AuthorityState { - activations: Vec, - coverage: Vec<(u64, u64)>, - reject_coverage: bool, - yield_coverage: bool, -} - -#[derive(Default)] -struct RecordingAuthority(Mutex); - -impl NodeLogAuthority for RecordingAuthority { - fn activate<'a>(&'a self, log_epoch: u64) -> BoxFuture<'a, Result<()>> { - Box::pin(async move { - self.0.lock().unwrap().activations.push(log_epoch); - Ok(()) - }) - } - - fn advance_coverage<'a>( - &'a self, - log_epoch: u64, - tiered_through: u64, - ) -> BoxFuture<'a, Result<()>> { - Box::pin(async move { - let yield_coverage = self.0.lock().unwrap().yield_coverage; - if yield_coverage { - tokio::task::yield_now().await; - } - let mut state = self.0.lock().unwrap(); - if state.reject_coverage { - return Err(Error::Node("coverage rejected")); - } - state.coverage.push((log_epoch, tiered_through)); - Ok(()) - }) - } - - fn close<'a>(&'a self, _barrier: &'a NodeLogRotationBarrier) -> BoxFuture<'a, Result<()>> { - Box::pin(async { Ok(()) }) - } -} - -struct BlockingAuthority { - activation_started: Option>, - coverage_started: Option>, -} - -impl NodeLogAuthority for BlockingAuthority { - fn activate<'a>(&'a self, _log_epoch: u64) -> BoxFuture<'a, Result<()>> { - let Some(started) = &self.activation_started else { - return Box::pin(async { Ok(()) }); - }; - let started = Arc::clone(started); - Box::pin(async move { - started.notify_one(); - std::future::pending().await - }) - } - - fn advance_coverage<'a>( - &'a self, - _log_epoch: u64, - _tiered_through: u64, - ) -> BoxFuture<'a, Result<()>> { - let Some(started) = &self.coverage_started else { - return Box::pin(async { Ok(()) }); - }; - let started = Arc::clone(started); - Box::pin(async move { - started.notify_one(); - std::future::pending().await - }) - } - - fn close<'a>(&'a self, _barrier: &'a NodeLogRotationBarrier) -> BoxFuture<'a, Result<()>> { - Box::pin(async { Ok(()) }) - } -} - -struct ImmediateTransport; - -impl NodeLogTransport for ImmediateTransport { - fn append<'a>( - &'a self, - _member: NodeId, - request: AppendRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - let last = request - .frames - .last() - .ok_or(Error::Node("empty test append"))?; - let frame = crab_ltx::inspect_node_frame(last.clone(), crab_ltx::Limits::default())?; - Ok(FollowerReceipt { - base_sequence: 1, - durable_through: frame.scope().node_sequence, - }) - }) - } - - fn seal<'a>( - &'a self, - _member: NodeId, - _request: SealRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("unused test seal")) }) - } - - fn tail<'a>( - &'a self, - _member: NodeId, - _request: TailRequest, - ) -> BoxFuture<'a, Result>> { - Box::pin(async { Err(Error::Node("unused test tail")) }) - } - - fn retire<'a>( - &'a self, - _member: NodeId, - _request: RetireRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("unused test retire")) }) - } -} - -fn session(byte: u8) -> SessionId { - SessionId::from_bytes([byte; 16]) -} - -fn node(byte: u8) -> NodeId { - NodeId::from_bytes([byte; 16]) -} - -fn lease() -> NodeLeaseGuard { - NodeLeaseGuard::new(1_000, 61_000).unwrap() -} - -fn capture() -> (tempfile::TempDir, crab_ltx::CaptureBatch) { - let directory = tempfile::TempDir::new().unwrap(); - let mut database = crab_ltx::Db::open( - &directory.path().join("durability.sqlite"), - crab_ltx::Limits::default(), - ) - .unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE items(value)")) - .unwrap(); - let cuts = database.capture().unwrap(); - database.close().unwrap(); - (directory, cuts) -} - -fn submission(cuts: &crab_ltx::CaptureBatch) -> NodeLogSubmission { - NodeLogSubmission::new( - ApplicationId::from_bytes([1; 16]), - CellId::from_bytes([2; 32]), - IncarnationId::from_bytes([3; 16]), - 4, - 5, - cuts, - ) - .unwrap() -} - -#[tokio::test] -async fn first_fsynced_ticket_activates_authority_before_fleet_proof() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - let transport: Arc = Arc::new(ImmediateTransport); - let shipper = NodeLogShipper::new( - gate.clone(), - Arc::clone(&transport), - crab_ltx::Limits::default(), - ) - .unwrap(); - let authority = Arc::new(RecordingAuthority::default()); - let durability = NodeDurability::new(gate, shipper, authority.clone(), transport, lease()); - let (_directory, cuts) = capture(); - let ticket = durability.submit(submission(&cuts)).await.unwrap(); - - let proof = durability.prove_fleet(ticket).await.unwrap(); - - assert_eq!(proof.source(), DurabilitySource::Fleet); - assert_eq!(authority.0.lock().unwrap().activations, vec![2]); -} - -#[tokio::test] -async fn fencing_cancels_a_blocked_fleet_activation() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - let transport: Arc = Arc::new(ImmediateTransport); - let shipper = NodeLogShipper::new( - gate.clone(), - Arc::clone(&transport), - crab_ltx::Limits::default(), - ) - .unwrap(); - let activation_started = Arc::new(Notify::new()); - let authority = Arc::new(BlockingAuthority { - activation_started: Some(Arc::clone(&activation_started)), - coverage_started: None, - }); - let lease = lease(); - let durability = Arc::new(NodeDurability::new( - gate, - shipper, - authority, - transport, - lease.clone(), - )); - let (_directory, cuts) = capture(); - let ticket = durability.submit(submission(&cuts)).await.unwrap(); - let task = tokio::spawn({ - let durability = Arc::clone(&durability); - async move { durability.prove_fleet(ticket).await } - }); - tokio::time::timeout( - std::time::Duration::from_secs(1), - activation_started.notified(), - ) - .await - .unwrap(); - lease.fence(); - assert!(matches!(task.await.unwrap(), Err(Error::Fenced))); - durability.shutdown().await.unwrap_err(); -} - -#[tokio::test] -async fn fencing_cancels_a_blocked_object_coverage_cas() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - let transport: Arc = Arc::new(ImmediateTransport); - let shipper = NodeLogShipper::new( - gate.clone(), - Arc::clone(&transport), - crab_ltx::Limits::default(), - ) - .unwrap(); - let coverage_started = Arc::new(Notify::new()); - let authority = Arc::new(BlockingAuthority { - activation_started: None, - coverage_started: Some(Arc::clone(&coverage_started)), - }); - let lease = lease(); - let durability = Arc::new(NodeDurability::new( - gate, - shipper, - authority, - transport, - lease.clone(), - )); - let (_directory, cuts) = capture(); - let ticket = durability.submit(submission(&cuts)).await.unwrap(); - let task = tokio::spawn({ - let durability = Arc::clone(&durability); - async move { durability.prove_object(ticket).await } - }); - tokio::time::timeout( - std::time::Duration::from_secs(1), - coverage_started.notified(), - ) - .await - .unwrap(); - lease.fence(); - assert!(matches!(task.await.unwrap(), Err(Error::Fenced))); - durability.shutdown().await.unwrap_err(); -} - -#[tokio::test] -async fn object_proof_advances_authoritative_contiguous_coverage() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - let transport: Arc = Arc::new(ImmediateTransport); - let shipper = NodeLogShipper::new( - gate.clone(), - Arc::clone(&transport), - crab_ltx::Limits::default(), - ) - .unwrap(); - let authority = Arc::new(RecordingAuthority::default()); - let durability = NodeDurability::new(gate, shipper, authority.clone(), transport, lease()); - let (_directory, cuts) = capture(); - let ticket = durability.submit(submission(&cuts)).await.unwrap(); - - let proof = durability.prove_object(ticket).await.unwrap(); - - assert_eq!(proof.source(), DurabilitySource::Object); - assert_eq!(authority.0.lock().unwrap().coverage, vec![(2, 1)]); -} - -#[tokio::test] -async fn concurrent_object_proofs_persist_the_complete_contiguous_prefix() { - for reverse in [false, true] { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - let transport: Arc = Arc::new(ImmediateTransport); - let shipper = NodeLogShipper::new( - gate.clone(), - Arc::clone(&transport), - crab_ltx::Limits::default(), - ) - .unwrap(); - let authority = Arc::new(RecordingAuthority(Mutex::new(AuthorityState { - yield_coverage: true, - ..AuthorityState::default() - }))); - let durability = - NodeDurability::new(gate.clone(), shipper, authority.clone(), transport, lease()); - let first = gate.issue(1).unwrap(); - let second = gate.issue(1).unwrap(); - let (left, right) = if reverse { - (second, first) - } else { - (first, second) - }; - - let (left, right) = tokio::join!( - durability.prove_object(left), - durability.prove_object(right), - ); - left.unwrap(); - right.unwrap(); - - assert_eq!(gate.tiered_through(), 2); - assert_eq!(authority.0.lock().unwrap().coverage.last(), Some(&(2, 2))); - } -} - -#[tokio::test] -async fn rejected_object_coverage_does_not_release_a_local_proof() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - let transport: Arc = Arc::new(ImmediateTransport); - let shipper = NodeLogShipper::new( - gate.clone(), - Arc::clone(&transport), - crab_ltx::Limits::default(), - ) - .unwrap(); - let authority = Arc::new(RecordingAuthority(Mutex::new(AuthorityState { - reject_coverage: true, - ..AuthorityState::default() - }))); - let durability = NodeDurability::new(gate.clone(), shipper, authority, transport, lease()); - let (_directory, cuts) = capture(); - let ticket = durability.submit(submission(&cuts)).await.unwrap(); - - assert!(durability.prove_object(ticket).await.is_err()); - assert_eq!(gate.tiered_through(), 0); - assert!( - tokio::time::timeout(std::time::Duration::from_millis(20), gate.prove(ticket)) - .await - .is_err() - ); -} - -#[tokio::test] -async fn rotation_threshold_tracks_issued_frames() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - let transport: Arc = Arc::new(ImmediateTransport); - let shipper = NodeLogShipper::new( - gate.clone(), - Arc::clone(&transport), - crab_ltx::Limits::default(), - ) - .unwrap(); - let authority = Arc::new(RecordingAuthority::default()); - let durability = NodeDurability::new(gate.clone(), shipper, authority, transport, lease()); - - assert!(!durability.needs_rotation(1)); - gate.issue(1).unwrap(); - assert!(durability.needs_rotation(1)); - assert!(!durability.needs_rotation(2)); - gate.stop_shipping(); - assert!(durability.needs_rotation(1_000_000)); -} - -#[tokio::test] -async fn shutdown_retries_after_object_coverage_and_is_idempotent() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - let transport: Arc = Arc::new(ImmediateTransport); - let shipper = NodeLogShipper::new( - gate.clone(), - Arc::clone(&transport), - crab_ltx::Limits::default(), - ) - .unwrap(); - let authority = Arc::new(RecordingAuthority::default()); - let durability = NodeDurability::new(gate, shipper, authority, transport, lease()); - let (_directory, cuts) = capture(); - let ticket = durability.submit(submission(&cuts)).await.unwrap(); - - assert!(matches!( - durability.shutdown().await, - Err(Error::PendingPublication) - )); - durability.prove_object(ticket).await.unwrap(); - durability.shutdown().await.unwrap(); - durability.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/src/node/lease.rs b/crates/crab-cell-runtime/src/node/lease.rs deleted file mode 100644 index cf929a74d..000000000 --- a/crates/crab-cell-runtime/src/node/lease.rs +++ /dev/null @@ -1,196 +0,0 @@ -//! Node lease guard and its renewal window. -use std::sync::{Arc, Mutex}; - -use tokio::sync::Notify; - -use crate::{Error, Result}; - -const MAX_NODE_LEASE_MS: i64 = 60_000; - -/// Process-wide monotonic guard for one published node-session lease. -/// -/// Only a successful authoritative create or refresh may advance the local -/// deadline. Expiry is terminal: a late object-store response cannot revive -/// dispatch or authorize a durability proof in the old process. -#[derive(Clone)] -pub struct NodeLeaseGuard { - inner: Arc, -} - -struct LeaseInner { - state: Mutex, - changed: Notify, -} - -struct LeaseState { - deadline: tokio::time::Instant, - fenced: bool, -} - -impl NodeLeaseGuard { - /// Starts a watchdog from one successfully published lease observation. - pub fn new(now_ms: i64, expires_at_ms: i64) -> Result { - let remaining = lease_remaining(now_ms, expires_at_ms)?; - let runtime = tokio::runtime::Handle::try_current().map_err(Error::RuntimeStart)?; - let inner = Arc::new(LeaseInner { - state: Mutex::new(LeaseState { - deadline: tokio::time::Instant::now() + remaining, - fenced: false, - }), - changed: Notify::new(), - }); - runtime.spawn(watchdog(Arc::clone(&inner))); - Ok(Self { inner }) - } - - /// Advances the deadline after an authoritative refresh succeeds. - pub fn renew(&self, now_ms: i64, expires_at_ms: i64) -> Result<()> { - let remaining = lease_remaining(now_ms, expires_at_ms)?; - let now = tokio::time::Instant::now(); - let next = now - .checked_add(remaining) - .ok_or(Error::Node("node lease deadline overflow"))?; - let mut state = self.lock()?; - if state.fenced || now >= state.deadline { - state.fenced = true; - drop(state); - self.inner.changed.notify_waiters(); - return Err(Error::Fenced); - } - if next <= state.deadline { - return Err(Error::Node("node lease deadline did not advance")); - } - state.deadline = next; - drop(state); - self.inner.changed.notify_waiters(); - Ok(()) - } - - /// Fails once the exact process can no longer prove a live session lease. - pub fn check(&self) -> Result<()> { - let mut state = self.lock()?; - if state.fenced || tokio::time::Instant::now() >= state.deadline { - state.fenced = true; - drop(state); - self.inner.changed.notify_waiters(); - return Err(Error::Fenced); - } - Ok(()) - } - - /// Returns the current lease time remaining, or zero after fencing. - #[must_use] - pub fn remaining(&self) -> std::time::Duration { - self.inner - .state - .lock() - .ok() - .filter(|state| !state.fenced) - .map_or(std::time::Duration::ZERO, |state| { - state - .deadline - .saturating_duration_since(tokio::time::Instant::now()) - }) - } - - /// Permanently closes the guard and wakes dispatch, output, and shutdown waiters. - pub fn fence(&self) { - if let Ok(mut state) = self.inner.state.lock() { - state.fenced = true; - } - self.inner.changed.notify_waiters(); - } - - /// Waits until expiry or an explicit terminal fence. - pub async fn wait_fenced(&self) { - loop { - let notified = self.inner.changed.notified(); - tokio::pin!(notified); - notified.as_mut().enable(); - if self.lock().map_or(true, |state| state.fenced) { - return; - } - notified.await; - } - } - - fn lock(&self) -> Result> { - self.inner - .state - .lock() - .map_err(|_| Error::Node("node lease guard poisoned")) - } -} - -async fn watchdog(inner: Arc) { - loop { - let notified = inner.changed.notified(); - tokio::pin!(notified); - notified.as_mut().enable(); - let deadline = match inner.state.lock() { - Ok(state) if state.fenced => return, - Ok(state) => state.deadline, - Err(_) => return, - }; - tokio::select! { - () = tokio::time::sleep_until(deadline) => { - if let Ok(mut state) = inner.state.lock() - && tokio::time::Instant::now() >= state.deadline - { - state.fenced = true; - drop(state); - inner.changed.notify_waiters(); - return; - } - } - () = &mut notified => {} - } - } -} - -fn lease_remaining(now_ms: i64, expires_at_ms: i64) -> Result { - let remaining_ms = expires_at_ms - .checked_sub(now_ms) - .filter(|remaining| (1..=MAX_NODE_LEASE_MS).contains(remaining)) - .ok_or(Error::Node("node lease bounds are invalid"))?; - Ok(std::time::Duration::from_millis(remaining_ms as u64)) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[tokio::test] - async fn expiry_is_terminal_and_late_renewal_cannot_revive_the_process() { - let guard = NodeLeaseGuard::new(1_000, 1_030).unwrap(); - - tokio::time::timeout(std::time::Duration::from_secs(1), guard.wait_fenced()) - .await - .unwrap(); - - assert!(matches!(guard.check(), Err(Error::Fenced))); - assert!(matches!(guard.renew(1_020, 1_050), Err(Error::Fenced))); - } - - #[tokio::test] - async fn authoritative_renewal_moves_the_monotonic_deadline_forward() { - let guard = NodeLeaseGuard::new(1_000, 1_050).unwrap(); - tokio::time::sleep(std::time::Duration::from_millis(20)).await; - guard.renew(1_020, 1_100).unwrap(); - tokio::time::sleep(std::time::Duration::from_millis(40)).await; - - guard.check().unwrap(); - guard.fence(); - guard.wait_fenced().await; - } - - #[tokio::test] - async fn invalid_or_nonadvancing_lease_is_rejected() { - assert!(NodeLeaseGuard::new(10, 10).is_err()); - assert!(NodeLeaseGuard::new(0, MAX_NODE_LEASE_MS + 1).is_err()); - - let guard = NodeLeaseGuard::new(1_000, 1_100).unwrap(); - assert!(guard.renew(1_050, 1_060).is_err()); - guard.fence(); - } -} diff --git a/crates/crab-cell-runtime/src/node/log.rs b/crates/crab-cell-runtime/src/node/log.rs deleted file mode 100644 index d6cfabbcb..000000000 --- a/crates/crab-cell-runtime/src/node/log.rs +++ /dev/null @@ -1,579 +0,0 @@ -//! Node log: durability gate, rotation barrier, and recovery overlays. -use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet}; -use std::path::Path; -use std::sync::{Arc, Mutex}; - -use futures_util::future::join_all; -use tokio::sync::Notify; - -use crate::identity::NodeId; -use crate::identity::SessionId; -use crate::node::log_transport::{NodeLogTransport, RetireRequest}; -use crate::node::{NodeDirectory, VersionedNodeAdvertisement}; -use crate::{Error, Result}; - -mod recovery; - -pub use recovery::*; - -pub(crate) const MAX_TICKET_FRAMES: u64 = 1_024; - -/// One actor-issued consecutive frame range awaiting a durability proof. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct CommitTicket { - leader_session: SessionId, - log_epoch: u64, - first_sequence: u64, - last_sequence: u64, -} - -impl CommitTicket { - /// Returns the leader session that issued the ticket. - #[must_use] - pub const fn leader_session(&self) -> SessionId { - self.leader_session - } - - /// Returns the node-log epoch the commit was written under. - #[must_use] - pub const fn log_epoch(&self) -> u64 { - self.log_epoch - } - - /// Returns the first node-log sequence the ticket covers. - #[must_use] - pub const fn first_sequence(&self) -> u64 { - self.first_sequence - } - - /// Returns the last node-log sequence the ticket covers. - #[must_use] - pub const fn last_sequence(&self) -> u64 { - self.last_sequence - } -} - -/// Durable path that authorized release of one committed result. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum DurabilitySource { - /// The commit was covered by the enrolled node-log lane. - Fleet, - /// The commit was covered by object storage. - Object, -} - -/// Non-forgeable proof issued by the gate after one complete path wins. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct DurabilityProof { - ticket: CommitTicket, - source: DurabilitySource, -} - -/// Proof that one gate stopped issuance after every old-epoch ticket became object-covered. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct NodeLogRotationBarrier { - leader_session: SessionId, - log_epoch: u64, - members: Vec, - covered_through: u64, -} - -/// New authoritative enrollment and inactive durability gate after rotation. -pub struct RotatedNodeLog { - /// Enrollment the rotation installed. - pub enrollment: VersionedNodeAdvertisement, - /// Inactive durability gate the rotated log starts from. - pub gate: DurabilityGate, -} - -impl NodeLogRotationBarrier { - /// Returns the session the barrier was issued for. - #[must_use] - pub const fn leader_session(&self) -> SessionId { - self.leader_session - } - - /// Returns the sealed log epoch. - #[must_use] - pub const fn log_epoch(&self) -> u64 { - self.log_epoch - } - - /// Returns the members the rotation enrolled. - #[must_use] - pub fn members(&self) -> &[NodeId] { - &self.members - } - - /// Returns the highest sequence the sealed log covered. - #[must_use] - pub const fn covered_through(&self) -> u64 { - self.covered_through - } -} - -impl DurabilityProof { - /// Returns the ticket this proof covers. - #[must_use] - pub const fn ticket(&self) -> CommitTicket { - self.ticket - } - - /// Returns which durability path issued the proof. - #[must_use] - pub const fn source(&self) -> DurabilitySource { - self.source - } -} - -/// Coordinates object and follower durability without exposing forgeable ACKs. -/// -/// The gate owns one leader session and one log epoch. Tickets are issued in -/// strict node-sequence order. Fleet proof is disabled until the session's -/// `active=false -> active=true` CAS has been observed by the caller. -#[derive(Clone)] -pub struct DurabilityGate { - inner: Arc>, - changed: Arc, -} - -struct GateState { - leader_session: SessionId, - log_epoch: u64, - members: HashSet, - follower_through: HashMap, - object_covered: BTreeSet, - tiered_through: u64, - next_sequence: u64, - fleet_active: bool, - rotating: bool, - fenced: bool, -} - -impl DurabilityGate { - /// Creates one inactive gate for the exact recruited follower ensemble. - pub fn new( - leader_session: SessionId, - leader_node: NodeId, - log_epoch: u64, - members: impl IntoIterator, - ) -> Result { - let members = members.into_iter().collect::>(); - if leader_session.as_bytes().iter().all(|byte| *byte == 0) - || leader_node.as_bytes().iter().all(|byte| *byte == 0) - || log_epoch == 0 - || members.is_empty() - || members.contains(&leader_node) - || members - .iter() - .any(|member| member.as_bytes().iter().all(|byte| *byte == 0)) - { - return Err(Error::Node("invalid node-log ensemble")); - } - let follower_through = members.iter().map(|member| (*member, 0)).collect(); - Ok(Self { - inner: Arc::new(Mutex::new(GateState { - leader_session, - log_epoch, - members, - follower_through, - object_covered: BTreeSet::new(), - tiered_through: 0, - next_sequence: 1, - fleet_active: false, - rotating: false, - fenced: false, - })), - changed: Arc::new(Notify::new()), - }) - } - - /// Allocates the next consecutive frame range after SQLite capture. - pub fn issue(&self, frame_count: u64) -> Result { - if !(1..=MAX_TICKET_FRAMES).contains(&frame_count) { - return Err(Error::Node("invalid node-log ticket size")); - } - let mut state = self.lock()?; - if state.fenced { - return Err(Error::Fenced); - } - if state.rotating { - return Err(Error::Node("node log is rotating")); - } - let first_sequence = state.next_sequence; - let last_sequence = first_sequence - .checked_add(frame_count - 1) - .ok_or(Error::Node("node sequence overflow"))?; - state.next_sequence = last_sequence - .checked_add(1) - .ok_or(Error::Node("node sequence overflow"))?; - Ok(CommitTicket { - leader_session: state.leader_session, - log_epoch: state.log_epoch, - first_sequence, - last_sequence, - }) - } - - pub(crate) fn preview(&self, frame_count: u64) -> Result { - if !(1..=MAX_TICKET_FRAMES).contains(&frame_count) { - return Err(Error::Node("invalid node-log ticket size")); - } - let state = self.lock()?; - if state.fenced { - return Err(Error::Fenced); - } - if state.rotating { - return Err(Error::Node("node log is rotating")); - } - let first_sequence = state.next_sequence; - let last_sequence = first_sequence - .checked_add(frame_count - 1) - .ok_or(Error::Node("node sequence overflow"))?; - Ok(CommitTicket { - leader_session: state.leader_session, - log_epoch: state.log_epoch, - first_sequence, - last_sequence, - }) - } - - pub(crate) fn commit(&self, ticket: CommitTicket) -> Result<()> { - let mut state = self.lock()?; - if state.fenced { - return Err(Error::Fenced); - } - if state.rotating { - return Err(Error::Node("node log is rotating")); - } - if ticket.leader_session != state.leader_session - || ticket.log_epoch != state.log_epoch - || ticket.first_sequence != state.next_sequence - || ticket.first_sequence == 0 - || ticket.first_sequence > ticket.last_sequence - || ticket.last_sequence.saturating_sub(ticket.first_sequence) >= MAX_TICKET_FRAMES - { - return Err(Error::Node("node-log ticket reservation changed")); - } - state.next_sequence = ticket - .last_sequence - .checked_add(1) - .ok_or(Error::Node("node sequence overflow"))?; - Ok(()) - } - - pub(crate) fn shipping_scope(&self) -> Result<(SessionId, u64, Vec)> { - let state = self.lock()?; - if state.fenced || state.rotating { - return Err(Error::Fenced); - } - let mut members = state.members.iter().copied().collect::>(); - members.sort_unstable_by(|left, right| left.as_bytes().cmp(right.as_bytes())); - Ok((state.leader_session, state.log_epoch, members)) - } - - pub(crate) fn stop_shipping(&self) { - if let Ok(mut state) = self.inner.lock() { - state.fleet_active = false; - state.rotating = true; - } - self.changed.notify_waiters(); - } - - /// Enables fleet proofs only after the authoritative active CAS succeeds. - pub fn activate_fleet(&self) -> Result<()> { - let mut state = self.lock()?; - if state.fenced { - return Err(Error::Fenced); - } - if state.rotating { - return Err(Error::Node("node log is rotating")); - } - state.fleet_active = true; - drop(state); - self.changed.notify_waiters(); - Ok(()) - } - - /// Records one authenticated follower's fsynced contiguous watermark. - pub fn acknowledge(&self, member: NodeId, durable_through: u64) -> Result<()> { - let mut state = self.lock()?; - if state.fenced { - return Err(Error::Fenced); - } - if state.rotating { - return Err(Error::Node("node log is rotating")); - } - let next_sequence = state.next_sequence; - let current = state - .follower_through - .get_mut(&member) - .ok_or(Error::Node("node-log ACK came from a non-member"))?; - if durable_through < *current || durable_through >= next_sequence { - return Err(Error::Node("node-log ACK watermark is invalid")); - } - *current = durable_through; - drop(state); - self.changed.notify_waiters(); - Ok(()) - } - - /// Waits until every enrolled follower has fsynced the complete ticket. - /// - /// This does not enable fleet durability. The caller must first complete - /// the authoritative inactive-to-active CAS and then call `activate_fleet`. - pub async fn wait_followers(&self, ticket: CommitTicket) -> Result<()> { - loop { - let notified = self.changed.notified(); - tokio::pin!(notified); - notified.as_mut().enable(); - { - let state = self.lock()?; - validate_ticket(&state, ticket)?; - if state.fenced { - return Err(Error::Fenced); - } - if state.rotating { - return Err(Error::Node("node log is rotating")); - } - if state.members.iter().all(|member| { - state - .follower_through - .get(member) - .is_some_and(|through| *through >= ticket.last_sequence) - }) { - return Ok(()); - } - } - notified.await; - } - } - - /// Marks exactly the frame range now reachable through an authoritative root. - pub fn prove_object(&self, ticket: CommitTicket) -> Result { - let mut state = self.lock()?; - validate_ticket(&state, ticket)?; - if state.fenced { - return Err(Error::Fenced); - } - for sequence in ticket.first_sequence..=ticket.last_sequence { - state.object_covered.insert(sequence); - } - while state.object_covered.contains(&(state.tiered_through + 1)) { - state.tiered_through += 1; - } - let tiered_through = state.tiered_through; - drop(state); - self.changed.notify_waiters(); - Ok(tiered_through) - } - - pub(crate) fn preview_object(&self, ticket: CommitTicket) -> Result { - let state = self.lock()?; - validate_ticket(&state, ticket)?; - if state.fenced { - return Err(Error::Fenced); - } - let mut tiered_through = state.tiered_through; - while let Some(next) = tiered_through.checked_add(1) { - if state.object_covered.contains(&next) - || (ticket.first_sequence..=ticket.last_sequence).contains(&next) - { - tiered_through = next; - } else { - break; - } - } - Ok(tiered_through) - } - - /// Returns the highest sequence the follower lane has made durable. - #[must_use] - pub fn tiered_through(&self) -> u64 { - self.lock().map_or(0, |state| state.tiered_through) - } - - /// Returns the highest sequence issued to the follower lane. - #[must_use] - pub fn issued_through(&self) -> u64 { - self.lock() - .map_or(0, |state| state.next_sequence.saturating_sub(1)) - } - - /// Stops ticket issuance after the entire old epoch is object-covered. - pub fn begin_rotation(&self) -> Result { - let mut state = self.lock()?; - if state.fenced { - return Err(Error::Fenced); - } - let issued_through = state.next_sequence.saturating_sub(1); - if state.tiered_through != issued_through { - return Err(Error::PendingPublication); - } - state.rotating = true; - state.fleet_active = false; - let mut members = state.members.iter().copied().collect::>(); - members.sort_unstable_by(|left, right| left.as_bytes().cmp(right.as_bytes())); - let barrier = NodeLogRotationBarrier { - leader_session: state.leader_session, - log_epoch: state.log_epoch, - members, - covered_through: state.tiered_through, - }; - drop(state); - self.changed.notify_waiters(); - Ok(barrier) - } - - /// Waits until either complete durability path covers the whole ticket. - pub async fn prove(&self, ticket: CommitTicket) -> Result { - loop { - let notified = self.changed.notified(); - tokio::pin!(notified); - notified.as_mut().enable(); - if let Some(proof) = self.proof(ticket)? { - return Ok(proof); - } - notified.await; - } - } - - /// Permanently rejects new tickets and wakes every waiting command. - pub fn fence(&self) { - if let Ok(mut state) = self.inner.lock() { - state.fenced = true; - } - self.changed.notify_waiters(); - } - - fn proof(&self, ticket: CommitTicket) -> Result> { - let state = self.lock()?; - validate_ticket(&state, ticket)?; - if state.fenced { - return Err(Error::Fenced); - } - if (ticket.first_sequence..=ticket.last_sequence) - .all(|sequence| state.object_covered.contains(&sequence)) - { - return Ok(Some(DurabilityProof { - ticket, - source: DurabilitySource::Object, - })); - } - if state.fleet_active - && state.members.iter().all(|member| { - state - .follower_through - .get(member) - .is_some_and(|through| *through >= ticket.last_sequence) - }) - { - return Ok(Some(DurabilityProof { - ticket, - source: DurabilitySource::Fleet, - })); - } - Ok(None) - } - - fn lock(&self) -> Result> { - self.inner - .lock() - .map_err(|_| Error::Node("node-log durability gate poisoned")) - } -} - -/// Best-effort retires old follower lanes before CASing a newly selected log epoch. -/// -/// Unreachable followers may retain inert data, but cannot block rotation. -/// A CAS failure leaves the old gate closed to new tickets. Retrying is safe: -/// the barrier and follower retire markers are exact and idempotent. -pub async fn rotate_node_log( - directory: &NodeDirectory, - transport: Arc, - observed: &VersionedNodeAdvertisement, - gate: &DurabilityGate, - required_follower_bytes: u64, - live_node_limit: usize, - now_ms: i64, -) -> Result { - let barrier = gate.begin_rotation()?; - retire_node_log(transport, &barrier).await?; - let enrollment = directory - .rotate_log( - observed, - &barrier, - required_follower_bytes, - live_node_limit, - now_ms, - ) - .await?; - let log = enrollment - .advertisement() - .log() - .ok_or(Error::Node("rotated node session lost its log"))?; - let gate = DurabilityGate::new( - enrollment.advertisement().session(), - enrollment.advertisement().node(), - log.epoch(), - log.members().iter().copied(), - )?; - Ok(RotatedNodeLog { enrollment, gate }) -} - -/// Best-effort retires a covered epoch and CAS-clears it for clean withdrawal. -/// -/// Unreachable followers may retain inert bytes. The authoritative clear -/// rejects every later append before the session can be withdrawn. -pub async fn close_node_log( - directory: &NodeDirectory, - transport: Arc, - observed: &VersionedNodeAdvertisement, - gate: &DurabilityGate, - now_ms: i64, -) -> Result { - let barrier = gate.begin_rotation()?; - retire_node_log(transport, &barrier).await?; - directory.close_log(observed, &barrier, now_ms).await -} - -pub(crate) async fn retire_node_log( - transport: Arc, - barrier: &NodeLogRotationBarrier, -) -> Result<()> { - let retirements = join_all(barrier.members().iter().map(|member| { - let transport = Arc::clone(&transport); - let member = *member; - let request = RetireRequest { - leader_session: barrier.leader_session(), - log_epoch: barrier.log_epoch(), - covered_through: barrier.covered_through(), - }; - async move { transport.retire(member, request).await } - })) - .await; - let expected_base = barrier.covered_through().saturating_add(1); - for receipt in retirements.into_iter().flatten() { - if receipt.base_sequence != expected_base - || receipt.durable_through != barrier.covered_through() - { - return Err(Error::Node("follower retire receipt differs")); - } - } - Ok(()) -} - -fn validate_ticket(state: &GateState, ticket: CommitTicket) -> Result<()> { - if ticket.leader_session != state.leader_session - || ticket.log_epoch != state.log_epoch - || ticket.first_sequence == 0 - || ticket.first_sequence > ticket.last_sequence - || ticket.last_sequence >= state.next_sequence - { - return Err(Error::Node("node-log ticket is not owned by this gate")); - } - Ok(()) -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/node/log/recovery.rs b/crates/crab-cell-runtime/src/node/log/recovery.rs deleted file mode 100644 index d07bcc57c..000000000 --- a/crates/crab-cell-runtime/src/node/log/recovery.rs +++ /dev/null @@ -1,321 +0,0 @@ -//! Rebuilding a recovered Cell tail from published node-log witnesses. -//! -//! The overlays here are what a restart replays: each one names the exact -//! published base, the retained witness frames above it, and the cell tail -//! they prove, so a recovery cannot invent frames it never verified. - -use super::*; - -/// Exact published Cell state used to validate a recovered node-log witness. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct RecoveryBase { - /// Application the recovered Cell belongs to. - pub application: [u8; 16], - /// Cell epoch the recovered root was published under. - pub cell_epoch: u64, - /// Published root the witness is validated against. - pub root: crab_ltx::RootRef, -} - -/// One Cell's verified tail selected from a complete node-log witness. -pub struct RecoveredCellTail { - /// Application the tail belongs to. - pub application: [u8; 16], - /// Cell epoch the tail extends. - pub cell_epoch: u64, - /// First node-log sequence the tail covers. - pub first_node_sequence: u64, - /// Last node-log sequence the tail covers. - pub last_node_sequence: u64, - /// Verified overlay the tail publishes. - pub overlay: crab_ltx::RecoveryOverlay, -} - -/// Splits one complete, contiguous witness into exact per-Cell overlays. -pub fn build_recovery_overlays( - frames: Vec, - bases: &[RecoveryBase], - limits: crab_ltx::Limits, -) -> Result> { - let first = frames - .first() - .ok_or(Error::Node("recovery witness is empty"))?; - let leader = first.scope().leader_session; - let log_epoch = first.scope().log_epoch; - if frames - .iter() - .any(|frame| frame.scope().leader_session != leader || frame.scope().log_epoch != log_epoch) - || !frames.windows(2).all(|pair| { - pair[0].scope().node_sequence.checked_add(1) == Some(pair[1].scope().node_sequence) - }) - { - return Err(Error::Node("recovery witness is not contiguous")); - } - - type Key = ([u8; 16], [u8; 32], [u8; 16], u64); - let base_by_key = bases - .iter() - .map(|base| { - ( - ( - base.application, - base.root.cell, - base.root.incarnation, - base.cell_epoch, - ), - *base, - ) - }) - .collect::>(); - if base_by_key.len() != bases.len() { - return Err(Error::Node("recovery bases contain duplicate Cell scope")); - } - - let mut grouped = BTreeMap::>::new(); - for frame in frames { - let scope = frame.scope(); - let key = ( - scope.application, - scope.cell, - scope.incarnation, - scope.cell_epoch, - ); - if !base_by_key.contains_key(&key) { - return Err(Error::Node("recovery frame has no exact published base")); - } - grouped.entry(key).or_default().push(frame); - } - - let mut recovered = Vec::with_capacity(grouped.len()); - for (key, mut cell_frames) in grouped { - let base = base_by_key - .get(&key) - .ok_or(Error::Node("recovery frame base disappeared"))?; - // A different Cell can hold back the global coverage watermark even - // after this Cell's root CAS. Discard only this exact root's covered - // prefix; a commit/position mismatch must never hide an uncovered cut. - for frame in &cell_frames { - let covered = frame.scope().commit_sequence <= base.root.commit_sequence; - let position = frame.segment().position(); - if covered != (position.txid <= base.root.position.txid) - || (position.txid == base.root.position.txid && position != base.root.position) - { - return Err(Error::Node("recovery frame disagrees with published base")); - } - } - cell_frames.retain(|frame| frame.scope().commit_sequence > base.root.commit_sequence); - if cell_frames.is_empty() { - continue; - } - let first_frame = cell_frames - .first() - .ok_or(Error::Node("recovery Cell tail is empty"))?; - let last = cell_frames - .last() - .ok_or(Error::Node("recovery Cell tail is empty"))?; - let first_commit = base - .root - .commit_sequence - .checked_add(1) - .ok_or(Error::Node("recovery commit sequence overflow"))?; - if first_frame.scope().commit_sequence != first_commit - || !cell_frames.windows(2).all(|pair| { - let left = pair[0].scope().commit_sequence; - let right = pair[1].scope().commit_sequence; - right == left || left.checked_add(1) == Some(right) - }) - { - return Err(Error::Node("recovery Cell commit sequence has a gap")); - } - let entries = cell_frames - .iter() - .map(|frame| { - crab_ltx::bundle::BundleEntry::for_cell( - base.root.cell, - base.root.incarnation, - frame.segment().clone(), - frame.body().to_vec(), - ) - }) - .collect(); - let bundle = crab_ltx::bundle::Bundle::encode(entries, limits)?; - recovered.push(RecoveredCellTail { - application: base.application, - cell_epoch: base.cell_epoch, - first_node_sequence: first_frame.scope().node_sequence, - last_node_sequence: last.scope().node_sequence, - overlay: crab_ltx::RecoveryOverlay::new( - base.root, - bundle, - last.segment().position(), - last.scope().commit_sequence, - ), - }); - } - Ok(recovered) -} - -/// Splits a complete witness into per-Cell overlays without retaining segment -/// bodies in the heap. Each Cell gets one scratch-backed bundle builder and -/// every frame body is released after its row is appended. -pub fn build_recovery_overlays_file_backed( - frames: Vec, - bases: &[RecoveryBase], - limits: crab_ltx::Limits, - scratch: &Path, -) -> Result> { - build_recovery_overlays_file_backed_stream(frames.into_iter().map(Ok), bases, limits, scratch) -} - -/// Streaming variant used by the bounded witness reader. The iterator may -/// yield one verified frame at a time; no complete witness is retained. -pub fn build_recovery_overlays_file_backed_stream( - frames: I, - bases: &[RecoveryBase], - limits: crab_ltx::Limits, - scratch: &Path, -) -> Result> -where - I: IntoIterator>, -{ - let mut frames = frames.into_iter(); - let first = frames - .next() - .ok_or(Error::Node("recovery witness is empty"))??; - let leader = first.scope().leader_session; - let log_epoch = first.scope().log_epoch; - let frames = std::iter::once(Ok(first)).chain(frames); - - type Key = ([u8; 16], [u8; 32], [u8; 16], u64); - let base_by_key = bases - .iter() - .map(|base| { - ( - ( - base.application, - base.root.cell, - base.root.incarnation, - base.cell_epoch, - ), - *base, - ) - }) - .collect::>(); - if base_by_key.len() != bases.len() { - return Err(Error::Node("recovery bases contain duplicate Cell scope")); - } - - struct CellBuilder { - base: RecoveryBase, - bundle: crab_ltx::bundle::BundleBuilder, - first_node_sequence: u64, - last_node_sequence: u64, - last_commit_sequence: u64, - final_position: crab_ltx::Position, - } - - let mut previous_node_sequence = None::; - let mut grouped = BTreeMap::::new(); - for frame in frames { - let frame = frame?; - let scope = frame.scope(); - if scope.leader_session != leader || scope.log_epoch != log_epoch { - return Err(Error::Node("recovery witness has mixed sessions")); - } - if previous_node_sequence - .is_some_and(|previous| previous.checked_add(1) != Some(scope.node_sequence)) - { - return Err(Error::Node("recovery witness is not contiguous")); - } - previous_node_sequence = Some(scope.node_sequence); - let key = ( - scope.application, - scope.cell, - scope.incarnation, - scope.cell_epoch, - ); - let base = base_by_key - .get(&key) - .copied() - .ok_or(Error::Node("recovery frame has no exact published base"))?; - let covered = scope.commit_sequence <= base.root.commit_sequence; - let position = frame.segment().position(); - if covered != (position.txid <= base.root.position.txid) - || (position.txid == base.root.position.txid && position != base.root.position) - { - return Err(Error::Node("recovery frame disagrees with published base")); - } - if covered { - if grouped.contains_key(&key) { - return Err(Error::Node( - "recovery Cell covered prefix follows uncovered tail", - )); - } - continue; - } - - if let Some(cell) = grouped.get_mut(&key) { - if scope.commit_sequence != cell.last_commit_sequence - && cell.last_commit_sequence.checked_add(1) != Some(scope.commit_sequence) - { - return Err(Error::Node("recovery Cell commit sequence has a gap")); - } - cell.bundle.push(crab_ltx::bundle::BundleEntry::for_cell( - base.root.cell, - base.root.incarnation, - frame.segment().clone(), - frame.body().to_vec(), - ))?; - cell.last_node_sequence = scope.node_sequence; - cell.last_commit_sequence = scope.commit_sequence; - cell.final_position = position; - continue; - } - - let first_commit = base - .root - .commit_sequence - .checked_add(1) - .ok_or(Error::Node("recovery commit sequence overflow"))?; - if scope.commit_sequence != first_commit { - return Err(Error::Node("recovery Cell commit sequence has a gap")); - } - let mut bundle = crab_ltx::bundle::BundleBuilder::new_temp(scratch, limits)?; - bundle.push(crab_ltx::bundle::BundleEntry::for_cell( - base.root.cell, - base.root.incarnation, - frame.segment().clone(), - frame.body().to_vec(), - ))?; - grouped.insert( - key, - CellBuilder { - base, - bundle, - first_node_sequence: scope.node_sequence, - last_node_sequence: scope.node_sequence, - last_commit_sequence: scope.commit_sequence, - final_position: position, - }, - ); - } - - grouped - .into_values() - .map(|cell| { - let bundle = cell.bundle.finish()?; - Ok(RecoveredCellTail { - application: cell.base.application, - cell_epoch: cell.base.cell_epoch, - first_node_sequence: cell.first_node_sequence, - last_node_sequence: cell.last_node_sequence, - overlay: crab_ltx::RecoveryOverlay::new( - cell.base.root, - bundle, - cell.final_position, - cell.last_commit_sequence, - ), - }) - }) - .collect() -} diff --git a/crates/crab-cell-runtime/src/node/log/tests.rs b/crates/crab-cell-runtime/src/node/log/tests.rs deleted file mode 100644 index 1f069ef17..000000000 --- a/crates/crab-cell-runtime/src/node/log/tests.rs +++ /dev/null @@ -1,369 +0,0 @@ -use super::*; -use bytes::Bytes; - -fn verified_frame( - sequence: u64, - commit_sequence: u64, - cell: [u8; 32], - incarnation: [u8; 16], - segment: &crab_ltx::LocalSegment, -) -> crab_ltx::VerifiedNodeFrame { - crab_ltx::encode_node_frame( - crab_ltx::NodeFrameScope { - leader_session: [1; 16], - log_epoch: 2, - node_sequence: sequence, - application: [9; 16], - cell, - incarnation, - cell_epoch: 3, - commit_sequence, - }, - segment.info().clone(), - Bytes::from(std::fs::read(segment.path()).unwrap()), - crab_ltx::Limits::default(), - ) - .unwrap() -} - -fn session(byte: u8) -> SessionId { - SessionId::from_bytes([byte; 16]) -} - -fn node(byte: u8) -> NodeId { - NodeId::from_bytes([byte; 16]) -} - -#[tokio::test] -async fn fleet_requires_activation_and_every_follower() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(3), node(4)]).unwrap(); - let ticket = gate.issue(2).unwrap(); - gate.acknowledge(node(3), 2).unwrap(); - gate.acknowledge(node(4), 2).unwrap(); - assert!(gate.proof(ticket).unwrap().is_none()); - gate.activate_fleet().unwrap(); - assert_eq!( - gate.prove(ticket).await.unwrap().source(), - DurabilitySource::Fleet - ); -} - -#[tokio::test] -async fn follower_wait_does_not_activate_fleet_proof() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(3), node(4)]).unwrap(); - let ticket = gate.issue(2).unwrap(); - gate.acknowledge(node(3), 2).unwrap(); - gate.acknowledge(node(4), 2).unwrap(); - - gate.wait_followers(ticket).await.unwrap(); - - assert!(gate.proof(ticket).unwrap().is_none()); -} - -#[tokio::test] -async fn follower_wait_wakes_after_ack_arrives() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(3)]).unwrap(); - let ticket = gate.issue(1).unwrap(); - let waiter = { - let gate = gate.clone(); - tokio::spawn(async move { gate.wait_followers(ticket).await }) - }; - tokio::task::yield_now().await; - gate.acknowledge(node(3), 1).unwrap(); - tokio::time::timeout(std::time::Duration::from_secs(1), waiter) - .await - .unwrap() - .unwrap() - .unwrap(); -} - -#[tokio::test] -async fn object_proof_wins_independently_and_watermark_stays_contiguous() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(3)]).unwrap(); - let first = gate.issue(1).unwrap(); - let second = gate.issue(1).unwrap(); - assert_eq!(gate.prove_object(second).unwrap(), 0); - assert_eq!( - gate.prove(second).await.unwrap().source(), - DurabilitySource::Object - ); - assert_eq!(gate.prove_object(first).unwrap(), 2); - assert_eq!(gate.tiered_through(), 2); -} - -#[tokio::test] -async fn proof_wakes_after_object_coverage_arrives() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(3)]).unwrap(); - let ticket = gate.issue(1).unwrap(); - let waiter = { - let gate = gate.clone(); - tokio::spawn(async move { gate.prove(ticket).await }) - }; - tokio::task::yield_now().await; - gate.prove_object(ticket).unwrap(); - assert_eq!( - tokio::time::timeout(std::time::Duration::from_secs(1), waiter) - .await - .unwrap() - .unwrap() - .unwrap() - .source(), - DurabilitySource::Object - ); -} - -#[tokio::test] -async fn rotation_waits_for_object_coverage_and_closes_the_old_gate() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(4), node(3)]).unwrap(); - let ticket = gate.issue(2).unwrap(); - - assert!(matches!( - gate.begin_rotation(), - Err(Error::PendingPublication) - )); - gate.prove_object(ticket).unwrap(); - let barrier = gate.begin_rotation().unwrap(); - assert_eq!(barrier.leader_session(), session(1)); - assert_eq!(barrier.log_epoch(), 2); - assert_eq!(barrier.members(), [node(3), node(4)]); - assert_eq!(barrier.covered_through(), 2); - assert_eq!(gate.begin_rotation().unwrap(), barrier); - assert!(gate.issue(1).is_err()); - assert!(gate.activate_fleet().is_err()); - assert!(gate.acknowledge(node(3), 2).is_err()); - assert_eq!( - gate.prove(ticket).await.unwrap().source(), - DurabilitySource::Object - ); -} - -#[tokio::test] -async fn fencing_wakes_waiters_and_rejects_late_acks() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(3)]).unwrap(); - let ticket = gate.issue(1).unwrap(); - let waiter = { - let gate = gate.clone(); - tokio::spawn(async move { gate.prove(ticket).await }) - }; - gate.fence(); - assert!(matches!(waiter.await.unwrap(), Err(Error::Fenced))); - assert!(matches!(gate.acknowledge(node(3), 1), Err(Error::Fenced))); -} - -#[test] -fn fencing_rejects_late_object_coverage() { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(3)]).unwrap(); - let ticket = gate.issue(1).unwrap(); - gate.fence(); - - assert!(matches!(gate.prove_object(ticket), Err(Error::Fenced))); - assert_eq!(gate.tiered_through(), 0); -} - -#[test] -fn recovery_witness_splits_interleaved_cells_against_exact_bases() { - let limits = crab_ltx::Limits::default(); - let directory = tempfile::TempDir::new().unwrap(); - let mut left = crab_ltx::Db::open(&directory.path().join("left.sqlite"), limits).unwrap(); - let mut right = crab_ltx::Db::open(&directory.path().join("right.sqlite"), limits).unwrap(); - for database in [&mut left, &mut right] { - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ - INSERT INTO events(body) VALUES ('base')", - ) - }) - .unwrap(); - } - let left_base = left.capture().unwrap(); - let right_base = right.capture().unwrap(); - for database in [&mut left, &mut right] { - database - .transaction(|transaction| { - transaction.execute("INSERT INTO events(body) VALUES ('tail')", [])?; - Ok(()) - }) - .unwrap(); - } - let left_tail = left.capture().unwrap(); - let right_tail = right.capture().unwrap(); - let left_cell = [4; 32]; - let left_incarnation = [5; 16]; - let right_cell = [6; 32]; - let right_incarnation = [7; 16]; - let frames = vec![ - verified_frame( - 1, - 2, - left_cell, - left_incarnation, - left_tail.segments.first().unwrap(), - ), - verified_frame( - 2, - 2, - right_cell, - right_incarnation, - right_tail.segments.first().unwrap(), - ), - ]; - let bases = [ - RecoveryBase { - application: [9; 16], - cell_epoch: 3, - root: crab_ltx::RootRef { - cell: left_cell, - incarnation: left_incarnation, - digest: [10; 32], - position: left_base.position, - commit_sequence: 1, - }, - }, - RecoveryBase { - application: [9; 16], - cell_epoch: 3, - root: crab_ltx::RootRef { - cell: right_cell, - incarnation: right_incarnation, - digest: [11; 32], - position: right_base.position, - commit_sequence: 1, - }, - }, - ]; - - let recovered = build_recovery_overlays(frames.clone(), &bases, limits).unwrap(); - assert_eq!(recovered.len(), 2); - assert_eq!(recovered[0].first_node_sequence, 1); - assert_eq!(recovered[0].overlay.final_position(), left_tail.position); - assert_eq!(recovered[1].last_node_sequence, 2); - assert_eq!(recovered[1].overlay.final_position(), right_tail.position); - // The second Cell can reach object storage before the first, leaving - // its already-published commit above the shared node watermark. - let mut advanced_bases = bases; - advanced_bases[1].root.position = right_tail.position; - advanced_bases[1].root.commit_sequence = 2; - let recovered = build_recovery_overlays(frames.clone(), &advanced_bases, limits).unwrap(); - assert_eq!(recovered.len(), 1); - assert_eq!(recovered[0].overlay.predecessor().cell, left_cell); - advanced_bases[1].root.position.checksum ^= 1; - assert!(build_recovery_overlays(frames.clone(), &advanced_bases, limits).is_err()); - advanced_bases[1].root.position = right_base.position; - assert!(build_recovery_overlays(frames, &advanced_bases, limits).is_err()); - left.close().unwrap(); - right.close().unwrap(); -} - -#[test] -fn recovery_witness_splits_two_interleaved_cuts_from_one_thousand_cells() { - const CELLS: usize = 1_000; - - let limits = crab_ltx::Limits::default(); - let directory = tempfile::TempDir::new().unwrap(); - let mut database = crab_ltx::Db::open(&directory.path().join("source.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ - INSERT INTO events(body) VALUES ('base')", - ) - }) - .unwrap(); - let base = database.capture().unwrap(); - database - .transaction(|transaction| { - transaction.execute("INSERT INTO events(body) VALUES ('first')", [])?; - Ok(()) - }) - .unwrap(); - let first = database.capture().unwrap(); - database - .transaction(|transaction| { - transaction.execute("INSERT INTO events(body) VALUES ('second')", [])?; - Ok(()) - }) - .unwrap(); - let second = database.capture().unwrap(); - let incarnation = [5; 16]; - let mut bases = Vec::with_capacity(CELLS); - let mut cells = Vec::with_capacity(CELLS); - for index in 0..CELLS { - let mut cell = [0_u8; 32]; - cell[..8].copy_from_slice(&(index as u64 + 1).to_be_bytes()); - cells.push(cell); - bases.push(RecoveryBase { - application: [9; 16], - cell_epoch: 3, - root: crab_ltx::RootRef { - cell, - incarnation, - digest: *blake3::hash(&cell).as_bytes(), - position: base.position, - commit_sequence: 1, - }, - }); - } - let mut frames = Vec::with_capacity(CELLS * 2); - for (index, cell) in cells.iter().copied().enumerate() { - frames.push(verified_frame( - index as u64 + 1, - 2, - cell, - incarnation, - first.segments.first().unwrap(), - )); - } - for (index, cell) in cells.iter().copied().enumerate() { - frames.push(verified_frame( - CELLS as u64 + index as u64 + 1, - 3, - cell, - incarnation, - second.segments.first().unwrap(), - )); - } - - let recovered = build_recovery_overlays(frames.clone(), &bases, limits).unwrap(); - - assert_eq!(recovered.len(), CELLS); - let file_backed = - build_recovery_overlays_file_backed(frames.clone(), &bases, limits, directory.path()) - .unwrap(); - assert_eq!(file_backed.len(), recovered.len()); - for (memory, disk) in recovered.iter().zip(&file_backed) { - assert_eq!(memory.first_node_sequence, disk.first_node_sequence); - assert_eq!(memory.last_node_sequence, disk.last_node_sequence); - assert_eq!( - memory.overlay.final_position(), - disk.overlay.final_position() - ); - assert_eq!( - memory.overlay.final_commit_sequence(), - disk.overlay.final_commit_sequence() - ); - assert_eq!( - memory.overlay.bundle().digest(), - disk.overlay.bundle().digest() - ); - } - for (index, tail) in recovered.into_iter().enumerate() { - assert_eq!(tail.overlay.predecessor().cell, cells[index]); - assert_eq!(tail.first_node_sequence, index as u64 + 1); - assert_eq!(tail.last_node_sequence, CELLS as u64 + index as u64 + 1); - assert_eq!(tail.overlay.final_position(), second.position); - assert_eq!(tail.overlay.final_commit_sequence(), 3); - } - for base in &mut bases { - base.root.position = first.position; - base.root.commit_sequence = 2; - } - let recovered = build_recovery_overlays(frames, &bases, limits).unwrap(); - assert_eq!(recovered.len(), CELLS); - for (index, tail) in recovered.iter().enumerate() { - assert_eq!(tail.first_node_sequence, CELLS as u64 + index as u64 + 1); - assert_eq!(tail.overlay.predecessor().position, first.position); - assert_eq!(tail.overlay.final_position(), second.position); - } - database.close().unwrap(); -} diff --git a/crates/crab-cell-runtime/src/node/log_recovery.rs b/crates/crab-cell-runtime/src/node/log_recovery.rs deleted file mode 100644 index 5eb68096b..000000000 --- a/crates/crab-cell-runtime/src/node/log_recovery.rs +++ /dev/null @@ -1,965 +0,0 @@ -//! Node-log recovery: inventory, witnesses, and owner-loss coordination. -use std::collections::BTreeMap; -use std::fs::{File, OpenOptions}; -use std::io::{Read, Seek, SeekFrom, Write}; -use std::path::{Path, PathBuf}; -use std::sync::Arc; - -use futures_util::{StreamExt, future::join_all, stream}; - -use crate::cell::catalog::{CatalogProof, CellCatalog}; -use crate::control::Transition; -use crate::control::authority::{CellAuthority, VersionedControl}; -use crate::follower::FollowerReceipt; -use crate::identity::NodeId; -use crate::identity::{ApplicationId, CellId, Digest, SessionId}; -use crate::node::log::{ - RecoveryBase, build_recovery_overlays_file_backed, build_recovery_overlays_file_backed_stream, -}; -use crate::node::log_state::NodeLogPhase; -use crate::node::log_transport::{NodeLogTransport, SealRequest, TailRequest}; -use crate::node::{FencedNodeSession, NodeDirectory, NodeTakeoverProof, SealedNodeLog}; -use crate::recovery::manifest::RecoveryManifestStore; -use crate::{Error, Result}; - -const MAX_RECOVERY_CATALOG_HEAD_READS: usize = 32; - -const MAX_RECOVERY_PAGE_BYTES: u64 = 1 << 20; -const MAX_RECOVERY_PAGE_FRAMES: usize = 4_096; - -mod tail; -mod witness; - -use tail::*; - -use witness::*; - -/// Bounded work counters emitted for one node-log recovery attempt. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct RecoveryWorkSummary { - /// Recovery candidates examined. - pub candidate_count: u64, - /// Cells whose tails needed recovery. - pub affected_cells: u64, - /// Catalog shards scanned. - pub catalog_shards: u64, - /// Catalog pages read. - pub catalog_pages: u64, - /// Control records read. - pub control_reads: u64, - /// Follower pages read. - pub follower_pages: u64, - /// Follower frames read. - pub follower_frames: u64, - /// Follower bytes read. - pub follower_bytes: u64, - /// Peer requests issued. - pub peer_requests: u64, - /// Bundle bytes transferred. - pub bundle_bytes: u64, - /// Object-store reads. - pub object_reads: u64, - /// Object-store writes. - pub object_writes: u64, -} - -impl RecoveryWorkSummary { - /// Adds another summary's counters, failing if one would overflow. - pub fn merge(&mut self, other: Self) -> Result<()> { - self.candidate_count = self - .candidate_count - .checked_add(other.candidate_count) - .ok_or(Error::Capacity("recovery candidate count"))?; - self.affected_cells = self - .affected_cells - .checked_add(other.affected_cells) - .ok_or(Error::Capacity("recovery affected Cell count"))?; - self.catalog_shards = self - .catalog_shards - .checked_add(other.catalog_shards) - .ok_or(Error::Capacity("recovery catalog shard count"))?; - self.catalog_pages = self - .catalog_pages - .checked_add(other.catalog_pages) - .ok_or(Error::Capacity("recovery catalog page count"))?; - self.control_reads = self - .control_reads - .checked_add(other.control_reads) - .ok_or(Error::Capacity("recovery control read count"))?; - self.follower_pages = self - .follower_pages - .checked_add(other.follower_pages) - .ok_or(Error::Capacity("recovery follower page count"))?; - self.follower_frames = self - .follower_frames - .checked_add(other.follower_frames) - .ok_or(Error::Capacity("recovery follower frame count"))?; - self.follower_bytes = self - .follower_bytes - .checked_add(other.follower_bytes) - .ok_or(Error::Capacity("recovery follower byte count"))?; - self.peer_requests = self - .peer_requests - .checked_add(other.peer_requests) - .ok_or(Error::Capacity("recovery peer request count"))?; - self.bundle_bytes = self - .bundle_bytes - .checked_add(other.bundle_bytes) - .ok_or(Error::Capacity("recovery bundle byte count"))?; - self.object_reads = self - .object_reads - .checked_add(other.object_reads) - .ok_or(Error::Capacity("recovery object read count"))?; - self.object_writes = self - .object_writes - .checked_add(other.object_writes) - .ok_or(Error::Capacity("recovery object write count"))?; - Ok(()) - } -} - -/// Catalog work counters collected while validating recovered frame scopes. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct RecoveryInventorySummary { - /// Cells owned by the dead session. - pub affected_cells: u64, - /// Catalog shards scanned. - pub catalog_shards: u64, - /// Catalog pages read. - pub catalog_pages: u64, - /// Control records read. - pub control_reads: u64, -} - -/// Recovery Cells together with the bounded catalog work needed to discover them. -pub struct RecoverableCellInventory { - /// Dead-session Cells the scan discovered. - pub cells: Vec, - /// Catalog work the scan performed. - pub summary: RecoveryInventorySummary, -} - -/// Verified uncovered suffix gathered after every reachable follower is sealed. -pub struct SealedSession { - /// Session whose log was sealed. - pub leader_session: SessionId, - /// Node-log epoch that was sealed. - pub log_epoch: u64, - /// Highest node-log sequence the sealed log tiered. - pub tiered_through: u64, - /// Highest node-log sequence the sealed log made durable. - pub durable_through: u64, - /// Verified frames of the recovered suffix. - pub frames: Vec, - witness: Option, - work: RecoveryWorkSummary, - // Keep admission until the caller has pinned or discarded the recovered - // bytes, not merely until their last network page arrives. - _reservation: Option, -} - -impl SealedSession { - /// Returns the number of frames in the recovered suffix. - #[must_use] - pub fn frame_count(&self) -> u64 { - self.witness - .as_ref() - .map_or(self.frames.len() as u64, |witness| witness.frame_count) - } - - /// Returns the work counters gathered while sealing. - #[must_use] - pub const fn work(&self) -> RecoveryWorkSummary { - self.work - } - - /// Returns authenticated frame scopes without materializing frame bodies. - pub fn scopes(&self, limits: crab_ltx::Limits) -> Result> { - if let Some(witness) = &self.witness { - return unique_scopes( - witness - .reader(limits)? - .map(|frame| frame.map(|frame| frame.scope())), - ); - } - unique_scopes(self.frames.iter().map(|frame| Ok(frame.scope()))) - } -} - -/// Retains one authenticated generation per Cell while validating every frame -/// in the witness. Recovery only needs the affected Cell set; retaining every -/// frame scope would make catalog discovery grow with the uncovered tail. -fn unique_scopes(scopes: I) -> Result> -where - I: IntoIterator>, -{ - let mut unique = BTreeMap::<[u8; 32], crab_ltx::NodeFrameScope>::new(); - for scope in scopes { - let scope = scope?; - if let Some(existing) = unique.get(&scope.cell) { - if existing.leader_session != scope.leader_session - || existing.log_epoch != scope.log_epoch - || existing.application != scope.application - || existing.incarnation != scope.incarnation - || existing.cell_epoch != scope.cell_epoch - { - return Err(Error::Control( - "recovery Cell scope has multiple generations", - )); - } - continue; - } - unique.insert(scope.cell, scope); - } - Ok(unique.into_values().collect()) -} - -/// Mechanical seal-and-gather coordinator for one already claimed dead session. -/// -/// Session expiry and recovery-claim CAS remain authority concerns. This type -/// never decides that a leader is dead; it only gathers the exact lane named by -/// its constructor. -pub struct NodeLogRecovery { - transport: Arc, - leader_session: SessionId, - log_epoch: u64, - members: Vec, - tiered_through: u64, - active: bool, - limits: crab_ltx::Limits, - recovery_disk: crab_ltx::DiskBudget, - recovery_scratch: Option, -} - -/// One dead-session Cell control that may need a recovered tail attached. -pub struct RecoveryCell { - /// Application the Cell belongs to. - pub application: ApplicationId, - /// Control authority the recovery writes through. - pub authority: CellAuthority, - /// Control record and token the scan observed. - pub observed: VersionedControl, -} - -/// Seals one claimed node log and pins every recovered Cell tail before return. -pub struct RecoveryCoordinator { - recovery: NodeLogRecovery, - manifests: RecoveryManifestStore, -} - -/// Completed dead-session recovery with every overlay pinned before log seal. -pub struct CompletedNodeRecovery { - /// Proof that the dead session's node log is sealed. - pub sealed: SealedNodeLog, - /// Recovered controls with their overlays attached. - pub controls: Vec, - /// Proof that the predecessor cannot add newer durable Cell state. - pub takeover: NodeTakeoverProof, -} - -/// Recovery controls and publication work produced by one sealed recovery. -pub struct RecoveryCoordinatorResult { - /// Controls the recovery updated. - pub controls: Vec, - /// Publication work the sealed recovery produced. - pub publication: crate::recovery::manifest::RecoveryPublicationSummary, -} - -/// Scans one application catalog for published Cells owned by a dead session. -pub async fn recoverable_cells( - catalog: &CellCatalog, - authority: &CellAuthority, - owner: SessionId, - limit: usize, -) -> Result> { - if owner.as_bytes().iter().all(|byte| *byte == 0) || limit == 0 { - return Err(Error::Node("node recovery inventory bound is invalid")); - } - let mut cells = Vec::new(); - let mut scans = stream::iter(0_u16..=u8::MAX.into()) - .map(|shard| catalog.scan_shard(shard as u8)) - .buffer_unordered(MAX_RECOVERY_CATALOG_HEAD_READS); - while let Some(scan) = scans.next().await { - let mut scan = scan?; - while let Some(page) = scan.next_page().await? { - for proof in page.entries() { - let Some(observed) = authority.load(proof.entry().cell()).await? else { - continue; - }; - if observed.value().owner.as_ref().map(|owner| owner.session) != Some(owner) - || observed.value().ltx_root().is_none() - { - continue; - } - if cells.len() == limit { - return Err(Error::Node( - "node recovery Cell inventory exceeds its limit", - )); - } - cells.push(RecoveryCell { - application: catalog.application(), - authority: authority.clone(), - observed, - }); - } - } - } - Ok(cells) -} - -/// Loads only catalog entries whose authenticated node-frame scopes are present -/// in a sealed witness. Each affected catalog shard is scanned once, then the -/// exact current Cell control is revalidated before recovery may proceed. -pub async fn recoverable_cells_from_frames( - catalog: &CellCatalog, - authority: &CellAuthority, - owner: SessionId, - frames: &[crab_ltx::VerifiedNodeFrame], - limit: usize, -) -> Result> { - let scopes = unique_scopes(frames.iter().map(|frame| Ok(frame.scope())))?; - recoverable_cells_from_scopes(catalog, authority, owner, &scopes, limit).await -} - -/// Loads only catalog entries named by authenticated witness scopes. -pub async fn recoverable_cells_from_scopes( - catalog: &CellCatalog, - authority: &CellAuthority, - owner: SessionId, - frame_scopes: &[crab_ltx::NodeFrameScope], - limit: usize, -) -> Result> { - Ok( - recoverable_cells_from_scopes_with_summary(catalog, authority, owner, frame_scopes, limit) - .await? - .cells, - ) -} - -/// Scans one application catalog for published Cells owned by a dead session, -/// bounded by the supplied frame scopes and `limit`. -pub async fn recoverable_cells_from_scopes_with_summary( - catalog: &CellCatalog, - authority: &CellAuthority, - owner: SessionId, - frame_scopes: &[crab_ltx::NodeFrameScope], - limit: usize, -) -> Result { - if owner.as_bytes().iter().all(|byte| *byte == 0) || limit == 0 { - return Err(Error::Node("node recovery inventory bound is invalid")); - } - type Scope = ([u8; 16], u64); - let mut scopes = BTreeMap::<[u8; 32], Scope>::new(); - let application = *catalog.application().as_bytes(); - for scope in frame_scopes { - if scope.leader_session != *owner.as_bytes() || scope.application != application { - return Err(Error::Node("recovery frame application or owner differs")); - } - let key = (scope.incarnation, scope.cell_epoch); - if let Some(existing) = scopes.get(&scope.cell) { - if *existing != key { - return Err(Error::Control( - "recovery Cell scope has multiple generations", - )); - } - } else { - if scopes.len() == limit { - return Err(Error::Node( - "node recovery Cell inventory exceeds its limit", - )); - } - scopes.insert(scope.cell, key); - } - } - if scopes.is_empty() { - return Ok(RecoverableCellInventory { - cells: Vec::new(), - summary: RecoveryInventorySummary::default(), - }); - } - - let mut needed = BTreeMap::>::new(); - for cell in scopes.keys() { - needed.entry(cell[0]).or_default().push(*cell); - } - let affected_cells = - u64::try_from(scopes.len()).map_err(|_| Error::Capacity("recovery affected Cell count"))?; - let catalog_shards = - u64::try_from(needed.len()).map_err(|_| Error::Capacity("recovery catalog shard count"))?; - let mut entries = BTreeMap::<[u8; 32], CatalogProof>::new(); - let mut catalog_pages = 0_u64; - for (shard, cells) in needed { - let mut scan = catalog.scan_shard(shard).await?; - while let Some(page) = scan.next_page().await? { - catalog_pages = catalog_pages - .checked_add(1) - .ok_or(Error::Capacity("recovery catalog page count"))?; - for proof in page.entries() { - if cells.binary_search(proof.entry().cell().as_bytes()).is_ok() { - entries.insert(*proof.entry().cell().as_bytes(), proof.clone()); - } - } - } - } - - let mut recovered = Vec::with_capacity(scopes.len()); - for (cell_bytes, (incarnation, cell_epoch)) in scopes { - let cell = CellId::from_bytes(cell_bytes); - let proof = entries - .remove(&cell_bytes) - .ok_or(Error::Catalog("recovery frame Cell is not cataloged"))?; - if proof.entry().cell() != cell { - return Err(Error::Catalog("recovery catalog proof scope differs")); - } - let observed = authority - .load(cell) - .await? - .ok_or(Error::Control("recovery Cell control is missing"))?; - let control = observed.value(); - if control.owner.as_ref().map(|current| current.session) != Some(owner) - || control.incarnation.as_bytes() != &incarnation - || control.epoch != cell_epoch - || control.ltx_root().is_none() - { - return Err(Error::Control("recovery Cell control scope differs")); - } - recovered.push(RecoveryCell { - application: catalog.application(), - authority: authority.clone(), - observed, - }); - } - Ok(RecoverableCellInventory { - summary: RecoveryInventorySummary { - affected_cells, - catalog_shards, - catalog_pages, - control_reads: u64::try_from(recovered.len()) - .map_err(|_| Error::Capacity("recovery control read count"))?, - }, - cells: recovered, - }) -} - -impl RecoveryCoordinator { - /// Creates a coordinator over one node-log recovery and manifest store. - #[must_use] - pub const fn new(recovery: NodeLogRecovery, manifests: RecoveryManifestStore) -> Self { - Self { - recovery, - manifests, - } - } - - /// Attaches immutable overlays to the exact dead-owner controls. - /// - /// Successful earlier attachments remain valid if a later Cell conflicts; - /// a retry must reload every control and rebuild against its exact root. - pub async fn recover( - &self, - fenced: FencedNodeSession, - cells: Vec, - ) -> Result> { - self.recovery.validate_fence(&fenced)?; - let sealed = self.recovery.ensure_sealed().await?; - self.recover_sealed(fenced, cells, sealed).await - } - - /// Attaches overlays from a witness that was already sealed by the caller. - /// Keeping the witness explicit lets orchestration derive affected Cells - /// from authenticated frame scopes before any catalog scan. - pub async fn recover_sealed( - &self, - fenced: FencedNodeSession, - cells: Vec, - sealed: SealedSession, - ) -> Result> { - Ok(self - .recover_sealed_with_summary(fenced, cells, sealed) - .await? - .controls) - } - - /// Attaches overlays from an already sealed witness and returns the work - /// counters for the recovery attempt. - pub async fn recover_sealed_with_summary( - &self, - fenced: FencedNodeSession, - cells: Vec, - sealed: SealedSession, - ) -> Result { - self.recovery.validate_fence(&fenced)?; - let mut bases = Vec::with_capacity(cells.len()); - for cell in &cells { - let control = cell.observed.value(); - let root = control - .ltx_root() - .ok_or(Error::Control("recovery Cell has no published root"))?; - if control.owner.as_ref().map(|owner| owner.session) != Some(fenced.session()) - || *cell.application.as_bytes() == [0; 16] - || control.recovery.as_ref().is_some_and(|recovery| { - recovery.leader_session != fenced.session() - || recovery.log_epoch != self.recovery.log_epoch - }) - { - return Err(Error::Control("recovery Cell scope differs")); - } - bases.push(RecoveryBase { - application: *cell.application.as_bytes(), - cell_epoch: control.epoch, - root, - }); - } - - if sealed.frame_count() == 0 { - return Ok(RecoveryCoordinatorResult { - controls: Vec::new(), - publication: crate::recovery::manifest::RecoveryPublicationSummary::default(), - }); - } - let scratch = self.manifests.recovery_scratch_directory(); - let tails = if let Some(witness) = &sealed.witness { - let reader = witness.reader(self.recovery.limits)?; - build_recovery_overlays_file_backed_stream( - reader, - &bases, - self.recovery.limits, - &scratch, - )? - } else { - build_recovery_overlays_file_backed( - sealed.frames, - &bases, - self.recovery.limits, - &scratch, - )? - }; - if tails.is_empty() { - return Ok(RecoveryCoordinatorResult { - controls: Vec::new(), - publication: crate::recovery::manifest::RecoveryPublicationSummary::default(), - }); - } - let pinned = self - .manifests - .pin_with_summary(sealed.leader_session, sealed.log_epoch, tails) - .await?; - if pinned.cells.len() > cells.len() { - return Err(Error::Node("recovery manifest exceeds Cell inventory")); - } - - let mut attached = Vec::with_capacity(pinned.cells.len()); - for pin in pinned.cells { - let cell = cells - .iter() - .find(|candidate| { - candidate.application == pin.application - && candidate.observed.value().cell == pin.cell - && candidate.observed.value().incarnation == pin.incarnation - && candidate.observed.value().epoch == pin.cell_epoch - }) - .ok_or(Error::Node("recovered Cell is absent from inventory"))?; - if let Some(current) = cell.observed.value().recovery.as_ref() { - if current == &pin.recovery { - attached.push(cell.observed.clone()); - continue; - } - return Err(Error::Control( - "different recovery overlay is already pinned", - )); - } - let successor = cell.observed.value().attach_recovery(pin.recovery)?; - let versioned = match cell - .authority - .transition( - &cell.observed, - successor.clone(), - Transition::AttachRecovery, - ) - .await - { - Ok(versioned) => versioned, - Err(error) => { - let current = cell - .authority - .load(cell.observed.value().cell) - .await? - .ok_or(Error::Fenced)?; - if current.value() == &successor { - current - } else { - return Err(error); - } - } - }; - attached.push(versioned); - } - Ok(RecoveryCoordinatorResult { - controls: attached, - publication: pinned.summary, - }) - } - - /// Pins every recovered Cell and then atomically seals the claimed node log. - pub async fn recover_and_seal( - &self, - directory: &NodeDirectory, - fenced: FencedNodeSession, - cells: Vec, - now_ms: i64, - ) -> Result { - let controls = self.recover(fenced.clone(), cells).await?; - self.finish(directory, fenced, controls, now_ms).await - } - - /// Seals one still-current claim after all recovered controls were pinned. - pub async fn finish( - &self, - directory: &NodeDirectory, - fenced: FencedNodeSession, - controls: Vec, - now_ms: i64, - ) -> Result { - self.recovery.validate_fence(&fenced)?; - let mut manifest = None::; - for control in &controls { - let recovery = control - .value() - .recovery - .as_ref() - .ok_or(Error::Control("recovered Cell has no pinned overlay"))?; - if recovery.leader_session != fenced.session() - || recovery.log_epoch != self.recovery.log_epoch - { - return Err(Error::Control("recovered Cell overlay scope differs")); - } - match manifest { - None => manifest = Some(recovery.manifest_digest), - Some(current) if current == recovery.manifest_digest => {} - Some(_) => { - return Err(Error::Control( - "recovered session produced multiple manifests", - )); - } - } - } - let sealed = directory.seal_recovery(&fenced, manifest, now_ms).await?; - let takeover = NodeTakeoverProof::after_recovery(&fenced, &sealed)?; - Ok(CompletedNodeRecovery { - sealed, - controls, - takeover, - }) - } -} - -impl NodeLogRecovery { - pub(crate) fn new( - transport: Arc, - leader_node: NodeId, - leader_session: SessionId, - log_epoch: u64, - members: Vec, - tiered_through: u64, - active: bool, - limits: crab_ltx::Limits, - ) -> Result { - if leader_node.as_bytes().iter().all(|byte| *byte == 0) - || leader_session.as_bytes().iter().all(|byte| *byte == 0) - || log_epoch == 0 - || members.is_empty() - || members.len() > 2 - || members.contains(&leader_node) - || members - .iter() - .any(|member| member.as_bytes().iter().all(|byte| *byte == 0)) - || !members - .windows(2) - .all(|pair| pair[0].as_bytes() < pair[1].as_bytes()) - { - return Err(Error::Node("invalid node-log recovery ensemble")); - } - Ok(Self { - transport, - leader_session, - log_epoch, - members, - tiered_through, - active, - limits, - recovery_disk: default_recovery_disk(limits), - recovery_scratch: None, - }) - } - - /// Builds recovery only from the exact CAS-protected failed-session log. - pub fn from_fenced( - transport: Arc, - fenced: &FencedNodeSession, - limits: crab_ltx::Limits, - ) -> Result { - Self::from_fenced_with_disk(transport, fenced, limits, default_recovery_disk(limits)) - } - - /// Builds recovery with a shared node-local disk budget. - pub fn from_fenced_with_disk( - transport: Arc, - fenced: &FencedNodeSession, - limits: crab_ltx::Limits, - recovery_disk: crab_ltx::DiskBudget, - ) -> Result { - let log = fenced - .log() - .ok_or(Error::Node("fenced session has no enrolled node log"))?; - let claim = log - .recovery() - .ok_or(Error::Node("fenced node log has no recovery claim"))?; - if log.phase() != NodeLogPhase::Recovering - || claim.claimant() != fenced.claimant() - || claim.generation() != fenced.claim_generation() - || claim.expires_at_ms() != fenced.claim_expires_at_ms() - { - return Err(Error::Fenced); - } - Self::new( - transport, - fenced.node(), - fenced.session(), - log.epoch(), - log.members().to_vec(), - log.tiered_through(), - log.active(), - limits, - ) - .map(|mut recovery| { - recovery.recovery_disk = recovery_disk; - recovery - }) - } - - /// Uses a shared budget for temporary follower-tail materialization. - pub fn with_recovery_disk(mut self, recovery_disk: crab_ltx::DiskBudget) -> Self { - self.recovery_disk = recovery_disk; - self - } - - /// Uses the runtime-owned session volume for the bounded witness file. - pub fn with_recovery_scratch(mut self, directory: PathBuf) -> Self { - self.recovery_scratch = Some(directory); - self - } - - fn validate_fence(&self, fenced: &FencedNodeSession) -> Result<()> { - let log = fenced.log().ok_or(Error::Fenced)?; - let claim = log.recovery().ok_or(Error::Fenced)?; - if fenced.session() != self.leader_session - || log.phase() != NodeLogPhase::Recovering - || log.epoch() != self.log_epoch - || log.members() != self.members - || log.tiered_through() != self.tiered_through - || log.active() != self.active - || claim.claimant() != fenced.claimant() - || claim.generation() != fenced.claim_generation() - || claim.expires_at_ms() != fenced.claim_expires_at_ms() - { - return Err(Error::Fenced); - } - Ok(()) - } - - /// Seals all reachable members, rejects conflicts, and returns a complete witness. - pub async fn ensure_sealed(&self) -> Result { - self.ensure_sealed_with_mode(false).await - } - - /// Seals all reachable members while retaining the selected witness on the - /// runtime scratch volume instead of the heap. - pub async fn ensure_sealed_bounded(&self) -> Result { - self.ensure_sealed_with_mode(true).await - } - - async fn ensure_sealed_with_mode(&self, bounded: bool) -> Result { - let receipts = join_all(self.members.iter().map(|member| { - let transport = Arc::clone(&self.transport); - let member = *member; - async move { - ( - member, - transport - .seal( - member, - SealRequest { - leader_session: self.leader_session, - log_epoch: self.log_epoch, - }, - ) - .await, - ) - } - })) - .await; - if receipts - .iter() - .filter_map(|(_, receipt)| receipt.as_ref().ok()) - .any(|receipt| { - (receipt.base_sequence == 0 && receipt.durable_through != 0) - || receipt.base_sequence > receipt.durable_through.saturating_add(1) - }) - { - return Err(Error::Node("follower seal receipt is invalid")); - } - let successful_seals = receipts - .iter() - .filter(|(_, receipt)| receipt.is_ok()) - .count(); - let durable_through = receipts - .iter() - .filter_map(|(_, receipt)| receipt.as_ref().ok()) - .map(|receipt| receipt.durable_through) - .max() - .unwrap_or(self.tiered_through); - if durable_through <= self.tiered_through { - if self.active && successful_seals == 0 { - return Err(Error::Node( - "active node log has no complete follower witness", - )); - } - return Ok(SealedSession { - leader_session: self.leader_session, - log_epoch: self.log_epoch, - tiered_through: self.tiered_through, - durable_through: self.tiered_through, - frames: Vec::new(), - witness: None, - work: RecoveryWorkSummary { - peer_requests: u64::try_from(self.members.len()) - .map_err(|_| Error::Capacity("recovery peer request count"))?, - ..RecoveryWorkSummary::default() - }, - _reservation: None, - }); - } - - let required_first = self - .tiered_through - .checked_add(1) - .ok_or(Error::Node("node-log recovery sequence overflow"))?; - let scratch = self - .recovery_scratch - .clone() - .unwrap_or_else(std::env::temp_dir); - let expected_frames = durable_through - .checked_sub(required_first) - .and_then(|count| count.checked_add(1)) - .ok_or(Error::Node("node-log recovery frame range overflows"))?; - let digest_bytes = expected_frames - .checked_mul(WITNESS_DIGEST_RECORD_BYTES) - .ok_or(Error::Capacity("recovery witness digest table"))?; - let digest_reservation = self.recovery_disk.try_reserve(digest_bytes)?; - let mut candidates = receipts - .iter() - .filter_map(|(member, receipt)| { - let receipt = receipt.as_ref().ok()?; - (receipt.durable_through == durable_through - && receipt.base_sequence <= required_first) - .then_some((*member, *receipt)) - }) - .collect::>(); - candidates.sort_unstable_by(|left, right| left.0.as_bytes().cmp(right.0.as_bytes())); - - let mut work = RecoveryWorkSummary { - peer_requests: u64::try_from(self.members.len()) - .map_err(|_| Error::Capacity("recovery peer request count"))?, - ..RecoveryWorkSummary::default() - }; - for (candidate_member, candidate_receipt) in candidates { - let candidate_reservation = self.recovery_disk.try_reserve(0)?; - let mut collector = if bounded { - WitnessCollector::file(&scratch)? - } else { - WitnessCollector::Memory(Vec::new()) - }; - let mut digest_writer = - WitnessDigestWriter::new(&scratch, required_first, expected_frames)?; - if !collect_member_tail( - candidate_member, - candidate_receipt, - TailReadContext { - transport: self.transport.as_ref(), - leader_session: self.leader_session, - log_epoch: self.log_epoch, - required_first, - limits: self.limits, - reservation: Some(&candidate_reservation), - work: Some(&mut work), - }, - TailSinks { - collector: Some(&mut collector), - digests: Some(&mut digest_writer), - compare: None, - }, - ) - .await? - { - continue; - } - if !collector.matches_range(required_first, durable_through) { - continue; - } - let mut digests = digest_writer.finish()?; - for (member, receipt) in &receipts { - if *member == candidate_member { - continue; - } - let Ok(receipt) = receipt else { - continue; - }; - if !collect_member_tail( - *member, - *receipt, - TailReadContext { - transport: self.transport.as_ref(), - leader_session: self.leader_session, - log_epoch: self.log_epoch, - required_first, - limits: self.limits, - reservation: None, - work: Some(&mut work), - }, - TailSinks { - collector: None, - digests: None, - compare: Some(&mut digests), - }, - ) - .await? - { - continue; - } - } - drop(digest_reservation); - let selected = collector.finish()?; - let (frames, witness) = match selected { - WitnessMaterial::Memory(frames) => (frames, None), - WitnessMaterial::File(witness) => (Vec::new(), Some(witness)), - }; - return Ok(SealedSession { - leader_session: self.leader_session, - log_epoch: self.log_epoch, - tiered_through: self.tiered_through, - durable_through, - frames, - witness, - work, - _reservation: Some(candidate_reservation), - }); - } - drop(digest_reservation); - Err(Error::Node( - "active node log has no complete follower witness", - )) - } -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/node/log_recovery/tail.rs b/crates/crab-cell-runtime/src/node/log_recovery/tail.rs deleted file mode 100644 index 87fea9f0b..000000000 --- a/crates/crab-cell-runtime/src/node/log_recovery/tail.rs +++ /dev/null @@ -1,167 +0,0 @@ -//! Streaming one follower's tail through the common framing checks. -//! -//! A transport or malformed-page failure returns false so another complete -//! witness may be tried; digest disagreement stays a hard recovery error. - -use super::*; - -/// Streams one follower tail through the common page/framing checks. A -/// transport or malformed-page failure returns `false` so another complete -/// witness may be tried; digest disagreement remains a hard recovery error. -pub(super) struct TailReadContext<'a> { - pub(super) transport: &'a dyn NodeLogTransport, - pub(super) leader_session: SessionId, - pub(super) log_epoch: u64, - pub(super) required_first: u64, - pub(super) limits: crab_ltx::Limits, - pub(super) reservation: Option<&'a crab_ltx::DiskReservation>, - pub(super) work: Option<&'a mut RecoveryWorkSummary>, -} - -pub(super) struct TailSinks<'a> { - pub(super) collector: Option<&'a mut WitnessCollector>, - pub(super) digests: Option<&'a mut WitnessDigestWriter>, - pub(super) compare: Option<&'a mut SealedWitnessDigests>, -} - -pub(super) async fn collect_member_tail( - member: NodeId, - receipt: FollowerReceipt, - context: TailReadContext<'_>, - mut sinks: TailSinks<'_>, -) -> Result { - let TailReadContext { - transport, - leader_session, - log_epoch, - required_first, - limits, - reservation, - mut work, - } = context; - if receipt.durable_through < required_first { - return Ok(false); - } - let mut first_sequence = required_first.max(receipt.base_sequence); - if first_sequence > receipt.durable_through { - return Ok(false); - } - let mut tail_bytes = 0_u64; - loop { - if let Some(work) = work.as_deref_mut() { - work.peer_requests = work - .peer_requests - .checked_add(1) - .ok_or(Error::Capacity("recovery peer request count"))?; - } - let page = match transport - .tail_page( - member, - TailRequest { - leader_session, - log_epoch, - first_sequence, - }, - ) - .await - { - Ok(page) => page, - Err(_) => return Ok(false), - }; - if page.frames.is_empty() || page.frames.len() > MAX_RECOVERY_PAGE_FRAMES { - return Ok(false); - } - let page_count = page.frames.len(); - if let Some(work) = work.as_deref_mut() { - work.follower_pages = work - .follower_pages - .checked_add(1) - .ok_or(Error::Capacity("recovery follower page count"))?; - } - let verified = match page - .frames - .into_iter() - .map(|bytes| crab_ltx::inspect_node_frame(bytes, limits)) - .collect::>>() - { - Ok(verified) => verified, - Err(_) => return Ok(false), - }; - if verified.iter().enumerate().any(|(offset, frame)| { - let scope = frame.scope(); - scope.leader_session != *leader_session.as_bytes() - || scope.log_epoch != log_epoch - || first_sequence.checked_add(offset as u64) != Some(scope.node_sequence) - || scope.node_sequence > receipt.durable_through - }) { - return Ok(false); - } - let Some(page_bytes) = verified.iter().try_fold(0_u64, |bytes, frame| { - bytes.checked_add(frame.encoded().len() as u64) - }) else { - return Ok(false); - }; - if let Some(work) = work.as_deref_mut() { - work.follower_frames = work - .follower_frames - .checked_add( - u64::try_from(verified.len()) - .map_err(|_| Error::Capacity("recovery follower frame count"))?, - ) - .ok_or(Error::Capacity("recovery follower frame count"))?; - work.follower_bytes = work - .follower_bytes - .checked_add(page_bytes) - .ok_or(Error::Capacity("recovery follower byte count"))?; - } - let page_limit = if page_count == 1 { - MAX_RECOVERY_PAGE_BYTES.saturating_add(limits.max_capture_bytes) - } else { - MAX_RECOVERY_PAGE_BYTES - }; - if page_bytes > page_limit { - return Ok(false); - } - tail_bytes = match tail_bytes.checked_add(page_bytes) { - Some(bytes) if bytes <= recovery_tail_reservation_bytes(limits) => bytes, - _ => return Ok(false), - }; - if let Some(reservation) = reservation { - reservation.try_grow(page_bytes)?; - } - for frame in &verified { - let sequence = frame.scope().node_sequence; - let digest = frame.digest(); - if let Some(writer) = sinks.digests.as_mut() { - writer.push(sequence, digest)?; - } - if let Some(table) = sinks.compare.as_mut() { - table.matches(sequence, digest)?; - } - } - let last_sequence = verified.last().map(|frame| frame.scope().node_sequence); - if let Some(output) = sinks.collector.as_mut() { - output.push(verified)?; - } - let Some(next_sequence) = page.next_sequence else { - return Ok(last_sequence == Some(receipt.durable_through)); - }; - let page_count = u64::try_from(page_count) - .map_err(|_| Error::Node("recovery page frame count overflows"))?; - let expected_next = first_sequence - .checked_add(page_count) - .ok_or(Error::Node("recovery page sequence overflows"))?; - if next_sequence != expected_next || next_sequence > receipt.durable_through { - return Ok(false); - } - first_sequence = next_sequence; - } -} - -pub(super) fn default_recovery_disk(limits: crab_ltx::Limits) -> crab_ltx::DiskBudget { - crab_ltx::DiskBudget::new(recovery_tail_reservation_bytes(limits)) -} - -fn recovery_tail_reservation_bytes(limits: crab_ltx::Limits) -> u64 { - limits.max_plan_bytes.min(512 << 20) -} diff --git a/crates/crab-cell-runtime/src/node/log_recovery/tests.rs b/crates/crab-cell-runtime/src/node/log_recovery/tests.rs deleted file mode 100644 index df70474d6..000000000 --- a/crates/crab-cell-runtime/src/node/log_recovery/tests.rs +++ /dev/null @@ -1,589 +0,0 @@ -use bytes::Bytes; -use futures_util::future::BoxFuture; - -use super::*; -use crate::follower::FollowerStore; -use crate::node::log_transport::{ - AppendRequest, LocalFollowerTransport, NodeLogTransport, RetireRequest, -}; - -fn scope(cell: u8, sequence: u64) -> crab_ltx::NodeFrameScope { - crab_ltx::NodeFrameScope { - leader_session: [1; 16], - log_epoch: 3, - node_sequence: sequence, - application: [4; 16], - cell: [cell; 32], - incarnation: [6; 16], - cell_epoch: 7, - commit_sequence: sequence, - } -} - -#[test] -fn scope_validation_keeps_one_generation_per_cell() { - let scopes = unique_scopes([Ok(scope(1, 1)), Ok(scope(1, 2)), Ok(scope(2, 3))]).unwrap(); - assert_eq!(scopes.len(), 2); - assert_eq!(scopes[0].cell, [1; 32]); - assert_eq!(scopes[1].cell, [2; 32]); -} - -#[test] -fn scope_validation_rejects_conflicting_cell_generations() { - let mut conflicting = scope(1, 2); - conflicting.cell_epoch = 8; - assert!(matches!( - unique_scopes([Ok(scope(1, 1)), Ok(conflicting)]), - Err(Error::Control( - "recovery Cell scope has multiple generations" - )) - )); -} - -#[test] -fn recovery_work_summary_merge_is_checked() { - let mut summary = RecoveryWorkSummary { - follower_bytes: u64::MAX, - ..RecoveryWorkSummary::default() - }; - assert!(matches!( - summary.merge(RecoveryWorkSummary { - follower_bytes: 1, - ..RecoveryWorkSummary::default() - }), - Err(Error::Capacity("recovery follower byte count")) - )); -} - -struct FailingFirstTransport { - failed: NodeId, - good: LocalFollowerTransport, - gap: bool, - oversized: bool, - replacement: Option, -} - -struct FleetTransport { - stores: Vec<(NodeId, FollowerStore)>, -} - -impl FleetTransport { - fn store(&self, member: NodeId) -> Result<&FollowerStore> { - self.stores - .iter() - .find_map(|(candidate, store)| (*candidate == member).then_some(store)) - .ok_or(Error::Node("follower is absent from test fleet")) - } -} - -impl NodeLogTransport for FleetTransport { - fn append<'a>( - &'a self, - member: NodeId, - request: AppendRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - self.store(member)? - .append( - request.leader_session, - request.log_epoch, - request.frames, - request.covered_through, - ) - .await - }) - } - - fn seal<'a>( - &'a self, - member: NodeId, - request: SealRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - self.store(member)? - .seal(request.leader_session, request.log_epoch) - .await - }) - } - - fn retire<'a>( - &'a self, - member: NodeId, - request: RetireRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - self.store(member)? - .retire( - request.leader_session, - request.log_epoch, - request.covered_through, - ) - .await - }) - } - - fn tail<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> BoxFuture<'a, Result>> { - Box::pin(async move { - self.store(member)? - .read_tail( - request.leader_session, - request.log_epoch, - request.first_sequence, - ) - .await - }) - } -} - -impl NodeLogTransport for FailingFirstTransport { - fn append<'a>( - &'a self, - member: NodeId, - request: AppendRequest, - ) -> BoxFuture<'a, Result> { - self.good.append(member, request) - } - - fn seal<'a>( - &'a self, - member: NodeId, - request: SealRequest, - ) -> BoxFuture<'a, Result> { - if member == self.failed { - return Box::pin(async { - Ok(crate::follower::FollowerReceipt { - base_sequence: 1, - durable_through: 1, - }) - }); - } - self.good.seal(member, request) - } - - fn retire<'a>( - &'a self, - member: NodeId, - request: RetireRequest, - ) -> BoxFuture<'a, Result> { - self.good.retire(member, request) - } - - fn tail<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> BoxFuture<'a, Result>> { - if member == self.failed { - return Box::pin(async { Err(Error::Node("injected follower read failure")) }); - } - self.good.tail(member, request) - } - - fn tail_page<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> BoxFuture<'a, Result> { - if member == self.failed { - return Box::pin(async { Err(Error::Node("injected follower read failure")) }); - } - let good = self.good.clone(); - let gap = self.gap; - let oversized = self.oversized; - let replacement = self.replacement.clone(); - Box::pin(async move { - let mut page = good.tail_page(member, request).await?; - if let Some(replacement) = replacement { - page.frames = vec![replacement]; - } - if oversized { - let first = page - .frames - .first() - .cloned() - .ok_or(Error::Node("test page has no frame"))?; - page.frames = std::iter::repeat_n(first, MAX_RECOVERY_PAGE_FRAMES + 1).collect(); - page.next_sequence = Some( - request - .first_sequence - .checked_add(MAX_RECOVERY_PAGE_FRAMES as u64 + 1) - .ok_or(Error::Node("test page sequence overflow"))?, - ); - } - if gap { - let count = u64::try_from(page.frames.len()) - .map_err(|_| Error::Node("test page frame count overflow"))?; - page.next_sequence = Some( - request - .first_sequence - .checked_add(count) - .and_then(|next| next.checked_add(1)) - .ok_or(Error::Node("test page sequence overflow"))?, - ); - } - Ok(page) - }) - } -} - -#[tokio::test] -async fn active_lane_requires_and_returns_a_complete_follower_tail() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = crab_ltx::Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE values_(v); INSERT INTO values_ VALUES(randomblob(2097152))", - ) - }) - .unwrap(); - let capture = database.capture().unwrap(); - let segment = capture.segments.first().unwrap(); - let leader = SessionId::from_bytes([1; 16]); - let member = NodeId::from_bytes([2; 16]); - let frame = crab_ltx::encode_node_frame( - crab_ltx::NodeFrameScope { - leader_session: *leader.as_bytes(), - log_epoch: 3, - node_sequence: 1, - application: [4; 16], - cell: [5; 32], - incarnation: [6; 16], - cell_epoch: 7, - commit_sequence: 1, - }, - segment.info().clone(), - Bytes::from(std::fs::read(segment.path()).unwrap()), - limits, - ) - .unwrap(); - assert!(frame.encoded().len() > MAX_RECOVERY_PAGE_BYTES as usize); - let root = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let transport: Arc = Arc::new(LocalFollowerTransport::new(member, store)); - transport - .append( - member, - AppendRequest { - leader_session: leader, - log_epoch: 3, - frames: vec![frame.encoded().clone()], - covered_through: 0, - }, - ) - .await - .unwrap(); - let rejected = NodeLogRecovery::new( - Arc::clone(&transport), - NodeId::from_bytes([1; 16]), - leader, - 3, - vec![member], - 0, - true, - limits, - ) - .unwrap() - .with_recovery_disk(crab_ltx::DiskBudget::new(0)); - assert!(rejected.ensure_sealed().await.is_err()); - let budget = crab_ltx::DiskBudget::new(1 << 30); - let recovery = NodeLogRecovery::new( - Arc::clone(&transport), - NodeId::from_bytes([1; 16]), - leader, - 3, - vec![member], - 0, - true, - limits, - ) - .unwrap() - .with_recovery_disk(budget.clone()); - let sealed = recovery.ensure_sealed().await.unwrap(); - assert_eq!(sealed.durable_through, 1); - assert_eq!(sealed.frames.len(), 1); - assert!(budget.used() > 0); - drop(sealed); - assert_eq!(budget.used(), 0); - let scratch = tempfile::TempDir::new().unwrap(); - let bounded = NodeLogRecovery::new( - Arc::clone(&transport), - NodeId::from_bytes([1; 16]), - leader, - 3, - vec![member], - 0, - true, - limits, - ) - .unwrap() - .with_recovery_disk(crab_ltx::DiskBudget::new(1 << 30)) - .with_recovery_scratch(scratch.path().to_owned()); - let sealed = bounded.ensure_sealed_bounded().await.unwrap(); - assert!(sealed.frames.is_empty()); - assert_eq!(sealed.frame_count(), 1); - assert_eq!(sealed.scopes(limits).unwrap().len(), 1); - drop(sealed); - assert!(scratch.path().read_dir().unwrap().next().is_none()); - database.close().unwrap(); -} - -#[tokio::test] -async fn recovery_uses_the_next_complete_witness_after_a_read_failure() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = crab_ltx::Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let segment = database.capture().unwrap().segments.remove(0); - let leader = SessionId::from_bytes([1; 16]); - let failed = NodeId::from_bytes([2; 16]); - let good = NodeId::from_bytes([3; 16]); - let frame = crab_ltx::encode_node_frame( - crab_ltx::NodeFrameScope { - leader_session: *leader.as_bytes(), - log_epoch: 3, - node_sequence: 1, - application: [4; 16], - cell: [5; 32], - incarnation: [6; 16], - cell_epoch: 7, - commit_sequence: 1, - }, - segment.info().clone(), - Bytes::from(std::fs::read(segment.path()).unwrap()), - limits, - ) - .unwrap(); - let root = tempfile::TempDir::new().unwrap(); - let local = LocalFollowerTransport::new( - good, - FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(), - ); - local - .append( - good, - AppendRequest { - leader_session: leader, - log_epoch: 3, - frames: vec![frame.encoded().clone()], - covered_through: 0, - }, - ) - .await - .unwrap(); - let transport: Arc = Arc::new(FailingFirstTransport { - failed, - good: local.clone(), - gap: false, - oversized: false, - replacement: None, - }); - let recovery = NodeLogRecovery::new( - transport, - NodeId::from_bytes([1; 16]), - leader, - 3, - vec![failed, good], - 0, - true, - limits, - ) - .unwrap(); - assert_eq!(recovery.ensure_sealed().await.unwrap().frames.len(), 1); - - let gapped: Arc = Arc::new(FailingFirstTransport { - failed, - good: local.clone(), - gap: true, - oversized: false, - replacement: None, - }); - let recovery = NodeLogRecovery::new( - gapped, - NodeId::from_bytes([1; 16]), - leader, - 3, - vec![good], - 0, - true, - limits, - ) - .unwrap(); - assert!(recovery.ensure_sealed().await.is_err()); - - let oversized: Arc = Arc::new(FailingFirstTransport { - failed, - good: local.clone(), - gap: false, - oversized: true, - replacement: None, - }); - let recovery = NodeLogRecovery::new( - oversized, - NodeId::from_bytes([1; 16]), - leader, - 3, - vec![good], - 0, - true, - limits, - ) - .unwrap(); - assert!(recovery.ensure_sealed().await.is_err()); - for wrong_epoch in [false, true] { - let mut scope = frame.scope(); - if wrong_epoch { - scope.log_epoch += 1; - } else { - scope.leader_session = [99; 16]; - } - let replacement = crab_ltx::encode_node_frame( - scope, - frame.segment().clone(), - frame.body().clone(), - limits, - ) - .unwrap(); - let transport = Arc::new(FailingFirstTransport { - failed, - good: local.clone(), - gap: false, - oversized: false, - replacement: Some(replacement.encoded().clone()), - }); - let recovery = NodeLogRecovery::new( - transport, - NodeId::from_bytes([1; 16]), - leader, - 3, - vec![good], - 0, - true, - limits, - ) - .unwrap(); - assert!(recovery.ensure_sealed().await.is_err()); - } - database.close().unwrap(); -} - -#[tokio::test] -async fn recovery_survives_a_simultaneous_follower_fleet_restart() { - let limits = crab_ltx::Limits::default(); - let source = tempfile::TempDir::new().unwrap(); - let mut database = crab_ltx::Db::open(&source.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let segment = database.capture().unwrap().segments.remove(0); - let leader = SessionId::from_bytes([1; 16]); - let leader_node = NodeId::from_bytes([1; 16]); - let members = [NodeId::from_bytes([2; 16]), NodeId::from_bytes([3; 16])]; - let frame = crab_ltx::encode_node_frame( - crab_ltx::NodeFrameScope { - leader_session: *leader.as_bytes(), - log_epoch: 3, - node_sequence: 1, - application: [4; 16], - cell: [5; 32], - incarnation: [6; 16], - cell_epoch: 7, - commit_sequence: 1, - }, - segment.info().clone(), - Bytes::from(std::fs::read(segment.path()).unwrap()), - limits, - ) - .unwrap(); - for conflict in [false, true] { - let roots = [ - tempfile::TempDir::new().unwrap(), - tempfile::TempDir::new().unwrap(), - ]; - for (index, (member, root)) in members.iter().zip(&roots).enumerate() { - let store = FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let mut scope = frame.scope(); - if conflict && index == 1 { - scope.cell = [99; 32]; - } - let frame = crab_ltx::encode_node_frame( - scope, - frame.segment().clone(), - frame.body().clone(), - limits, - ) - .unwrap(); - store - .append(leader, 3, vec![frame.encoded().clone()], 0) - .await - .unwrap(); - drop(store); - assert!(root.path().join("followers").exists(), "{member:?}"); - } - - let transport: Arc = Arc::new(FleetTransport { - stores: members - .iter() - .zip(&roots) - .map(|(member, root)| { - ( - *member, - FollowerStore::open( - root.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(), - ) - }) - .collect(), - }); - let recovery = NodeLogRecovery::new( - transport, - leader_node, - leader, - 3, - members.to_vec(), - 0, - true, - limits, - ) - .unwrap(); - - let result = recovery.ensure_sealed().await; - if conflict { - assert!(matches!( - result, - Err(Error::Node("follower witnesses disagree")) - )); - continue; - } - let sealed = result.unwrap(); - assert_eq!(sealed.durable_through, 1); - assert_eq!(sealed.frames.len(), 1); - assert_eq!(sealed.frames[0].scope().node_sequence, 1); - } - database.close().unwrap(); -} diff --git a/crates/crab-cell-runtime/src/node/log_recovery/witness.rs b/crates/crab-cell-runtime/src/node/log_recovery/witness.rs deleted file mode 100644 index ba43b1d0c..000000000 --- a/crates/crab-cell-runtime/src/node/log_recovery/witness.rs +++ /dev/null @@ -1,268 +0,0 @@ -//! Durable witness records for one node-log recovery attempt. -//! -//! Recovery writes what it sealed (and later verifies it) through these -//! bounded, checksummed records so a retry can prove it replayed the same -//! frames instead of trusting an in-memory summary. - -use super::*; - -const WITNESS_RECORD_HEADER_BYTES: usize = 8 + 32; -pub(super) const WITNESS_DIGEST_RECORD_BYTES: u64 = 8 + 32; - -pub(super) struct WitnessWriter { - path: tempfile::TempPath, - file: File, - first_sequence: Option, - last_sequence: Option, - frame_count: u64, -} - -impl WitnessWriter { - fn new(directory: &Path) -> Result { - let temporary = tempfile::Builder::new() - .prefix(".crab-witness-") - .tempfile_in(directory)?; - let path = temporary.into_temp_path(); - let file = OpenOptions::new().append(true).open(&path)?; - Ok(Self { - path, - file, - first_sequence: None, - last_sequence: None, - frame_count: 0, - }) - } - - fn push(&mut self, frame: &crab_ltx::VerifiedNodeFrame) -> Result<()> { - let encoded = frame.encoded(); - let length = u64::try_from(encoded.len()) - .map_err(|_| Error::Node("recovery witness frame length overflows"))?; - self.file.write_all(&length.to_le_bytes())?; - self.file.write_all(&frame.digest())?; - self.file.write_all(encoded)?; - self.first_sequence - .get_or_insert(frame.scope().node_sequence); - self.last_sequence = Some(frame.scope().node_sequence); - self.frame_count = self - .frame_count - .checked_add(1) - .ok_or(Error::Node("recovery witness frame count overflows"))?; - Ok(()) - } - - fn matches_range(&self, first: u64, last: u64) -> bool { - self.first_sequence == Some(first) - && self.last_sequence == Some(last) - && self.frame_count == last.saturating_sub(first).saturating_add(1) - } - - fn finish(self) -> Result { - self.file.sync_all()?; - if self.first_sequence.is_none() || self.last_sequence.is_none() { - return Err(Error::Node("recovery witness is empty")); - } - Ok(SealedWitness { - path: self.path, - frame_count: self.frame_count, - }) - } -} - -pub(super) struct SealedWitness { - path: tempfile::TempPath, - pub(super) frame_count: u64, -} - -impl SealedWitness { - pub(super) fn reader(&self, limits: crab_ltx::Limits) -> Result { - Ok(WitnessReader { - file: File::open(&self.path)?, - limits, - remaining: self.frame_count, - }) - } -} - -/// Disk-backed sequence-to-digest table used to compare every reachable -/// follower without retaining one digest map per recovered frame in memory. -pub(super) struct WitnessDigestWriter { - path: tempfile::TempPath, - file: File, - first_sequence: u64, - expected_frames: u64, - written_frames: u64, -} - -pub(super) struct SealedWitnessDigests { - _path: tempfile::TempPath, - file: File, - first_sequence: u64, - frame_count: u64, -} - -impl WitnessDigestWriter { - pub(super) fn new(directory: &Path, first_sequence: u64, expected_frames: u64) -> Result { - if expected_frames == 0 { - return Err(Error::Node("recovery witness digest range is empty")); - } - let temporary = tempfile::Builder::new() - .prefix(".crab-witness-digests-") - .tempfile_in(directory)?; - let path = temporary.into_temp_path(); - let file = OpenOptions::new().append(true).open(&path)?; - Ok(Self { - path, - file, - first_sequence, - expected_frames, - written_frames: 0, - }) - } - - pub(super) fn push(&mut self, sequence: u64, digest: [u8; 32]) -> Result<()> { - let expected = self - .first_sequence - .checked_add(self.written_frames) - .ok_or(Error::Node("recovery witness digest sequence overflow"))?; - if sequence != expected || self.written_frames == self.expected_frames { - return Err(Error::Node("recovery witness digest range differs")); - } - self.file.write_all(&sequence.to_le_bytes())?; - self.file.write_all(&digest)?; - self.written_frames = self - .written_frames - .checked_add(1) - .ok_or(Error::Node("recovery witness digest count overflow"))?; - Ok(()) - } - - pub(super) fn finish(self) -> Result { - if self.written_frames != self.expected_frames { - return Err(Error::Node("recovery witness digest range is incomplete")); - } - self.file.sync_all()?; - drop(self.file); - let file = File::open(&self.path)?; - Ok(SealedWitnessDigests { - _path: self.path, - file, - first_sequence: self.first_sequence, - frame_count: self.written_frames, - }) - } -} - -impl SealedWitnessDigests { - pub(super) fn matches(&mut self, sequence: u64, digest: [u8; 32]) -> Result<()> { - if sequence < self.first_sequence { - return Ok(()); - } - let offset = sequence - .checked_sub(self.first_sequence) - .and_then(|index| index.checked_mul(WITNESS_DIGEST_RECORD_BYTES)) - .ok_or(Error::Node("recovery witness digest offset overflow"))?; - if sequence - >= self - .first_sequence - .checked_add(self.frame_count) - .ok_or(Error::Node("recovery witness digest range overflow"))? - { - return Ok(()); - } - self.file.seek(SeekFrom::Start(offset))?; - let mut record = [0_u8; WITNESS_DIGEST_RECORD_BYTES as usize]; - self.file.read_exact(&mut record)?; - let stored_sequence = u64::from_le_bytes( - record[..8] - .try_into() - .map_err(|_| Error::Node("recovery witness digest record is invalid"))?, - ); - if stored_sequence != sequence || record[8..] != digest { - return Err(Error::Node("follower witnesses disagree")); - } - Ok(()) - } -} - -pub(super) struct WitnessReader { - file: File, - limits: crab_ltx::Limits, - remaining: u64, -} - -impl Iterator for WitnessReader { - type Item = Result; - - fn next(&mut self) -> Option { - if self.remaining == 0 { - return None; - } - let mut header = [0_u8; WITNESS_RECORD_HEADER_BYTES]; - if let Err(error) = self.file.read_exact(&mut header) { - return Some(Err(error.into())); - } - let length = u64::from_le_bytes(header[..8].try_into().ok()?); - let max_encoded = self.limits.max_capture_bytes.saturating_add(240); - if length > max_encoded || length > usize::MAX as u64 { - return Some(Err(Error::Node("recovery witness frame exceeds limit"))); - } - let mut encoded = vec![0_u8; length as usize]; - if let Err(error) = self.file.read_exact(&mut encoded) { - return Some(Err(error.into())); - } - if *blake3::hash(&encoded).as_bytes() != header[8..] { - return Some(Err(Error::Node("recovery witness frame digest differs"))); - } - self.remaining = self.remaining.saturating_sub(1); - Some(crab_ltx::inspect_node_frame(encoded.into(), self.limits).map_err(Into::into)) - } -} - -pub(super) enum WitnessCollector { - Memory(Vec), - File(WitnessWriter), -} - -pub(super) enum WitnessMaterial { - Memory(Vec), - File(SealedWitness), -} - -impl WitnessCollector { - pub(super) fn file(directory: &Path) -> Result { - Ok(Self::File(WitnessWriter::new(directory)?)) - } - - pub(super) fn push(&mut self, frames: Vec) -> Result<()> { - match self { - Self::Memory(existing) => existing.extend(frames), - Self::File(writer) => { - for frame in &frames { - writer.push(frame)?; - } - } - } - Ok(()) - } - - pub(super) fn matches_range(&self, first: u64, last: u64) -> bool { - match self { - Self::Memory(frames) => { - frames.first().map(|frame| frame.scope().node_sequence) == Some(first) - && frames.last().map(|frame| frame.scope().node_sequence) == Some(last) - && frames.windows(2).all(|pair| { - pair[0].scope().node_sequence.checked_add(1) - == Some(pair[1].scope().node_sequence) - }) - } - Self::File(writer) => writer.matches_range(first, last), - } - } - - pub(super) fn finish(self) -> Result { - match self { - Self::Memory(frames) => Ok(WitnessMaterial::Memory(frames)), - Self::File(writer) => Ok(WitnessMaterial::File(writer.finish()?)), - } - } -} diff --git a/crates/crab-cell-runtime/src/node/log_shipper.rs b/crates/crab-cell-runtime/src/node/log_shipper.rs deleted file mode 100644 index bfd5bfe5a..000000000 --- a/crates/crab-cell-runtime/src/node/log_shipper.rs +++ /dev/null @@ -1,505 +0,0 @@ -//! Shipping verified frames to follower lanes and retiring them. -use std::{collections::VecDeque, io::Read as _, sync::Arc, time::Duration}; - -use bytes::Bytes; -use futures_util::future::join_all; -use tokio::sync::{OwnedSemaphorePermit, Semaphore, mpsc}; - -use crate::identity::{ApplicationId, CellId}; -use crate::identity::{IncarnationId, NodeId}; -use crate::node::log::{CommitTicket, DurabilityGate}; -use crate::node::log_transport::{AppendRequest, NodeLogTransport}; -use crate::{Error, Result}; - -const MAX_BATCH_FRAMES: usize = 64; -const MAX_QUEUED_SUBMISSIONS: usize = 512; -const NODE_FRAME_HEADER_BYTES: u64 = 240; -const BATCH_INTERVAL: Duration = Duration::from_millis(1); - -/// One captured Cell commit awaiting ordered node-log assignment. -pub struct NodeLogSubmission { - application: ApplicationId, - cell: CellId, - incarnation: IncarnationId, - cell_epoch: u64, - commit_sequence: u64, - segments: Vec, - encoded_bytes: u64, -} - -impl NodeLogSubmission { - /// Binds one nonempty capture batch to its exact Cell authority generation. - pub fn new( - application: ApplicationId, - cell: CellId, - incarnation: IncarnationId, - cell_epoch: u64, - commit_sequence: u64, - cuts: &crab_ltx::CaptureBatch, - ) -> Result { - let encoded_bytes = cuts.segments.iter().try_fold(0_u64, |total, segment| { - total - .checked_add(segment.info().size_bytes) - .and_then(|bytes| bytes.checked_add(NODE_FRAME_HEADER_BYTES)) - }); - if application.as_bytes().iter().all(|byte| *byte == 0) - || cell.as_bytes().iter().all(|byte| *byte == 0) - || incarnation.as_bytes().iter().all(|byte| *byte == 0) - || cell_epoch == 0 - || commit_sequence == 0 - || cuts.segments.is_empty() - || encoded_bytes.is_none() - { - return Err(Error::Node("invalid node-log submission")); - } - Ok(Self { - application, - cell, - incarnation, - cell_epoch, - commit_sequence, - segments: cuts.segments.clone(), - encoded_bytes: encoded_bytes.ok_or(Error::Node("node-log byte count overflow"))?, - }) - } - - fn frame_count(&self) -> Result { - u64::try_from(self.segments.len()).map_err(|_| Error::Node("node-log frame count overflow")) - } - - fn load(self, limits: crab_ltx::Limits) -> Result { - let segments = self - .segments - .into_iter() - .map(|segment| { - let body = Bytes::from(read_segment( - segment.path(), - segment.info().size_bytes, - limits.max_capture_bytes, - )?); - Ok((segment.info().clone(), body)) - }) - .collect::>>()?; - Ok(LoadedNodeLogSubmission { - application: self.application, - cell: self.cell, - incarnation: self.incarnation, - cell_epoch: self.cell_epoch, - commit_sequence: self.commit_sequence, - segments, - }) - } -} - -struct LoadedNodeLogSubmission { - application: ApplicationId, - cell: CellId, - incarnation: IncarnationId, - cell_epoch: u64, - commit_sequence: u64, - segments: Vec<(crab_ltx::SegmentInfo, Bytes)>, -} - -impl LoadedNodeLogSubmission { - fn encode(self, ticket: CommitTicket, limits: crab_ltx::Limits) -> Result> { - self.segments - .into_iter() - .enumerate() - .map(|(offset, (segment, body))| { - let offset = u64::try_from(offset) - .map_err(|_| Error::Node("node-log frame offset overflow"))?; - let node_sequence = ticket - .first_sequence() - .checked_add(offset) - .ok_or(Error::Node("node-log sequence overflow"))?; - crab_ltx::encode_node_frame( - crab_ltx::NodeFrameScope { - leader_session: *ticket.leader_session().as_bytes(), - log_epoch: ticket.log_epoch(), - node_sequence, - application: *self.application.as_bytes(), - cell: *self.cell.as_bytes(), - incarnation: *self.incarnation.as_bytes(), - cell_epoch: self.cell_epoch, - commit_sequence: self.commit_sequence, - }, - segment, - body, - limits, - ) - .map(|frame| frame.encoded().clone()) - .map_err(Error::from) - }) - .collect() - } -} - -fn read_segment(path: &std::path::Path, expected_bytes: u64, limit: u64) -> Result> { - let length = usize::try_from(expected_bytes) - .ok() - .filter(|_| expected_bytes <= limit) - .ok_or(Error::Capacity("node-log frame bytes"))?; - let mut file = std::fs::File::open(path)?; - if file.metadata()?.len() != expected_bytes { - return Err(Error::Node("node-log segment size changed")); - } - let mut body = vec![0_u8; length]; - file.read_exact(&mut body)?; - let mut trailing = [0_u8; 1]; - if file.read(&mut trailing)? != 0 { - return Err(Error::Node("node-log segment grew while reading")); - } - Ok(body) -} - -/// Bounded node-wide multiplexer for one leader session and log epoch. -/// -/// Submissions receive consecutive tickets before entering the ordered queue. -/// The worker batches frames across Cells and credits fleet durability only -/// after every selected member returns an fsynced contiguous watermark. -pub struct NodeLogShipper { - sender: std::sync::Mutex>>, - worker: std::sync::Mutex>>, - bytes: Arc, - order: tokio::sync::Mutex<()>, - max_outstanding_bytes: u64, - gate: DurabilityGate, - limits: crab_ltx::Limits, -} - -impl NodeLogShipper { - /// Starts one shipper for the exact ensemble owned by `gate`. - pub fn new( - gate: DurabilityGate, - transport: Arc, - limits: crab_ltx::Limits, - ) -> Result { - Self::start( - gate, - transport, - limits, - crate::fleet::telemetry::CellTelemetryHandle::default(), - BATCH_INTERVAL, - ) - } - - /// Starts a shipper with a bounded operational telemetry sink. - pub fn new_with_telemetry( - gate: DurabilityGate, - transport: Arc, - limits: crab_ltx::Limits, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, - ) -> Result { - Self::start(gate, transport, limits, telemetry, BATCH_INTERVAL) - } - - fn start( - gate: DurabilityGate, - transport: Arc, - limits: crab_ltx::Limits, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, - interval: Duration, - ) -> Result { - let (leader, log_epoch, members) = gate.shipping_scope()?; - let batch_bytes = limits - .max_capture_bytes - .checked_add((MAX_BATCH_FRAMES as u64) * NODE_FRAME_HEADER_BYTES) - .ok_or(Error::Capacity("node-log outstanding bytes"))?; - let permits = usize::try_from(batch_bytes) - .ok() - .filter(|bytes| *bytes <= Semaphore::MAX_PERMITS && *bytes <= u32::MAX as usize) - .ok_or(Error::Capacity("node-log outstanding bytes"))?; - let runtime = tokio::runtime::Handle::try_current().map_err(Error::RuntimeStart)?; - let (sender, receiver) = mpsc::channel(MAX_QUEUED_SUBMISSIONS); - let bytes = Arc::new(Semaphore::new(permits)); - let worker_gate = gate.clone(); - let worker = runtime.spawn(run_shipper( - receiver, - worker_gate, - Arc::clone(&bytes), - transport, - leader, - log_epoch, - members, - batch_bytes, - telemetry.clone(), - interval, - )); - Ok(Self { - sender: std::sync::Mutex::new(Some(sender)), - worker: std::sync::Mutex::new(Some(worker)), - bytes, - order: tokio::sync::Mutex::new(()), - max_outstanding_bytes: batch_bytes, - gate, - limits, - }) - } - - /// Assigns a consecutive ticket and retains the encoded frames for shipping. - /// - /// Queue, byte admission, disk reads, and canonical encoding happen before - /// the ticket reservation commits, so failures cannot create a sequence gap. - pub async fn submit(&self, submission: NodeLogSubmission) -> Result { - let frame_count = submission.frame_count()?; - if frame_count > crate::node::log::MAX_TICKET_FRAMES - || submission.encoded_bytes > self.max_outstanding_bytes - { - return Err(Error::Capacity("node-log submission")); - } - let permit_count = u32::try_from(submission.encoded_bytes) - .ok() - .filter(|bytes| *bytes != 0) - .ok_or(Error::Capacity("node-log outstanding bytes"))?; - let reservation = Arc::clone(&self.bytes) - .acquire_many_owned(permit_count) - .await - .map_err(|_| Error::RuntimeClosed)?; - let sender = self - .sender - .lock() - .map_err(|_| Error::Node("node-log shipper lock poisoned"))? - .clone() - .ok_or(Error::RuntimeClosed)?; - let slot = sender - .reserve_owned() - .await - .map_err(|_| Error::RuntimeClosed)?; - let limits = self.limits; - let loaded = tokio::task::spawn_blocking(move || submission.load(limits)) - .await - .map_err(Error::FollowerWorkerJoin)??; - let _ordered = self.order.lock().await; - let ticket = self.gate.preview(frame_count)?; - let encoded = match tokio::task::spawn_blocking(move || loaded.encode(ticket, limits)).await - { - Ok(Ok(encoded)) => encoded, - Ok(Err(error)) => return Err(error), - Err(error) => return Err(Error::FollowerWorkerJoin(error)), - }; - self.gate.commit(ticket)?; - let reservation = Arc::new(OutstandingBytes { - _permit: reservation, - }); - let frames = encoded - .into_iter() - .enumerate() - .map(|(offset, encoded)| QueuedFrame { - sequence: ticket.first_sequence().saturating_add(offset as u64), - encoded, - _reservation: Arc::clone(&reservation), - }) - .collect(); - slot.send(QueuedSubmission { frames }); - Ok(ticket) - } - - /// Closes admission and drains every accepted frame to the current epoch. - pub async fn shutdown(&self) -> Result<()> { - self.bytes.close(); - self.sender - .lock() - .map_err(|_| Error::Node("node-log shipper lock poisoned"))? - .take(); - let worker = self - .worker - .lock() - .map_err(|_| Error::Node("node-log shipper lock poisoned"))? - .take(); - if let Some(worker) = worker { - worker.await.map_err(Error::FollowerWorkerJoin)?; - } - self.gate.stop_shipping(); - Ok(()) - } -} - -impl Drop for NodeLogShipper { - fn drop(&mut self) { - self.gate.stop_shipping(); - self.bytes.close(); - } -} - -struct OutstandingBytes { - _permit: OwnedSemaphorePermit, -} - -struct QueuedSubmission { - frames: Vec, -} - -struct QueuedFrame { - sequence: u64, - encoded: Bytes, - _reservation: Arc, -} - -#[expect( - clippy::too_many_arguments, - reason = "the worker keeps the exact log epoch, ensemble and bounded admission explicit" -)] -async fn run_shipper( - mut receiver: mpsc::Receiver, - gate: DurabilityGate, - bytes: Arc, - transport: Arc, - leader: crate::SessionId, - log_epoch: u64, - members: Vec, - max_batch_bytes: u64, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, - interval: Duration, -) { - let mut pending = VecDeque::::new(); - let mut closed = false; - loop { - if pending.is_empty() { - if closed { - bytes.close(); - return; - } - match receiver.recv().await { - Some(submission) => pending.extend(submission.frames), - None => { - bytes.close(); - return; - } - } - } - - let deadline = tokio::time::Instant::now() + interval; - let mut batch = Vec::::new(); - let mut batch_bytes = 0_u64; - loop { - while batch.len() < MAX_BATCH_FRAMES { - let Some(next) = pending.front() else { - break; - }; - let Some(next_bytes) = batch_bytes.checked_add(next.encoded.len() as u64) else { - stop_shipper(&gate, &bytes); - return; - }; - if !batch.is_empty() && next_bytes > max_batch_bytes { - break; - } - if next_bytes > max_batch_bytes { - stop_shipper(&gate, &bytes); - return; - } - let Some(next) = pending.pop_front() else { - stop_shipper(&gate, &bytes); - return; - }; - batch_bytes = next_bytes; - batch.push(next); - } - if batch.len() == MAX_BATCH_FRAMES - || batch_bytes == max_batch_bytes - || closed - || !pending.is_empty() - { - break; - } - match tokio::time::timeout_at(deadline, receiver.recv()).await { - Ok(Some(submission)) => pending.extend(submission.frames), - Ok(None) => { - closed = true; - break; - } - Err(_) => break, - } - } - - let append_bytes = batch - .iter() - .try_fold(0_u64, |total, frame| { - total.checked_add(frame.encoded.len() as u64) - }) - .and_then(|bytes| bytes.checked_mul(members.len() as u64)) - .unwrap_or(u64::MAX); - let result = append_batch( - &gate, - Arc::clone(&transport), - leader, - log_epoch, - &members, - batch, - ) - .await; - telemetry.node_log_append(result.is_ok(), append_bytes); - if result.is_err() { - stop_shipper(&gate, &bytes); - receiver.close(); - return; - } - } -} - -fn stop_shipper(gate: &DurabilityGate, bytes: &Semaphore) { - gate.stop_shipping(); - bytes.close(); -} - -async fn append_batch( - gate: &DurabilityGate, - transport: Arc, - leader: crate::SessionId, - log_epoch: u64, - members: &[NodeId], - batch: Vec, -) -> Result<()> { - let first = batch - .first() - .ok_or(Error::Node("node-log append batch is empty"))? - .sequence; - let last = batch - .last() - .ok_or(Error::Node("node-log append batch is empty"))? - .sequence; - if !batch - .windows(2) - .all(|pair| pair[0].sequence.checked_add(1) == Some(pair[1].sequence)) - { - return Err(Error::Node("node-log append batch is not contiguous")); - } - let frames = batch - .iter() - .map(|frame| frame.encoded.clone()) - .collect::>(); - let covered_through = gate.tiered_through(); - let replies = join_all(members.iter().map(|member| { - let transport = Arc::clone(&transport); - let request = AppendRequest { - leader_session: leader, - log_epoch, - frames: frames.clone(), - covered_through, - }; - let member = *member; - async move { (member, transport.append(member, request).await) } - })) - .await; - let mut acknowledgements = Vec::with_capacity(replies.len()); - for (member, reply) in replies { - let receipt = reply?; - // A queued batch can include an already object-covered prefix. Only - // its uncovered suffix must remain on the follower for a fleet proof. - let required_first = first.max(covered_through.saturating_add(1)); - if receipt.durable_through < last - || receipt.base_sequence == 0 - || receipt.base_sequence > receipt.durable_through.saturating_add(1) - || (last > covered_through && receipt.base_sequence > required_first) - { - return Err(Error::Node("node-log append receipt differs")); - } - acknowledgements.push((member, receipt.durable_through)); - } - for (member, durable_through) in acknowledgements { - gate.acknowledge(member, durable_through)?; - } - Ok(()) -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/node/log_shipper/tests.rs b/crates/crab-cell-runtime/src/node/log_shipper/tests.rs deleted file mode 100644 index b518573b8..000000000 --- a/crates/crab-cell-runtime/src/node/log_shipper/tests.rs +++ /dev/null @@ -1,514 +0,0 @@ -use std::sync::{Arc, Mutex}; - -use futures_util::future::BoxFuture; - -use super::*; -use crate::follower::FollowerReceipt; -use crate::follower::FollowerStore; -use crate::identity::SessionId; -use crate::node::log_transport::{LocalFollowerTransport, RetireRequest, SealRequest, TailRequest}; - -#[derive(Default)] -struct RecordingTransport { - batches: Mutex)>>, - fail: Option, - delay: Option, - receipt: Option, -} - -#[derive(Default)] -struct RecordingTelemetry { - appends: Mutex>, -} - -impl crate::fleet::telemetry::CellTelemetry for RecordingTelemetry { - fn node_log_append(&self, acknowledged: bool, bytes: u64) { - self.appends.lock().unwrap().push((acknowledged, bytes)); - } -} - -struct LostAckTransport { - inner: LocalFollowerTransport, -} - -impl NodeLogTransport for LostAckTransport { - fn append<'a>( - &'a self, - member: NodeId, - request: AppendRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - self.inner.append(member, request).await?; - Err(Error::Node("injected lost follower acknowledgement")) - }) - } - - fn seal<'a>( - &'a self, - member: NodeId, - request: SealRequest, - ) -> BoxFuture<'a, Result> { - self.inner.seal(member, request) - } - - fn retire<'a>( - &'a self, - member: NodeId, - request: RetireRequest, - ) -> BoxFuture<'a, Result> { - self.inner.retire(member, request) - } - - fn tail<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> BoxFuture<'a, Result>> { - self.inner.tail(member, request) - } -} - -impl RecordingTransport { - fn failing(member: NodeId) -> Self { - Self { - batches: Mutex::new(Vec::new()), - fail: Some(member), - delay: None, - receipt: None, - } - } - - fn slow(delay: Duration) -> Self { - Self { - batches: Mutex::new(Vec::new()), - fail: None, - delay: Some(delay), - receipt: None, - } - } - - fn batch_sizes(&self, member: NodeId) -> Vec { - self.batches - .lock() - .unwrap() - .iter() - .filter(|(observed, _)| *observed == member) - .map(|(_, sequences)| sequences.len()) - .collect() - } -} - -impl NodeLogTransport for RecordingTransport { - fn append<'a>( - &'a self, - member: NodeId, - request: AppendRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - if self.fail == Some(member) { - return Err(Error::Node("injected follower failure")); - } - if let Some(delay) = self.delay { - tokio::time::sleep(delay).await; - } - let sequences = request - .frames - .iter() - .map(|frame| { - crab_ltx::inspect_node_frame(frame.clone(), crab_ltx::Limits::default()) - .map(|frame| frame.scope().node_sequence) - .map_err(Error::from) - }) - .collect::>>()?; - let first = *sequences.first().ok_or(Error::Node("empty test append"))?; - let last = *sequences.last().ok_or(Error::Node("empty test append"))?; - self.batches.lock().unwrap().push((member, sequences)); - Ok(self.receipt.unwrap_or(FollowerReceipt { - base_sequence: first, - durable_through: last, - })) - }) - } - - fn seal<'a>( - &'a self, - _member: NodeId, - _request: SealRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("unused test seal")) }) - } - - fn retire<'a>( - &'a self, - _member: NodeId, - _request: RetireRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("unused test retire")) }) - } - - fn tail<'a>( - &'a self, - _member: NodeId, - _request: TailRequest, - ) -> BoxFuture<'a, Result>> { - Box::pin(async { Err(Error::Node("unused test tail")) }) - } -} - -fn session(byte: u8) -> SessionId { - SessionId::from_bytes([byte; 16]) -} - -fn node(byte: u8) -> NodeId { - NodeId::from_bytes([byte; 16]) -} - -fn capture() -> (tempfile::TempDir, crab_ltx::CaptureBatch) { - let directory = tempfile::TempDir::new().unwrap(); - let mut database = crab_ltx::Db::open( - &directory.path().join("shipper.sqlite"), - crab_ltx::Limits::default(), - ) - .unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE items(value)")) - .unwrap(); - let cuts = database.capture().unwrap(); - database.close().unwrap(); - (directory, cuts) -} - -fn submission(cuts: &crab_ltx::CaptureBatch) -> NodeLogSubmission { - NodeLogSubmission::new( - ApplicationId::from_bytes([9; 16]), - CellId::from_bytes([8; 32]), - IncarnationId::from_bytes([7; 16]), - 3, - 4, - cuts, - ) - .unwrap() -} - -#[tokio::test] -async fn concurrent_submissions_stay_ordered_and_require_every_member_ack() { - let (_directory, cuts) = capture(); - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2), node(3)]).unwrap(); - gate.activate_fleet().unwrap(); - let transport = Arc::new(RecordingTransport::default()); - let shipper = NodeLogShipper::start( - gate.clone(), - transport.clone(), - crab_ltx::Limits::default(), - crate::fleet::telemetry::CellTelemetryHandle::default(), - Duration::from_millis(50), - ) - .unwrap(); - - let (first, second) = tokio::join!( - shipper.submit(submission(&cuts)), - shipper.submit(submission(&cuts)) - ); - let first = first.unwrap(); - let second = second.unwrap(); - assert_eq!( - gate.prove(first).await.unwrap().source(), - crate::node::log::DurabilitySource::Fleet - ); - assert_eq!( - gate.prove(second).await.unwrap().source(), - crate::node::log::DurabilitySource::Fleet - ); - shipper.shutdown().await.unwrap(); - - assert_eq!(transport.batch_sizes(node(2)), [2]); - assert_eq!(transport.batch_sizes(node(3)), [2]); - assert!(gate.issue(1).is_err()); -} - -#[tokio::test] -async fn append_telemetry_records_one_result_for_each_batch() { - let (_directory, cuts) = capture(); - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - gate.activate_fleet().unwrap(); - let transport = Arc::new(RecordingTransport::default()); - let telemetry = Arc::new(RecordingTelemetry::default()); - let handle = crate::fleet::telemetry::CellTelemetryHandle::default(); - handle.install(telemetry.clone()).unwrap(); - let shipper = - NodeLogShipper::new_with_telemetry(gate, transport, crab_ltx::Limits::default(), handle) - .unwrap(); - - shipper.submit(submission(&cuts)).await.unwrap(); - shipper.shutdown().await.unwrap(); - - assert_eq!(telemetry.appends.lock().unwrap().len(), 1); - assert!(telemetry.appends.lock().unwrap()[0].0); - assert!(telemetry.appends.lock().unwrap()[0].1 > 0); -} - -#[tokio::test] -async fn splits_large_submission_at_sixty_four_frames() { - let (_directory, mut cuts) = capture(); - cuts.segments = std::iter::repeat_n(cuts.segments[0].clone(), 65).collect(); - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - gate.activate_fleet().unwrap(); - let transport = Arc::new(RecordingTransport::default()); - let shipper = NodeLogShipper::start( - gate.clone(), - transport.clone(), - crab_ltx::Limits::default(), - crate::fleet::telemetry::CellTelemetryHandle::default(), - Duration::from_millis(50), - ) - .unwrap(); - - let ticket = shipper.submit(submission(&cuts)).await.unwrap(); - assert_eq!( - gate.prove(ticket).await.unwrap().source(), - crate::node::log::DurabilitySource::Fleet - ); - shipper.shutdown().await.unwrap(); - - assert_eq!(ticket.first_sequence(), 1); - assert_eq!(ticket.last_sequence(), 65); - assert_eq!(transport.batch_sizes(node(2)), [64, 1]); - assert!(gate.issue(1).is_err()); -} - -#[tokio::test] -async fn covered_queued_prefix_keeps_the_uncovered_suffix_fleet_durable() { - let (_directory, cuts) = capture(); - let limits = crab_ltx::Limits::default(); - let leader = session(1); - let member = node(2); - let gate = DurabilityGate::new(leader, node(1), 2, [member]).unwrap(); - gate.activate_fleet().unwrap(); - let first = gate.issue(1).unwrap(); - let second = gate.issue(1).unwrap(); - gate.prove_object(first).unwrap(); - let follower_directory = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - follower_directory.path().to_owned(), - limits, - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let transport = Arc::new(LocalFollowerTransport::new(member, store.clone())); - let reservation = Arc::new(OutstandingBytes { - _permit: Arc::new(Semaphore::new(1)).acquire_owned().await.unwrap(), - }); - let batch = [first, second] - .into_iter() - .flat_map(|ticket| { - submission(&cuts) - .load(limits) - .unwrap() - .encode(ticket, limits) - .unwrap() - }) - .enumerate() - .map(|(offset, encoded)| QueuedFrame { - sequence: offset as u64 + 1, - encoded, - _reservation: Arc::clone(&reservation), - }) - .collect(); - - append_batch(&gate, transport, leader, 2, &[member], batch) - .await - .unwrap(); - - assert_eq!( - gate.prove(second).await.unwrap().source(), - crate::node::log::DurabilitySource::Fleet - ); - assert_eq!(store.seal(leader, 2).await.unwrap().base_sequence, 2); - assert_eq!(store.read_tail(leader, 2, 2).await.unwrap().len(), 1); -} - -#[tokio::test] -async fn receipts_without_the_uncovered_frame_never_authorize_fleet_proof() { - let (_directory, cuts) = capture(); - for (base_sequence, durable_through) in [(2, 1), (0, 1), (3, 1), (1, 0)] { - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - gate.activate_fleet().unwrap(); - let transport = Arc::new(RecordingTransport { - receipt: Some(FollowerReceipt { - base_sequence, - durable_through, - }), - ..RecordingTransport::default() - }); - let shipper = - NodeLogShipper::new(gate.clone(), transport, crab_ltx::Limits::default()).unwrap(); - let ticket = shipper.submit(submission(&cuts)).await.unwrap(); - assert!( - tokio::time::timeout(Duration::from_secs(1), gate.wait_followers(ticket)) - .await - .unwrap() - .is_err() - ); - shipper.shutdown().await.unwrap(); - assert!( - tokio::time::timeout(Duration::from_millis(10), gate.prove(ticket)) - .await - .is_err() - ); - gate.prove_object(ticket).unwrap(); - assert_eq!( - gate.prove(ticket).await.unwrap().source(), - crate::node::log::DurabilitySource::Object - ); - } -} - -#[tokio::test] -async fn follower_failure_stops_fleet_issuance_but_preserves_object_proof() { - let (_directory, cuts) = capture(); - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2), node(3)]).unwrap(); - gate.activate_fleet().unwrap(); - let transport = Arc::new(RecordingTransport::failing(node(3))); - let shipper = NodeLogShipper::start( - gate.clone(), - transport, - crab_ltx::Limits::default(), - crate::fleet::telemetry::CellTelemetryHandle::default(), - Duration::from_millis(1), - ) - .unwrap(); - - let ticket = shipper.submit(submission(&cuts)).await.unwrap(); - for _ in 0..100 { - if gate.shipping_scope().is_err() { - break; - } - tokio::time::sleep(Duration::from_millis(1)).await; - } - assert!(gate.shipping_scope().is_err()); - let retry = tokio::time::timeout(Duration::from_secs(1), shipper.submit(submission(&cuts))) - .await - .expect("failed shipper must release blocked byte admission"); - assert!(retry.is_err()); - shipper.shutdown().await.unwrap(); - - assert!(gate.issue(1).is_err()); - gate.prove_object(ticket).unwrap(); - assert_eq!( - gate.prove(ticket).await.unwrap().source(), - crate::node::log::DurabilitySource::Object - ); -} - -#[tokio::test] -async fn slow_follower_delays_fleet_proof_until_every_member_acknowledges() { - let (_directory, cuts) = capture(); - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2), node(3)]).unwrap(); - gate.activate_fleet().unwrap(); - let delay = Duration::from_millis(40); - let transport = Arc::new(RecordingTransport::slow(delay)); - let shipper = NodeLogShipper::start( - gate.clone(), - transport, - crab_ltx::Limits::default(), - crate::fleet::telemetry::CellTelemetryHandle::default(), - Duration::from_millis(1), - ) - .unwrap(); - - let started = tokio::time::Instant::now(); - let ticket = shipper.submit(submission(&cuts)).await.unwrap(); - let proof = gate.prove(ticket).await.unwrap(); - assert_eq!(proof.source(), crate::node::log::DurabilitySource::Fleet); - assert!(started.elapsed() >= delay); - shipper.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn lost_ack_keeps_the_durable_follower_tail_without_issuing_fleet_proof() { - let (_directory, cuts) = capture(); - let leader = session(1); - let member = node(2); - let gate = DurabilityGate::new(leader, node(1), 2, [member]).unwrap(); - gate.activate_fleet().unwrap(); - let follower_directory = tempfile::TempDir::new().unwrap(); - let store = FollowerStore::open( - follower_directory.path().to_owned(), - crab_ltx::Limits::default(), - crab_ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let transport: Arc = Arc::new(LostAckTransport { - inner: LocalFollowerTransport::new(member, store.clone()), - }); - let shipper = NodeLogShipper::start( - gate.clone(), - transport, - crab_ltx::Limits::default(), - crate::fleet::telemetry::CellTelemetryHandle::default(), - Duration::from_millis(1), - ) - .unwrap(); - - let ticket = shipper.submit(submission(&cuts)).await.unwrap(); - shipper.shutdown().await.unwrap(); - - assert!(gate.issue(1).is_err()); - assert_eq!(store.seal(leader, 2).await.unwrap().durable_through, 1); - assert_eq!(store.read_tail(leader, 2, 1).await.unwrap().len(), 1); - gate.prove_object(ticket).unwrap(); - assert_eq!( - gate.prove(ticket).await.unwrap().source(), - crate::node::log::DurabilitySource::Object - ); -} - -#[tokio::test] -async fn oversized_submission_is_rejected_before_waiting_for_capacity() { - let (_directory, mut cuts) = capture(); - let mut info = cuts.segments[0].info().clone(); - info.size_bytes = crab_ltx::Limits::default() - .max_capture_bytes - .saturating_add((MAX_BATCH_FRAMES as u64) * NODE_FRAME_HEADER_BYTES); - cuts.segments[0] = crab_ltx::LocalSegment::new(cuts.segments[0].path().to_owned(), info); - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - let shipper = NodeLogShipper::new( - gate, - Arc::new(RecordingTransport::default()), - crab_ltx::Limits::default(), - ) - .unwrap(); - - assert!(matches!( - shipper.submit(submission(&cuts)).await, - Err(Error::Capacity("node-log submission")) - )); - shipper.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn encoding_failure_does_not_consume_a_node_sequence() { - let (_directory, cuts) = capture(); - let mut invalid = cuts.clone(); - let mut info = invalid.segments[0].info().clone(); - info.blake3 = [0; 32]; - invalid.segments[0] = crab_ltx::LocalSegment::new(invalid.segments[0].path().to_owned(), info); - let gate = DurabilityGate::new(session(1), node(1), 2, [node(2)]).unwrap(); - gate.activate_fleet().unwrap(); - let shipper = NodeLogShipper::new( - gate.clone(), - Arc::new(RecordingTransport::default()), - crab_ltx::Limits::default(), - ) - .unwrap(); - - assert!(shipper.submit(submission(&invalid)).await.is_err()); - let ticket = shipper.submit(submission(&cuts)).await.unwrap(); - - assert_eq!(ticket.first_sequence(), 1); - assert_eq!( - gate.prove(ticket).await.unwrap().source(), - crate::node::log::DurabilitySource::Fleet - ); - shipper.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/src/node/log_state.rs b/crates/crab-cell-runtime/src/node/log_state.rs deleted file mode 100644 index 44e6efd8b..000000000 --- a/crates/crab-cell-runtime/src/node/log_state.rs +++ /dev/null @@ -1,397 +0,0 @@ -//! Node-log lifecycle state and recovery claims. -use crate::identity::NodeId; -use crate::identity::{Digest, SessionId}; -use crate::{Error, Result}; - -pub(crate) const RECOVERY_CLAIM_LIFETIME_MS: i64 = 30_000; -pub(crate) const MAX_NODE_LOG_MEMBERS: usize = 2; - -/// Authoritative lifecycle state for one node-session durability log. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum NodeLogPhase { - /// The log accepts appends from its enrolled leader. - Open, - /// The log is being recovered by a claimed session. - Recovering, - /// The log accepts no further appends and can be recovered. - Sealed, - /// The log's lane was deleted after object coverage. - Retired, -} - -/// Bounded lease held by the live node session recovering a failed owner. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct NodeRecoveryClaim { - claimant: SessionId, - generation: u64, - expires_at_ms: i64, -} - -impl NodeRecoveryClaim { - /// Returns the session that holds the claim. - #[must_use] - pub const fn claimant(&self) -> SessionId { - self.claimant - } - - /// Returns the claim generation. - #[must_use] - pub const fn generation(&self) -> u64 { - self.generation - } - - /// Returns the logical time the claim expires. - #[must_use] - pub const fn expires_at_ms(&self) -> i64 { - self.expires_at_ms - } - - pub(crate) fn new(claimant: SessionId, generation: u64, now_ms: i64) -> Result { - let expires_at_ms = now_ms - .checked_add(RECOVERY_CLAIM_LIFETIME_MS) - .ok_or(Error::Node("node recovery claim time overflow"))?; - let claim = Self { - claimant, - generation, - expires_at_ms, - }; - claim.validate()?; - Ok(claim) - } - - pub(crate) fn from_parts( - claimant: SessionId, - generation: u64, - expires_at_ms: i64, - ) -> Result { - let claim = Self { - claimant, - generation, - expires_at_ms, - }; - claim.validate()?; - Ok(claim) - } - - pub(crate) fn renew(self, claimant: SessionId, now_ms: i64) -> Result { - if self.claimant != claimant || now_ms >= self.expires_at_ms { - return Err(Error::Fenced); - } - let renewed = Self::new(claimant, self.generation, now_ms)?; - if renewed.expires_at_ms <= self.expires_at_ms { - return Err(Error::Node("node recovery claim expiry did not advance")); - } - Ok(renewed) - } - - pub(crate) fn take_over(self, claimant: SessionId, now_ms: i64) -> Result { - if now_ms < self.expires_at_ms { - return Err(Error::Node("node recovery is already claimed")); - } - Self::new( - claimant, - self.generation - .checked_add(1) - .ok_or(Error::Node("node recovery claim generation overflow"))?, - now_ms, - ) - } - - pub(crate) fn validate(&self) -> Result<()> { - if self.claimant.as_bytes().iter().all(|byte| *byte == 0) - || self.generation == 0 - || self.expires_at_ms < 0 - { - return Err(Error::Node("node recovery claim is invalid")); - } - Ok(()) - } -} - -/// CAS-protected follower ensemble and object-coverage watermark. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct NodeLogStatus { - phase: NodeLogPhase, - epoch: u64, - members: Vec, - active: bool, - tiered_through: u64, - recovery: Option, - recovery_manifest: Option, -} - -impl NodeLogStatus { - pub(crate) fn open(leader: NodeId, epoch: u64, members: Vec) -> Result { - let status = Self { - phase: NodeLogPhase::Open, - epoch, - members, - active: false, - tiered_through: 0, - recovery: None, - recovery_manifest: None, - }; - status.validate(leader)?; - Ok(status) - } - - pub(crate) fn from_parts( - leader: NodeId, - phase: NodeLogPhase, - epoch: u64, - members: Vec, - active: bool, - tiered_through: u64, - recovery: Option, - recovery_manifest: Option, - ) -> Result { - let status = Self { - phase, - epoch, - members, - active, - tiered_through, - recovery, - recovery_manifest, - }; - status.validate(leader)?; - Ok(status) - } - - /// Returns the durable log phase. - #[must_use] - pub const fn phase(&self) -> NodeLogPhase { - self.phase - } - - /// Returns the node-log epoch. - #[must_use] - pub const fn epoch(&self) -> u64 { - self.epoch - } - - /// Returns the members the log is enrolled with. - #[must_use] - pub fn members(&self) -> &[NodeId] { - &self.members - } - - /// Reports whether the log currently accepts appends. - #[must_use] - pub const fn active(&self) -> bool { - self.active - } - - /// Returns the highest sequence tiered durably. - #[must_use] - pub const fn tiered_through(&self) -> u64 { - self.tiered_through - } - - /// Returns the recovery claim, while one is held. - #[must_use] - pub const fn recovery(&self) -> Option { - self.recovery - } - - /// Returns the published recovery manifest digest, once recovery sealed the - /// log. - #[must_use] - pub const fn recovery_manifest(&self) -> Option { - self.recovery_manifest - } - - pub(crate) fn activate(&self, leader: NodeId) -> Result { - if self.phase != NodeLogPhase::Open || self.active { - return Err(Error::Node("node log cannot be activated")); - } - Self::from_parts( - leader, - self.phase, - self.epoch, - self.members.clone(), - true, - self.tiered_through, - None, - None, - ) - } - - pub(crate) fn advance_tiered(&self, leader: NodeId, through: u64) -> Result { - if self.phase != NodeLogPhase::Open || through < self.tiered_through { - return Err(Error::Node("node log object coverage regressed")); - } - Self::from_parts( - leader, - self.phase, - self.epoch, - self.members.clone(), - self.active, - through, - None, - None, - ) - } - - pub(crate) fn begin_recovery( - &self, - leader: NodeId, - claimant: SessionId, - now_ms: i64, - ) -> Result { - match (self.phase, self.recovery) { - (NodeLogPhase::Open, None) => Self::from_parts( - leader, - NodeLogPhase::Recovering, - self.epoch, - self.members.clone(), - self.active, - self.tiered_through, - Some(NodeRecoveryClaim::new(claimant, 1, now_ms)?), - None, - ), - (NodeLogPhase::Recovering, Some(current)) if current.claimant == claimant => { - if now_ms < current.expires_at_ms { - return Ok(self.clone()); - } - Self::from_parts( - leader, - NodeLogPhase::Recovering, - self.epoch, - self.members.clone(), - self.active, - self.tiered_through, - Some(current.take_over(claimant, now_ms)?), - None, - ) - } - (NodeLogPhase::Recovering, Some(current)) => Self::from_parts( - leader, - NodeLogPhase::Recovering, - self.epoch, - self.members.clone(), - self.active, - self.tiered_through, - Some(current.take_over(claimant, now_ms)?), - None, - ), - _ => Err(Error::Node("node log cannot enter recovery")), - } - } - - pub(crate) fn renew_recovery( - &self, - leader: NodeId, - claimant: SessionId, - generation: u64, - now_ms: i64, - ) -> Result { - let current = self - .recovery - .filter(|claim| claim.generation == generation) - .ok_or(Error::Fenced)?; - Self::from_parts( - leader, - NodeLogPhase::Recovering, - self.epoch, - self.members.clone(), - self.active, - self.tiered_through, - Some(current.renew(claimant, now_ms)?), - None, - ) - } - - pub(crate) fn seal_recovery( - &self, - leader: NodeId, - claimant: SessionId, - generation: u64, - manifest: Option, - ) -> Result { - let current = self.recovery.ok_or(Error::Fenced)?; - if self.phase != NodeLogPhase::Recovering - || current.claimant != claimant - || current.generation != generation - { - return Err(Error::Fenced); - } - Self::from_parts( - leader, - NodeLogPhase::Sealed, - self.epoch, - self.members.clone(), - self.active, - self.tiered_through, - None, - manifest, - ) - } - - pub(crate) fn permits_append(&self, leader: NodeId, member: NodeId, epoch: u64) -> Result<()> { - self.validate(leader)?; - if self.phase != NodeLogPhase::Open - || self.epoch != epoch - || !self.members.contains(&member) - { - return Err(Error::PeerAuthorization( - "node-log append is outside the enrolled ensemble", - )); - } - Ok(()) - } - - pub(crate) fn permits_recovery_read( - &self, - leader: NodeId, - claimant: SessionId, - member: NodeId, - epoch: u64, - now_ms: i64, - ) -> Result<()> { - self.validate(leader)?; - let claim = self - .recovery - .ok_or(Error::PeerAuthorization("node-log recovery is not claimed"))?; - if self.phase != NodeLogPhase::Recovering - || self.epoch != epoch - || claim.claimant != claimant - || claim.expires_at_ms <= now_ms - || !self.members.contains(&member) - { - return Err(Error::PeerAuthorization( - "node-log recovery claim or member differs", - )); - } - Ok(()) - } - - pub(crate) fn validate(&self, leader: NodeId) -> Result<()> { - if self.epoch == 0 - || self.members.is_empty() - || self.members.len() > MAX_NODE_LOG_MEMBERS - || self.members.contains(&leader) - || self - .members - .iter() - .any(|member| member.as_bytes().iter().all(|byte| *byte == 0)) - || !self - .members - .windows(2) - .all(|pair| pair[0].as_bytes() < pair[1].as_bytes()) - { - return Err(Error::Node("node log ensemble is invalid")); - } - match (self.phase, self.recovery, self.recovery_manifest) { - (NodeLogPhase::Open, None, None) => {} - (NodeLogPhase::Recovering, Some(claim), None) => claim.validate()?, - (NodeLogPhase::Sealed, None, manifest) | (NodeLogPhase::Retired, None, manifest) => { - if manifest.is_some_and(|digest| digest.as_bytes().iter().all(|byte| *byte == 0)) { - return Err(Error::Node("node recovery manifest digest is zero")); - } - } - _ => return Err(Error::Node("node log state is invalid")), - } - Ok(()) - } -} diff --git a/crates/crab-cell-runtime/src/node/log_transport.rs b/crates/crab-cell-runtime/src/node/log_transport.rs deleted file mode 100644 index 06734315c..000000000 --- a/crates/crab-cell-runtime/src/node/log_transport.rs +++ /dev/null @@ -1,238 +0,0 @@ -//! Node-log transport requests: append, retire, seal, and tail. -use bytes::Bytes; -use futures_util::future::BoxFuture; - -use crate::follower::FollowerStore; -use crate::follower::{FollowerReceipt, FollowerTailPage}; -use crate::identity::NodeId; -use crate::identity::SessionId; -use crate::{Error, Result}; - -/// One ordered follower append with the leader's safe truncation watermark. -pub struct AppendRequest { - /// Leader issuing the append. - pub leader_session: SessionId, - /// Node-log epoch the append belongs to. - pub log_epoch: u64, - /// Ordered frames to append. - pub frames: Vec, - /// Highest sequence the leader knows is durable elsewhere, so the follower - /// may truncate below it. - pub covered_through: u64, -} - -/// Recovery request that atomically closes one follower lane to new appends. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct SealRequest { - /// Leader whose lane is being sealed. - pub leader_session: SessionId, - /// Epoch of the lane to seal. - pub log_epoch: u64, -} - -/// Leader-authorized deletion of one fully object-covered follower lane. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct RetireRequest { - /// Leader authorizing the retirement. - pub leader_session: SessionId, - /// Epoch of the lane to retire. - pub log_epoch: u64, - /// Highest sequence object storage covers. - pub covered_through: u64, -} - -/// Bounded read of one already sealed follower tail. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct TailRequest { - /// Leader whose sealed tail is read. - pub leader_session: SessionId, - /// Epoch of the sealed lane. - pub log_epoch: u64, - /// First sequence the follower should return. - pub first_sequence: u64, -} - -/// Authenticated peer transport used by node-log shipping and recovery. -/// -/// Implementations own mTLS, peer enrollment, request deadlines, and response -/// size limits. Follower storage and LTX verification remain runtime concerns. -pub trait NodeLogTransport: Send + Sync { - /// Appends one ordered batch to a member's lane. - fn append<'a>( - &'a self, - member: NodeId, - request: AppendRequest, - ) -> BoxFuture<'a, Result>; - - /// Closes the member's lane to new appends. - fn seal<'a>( - &'a self, - member: NodeId, - request: SealRequest, - ) -> BoxFuture<'a, Result>; - - /// Deletes a fully object-covered member lane. - fn retire<'a>( - &'a self, - member: NodeId, - request: RetireRequest, - ) -> BoxFuture<'a, Result>; - - /// Reads the sealed tail starting at the requested sequence. - fn tail<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> BoxFuture<'a, Result>>; - - /// Reads one bounded page of a sealed follower tail. - /// - /// Implementations with a paged transport should override this method. - /// The default keeps older transports source-compatible while applying the - /// same one-megabyte/4096-frame page boundary in memory. - fn tail_page<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - let frames = self.tail(member, request).await?; - let mut bytes = 0_usize; - let mut count = 0_usize; - for frame in &frames { - let next_bytes = bytes - .checked_add(frame.len()) - .ok_or(Error::Node("follower tail page byte count overflow"))?; - if count == 4096 || (count != 0 && next_bytes > 1 << 20) { - break; - } - bytes = next_bytes; - count += 1; - } - let next_sequence = if count < frames.len() { - let count = u64::try_from(count) - .map_err(|_| Error::Node("follower tail page frame count overflow"))?; - Some( - request - .first_sequence - .checked_add(count) - .ok_or(Error::Node("follower tail page sequence overflow"))?, - ) - } else { - None - }; - Ok(FollowerTailPage { - frames: frames.into_iter().take(count).collect(), - next_sequence, - }) - }) - } -} - -/// In-process transport for deterministic tests and single-process recovery. -#[derive(Clone)] -pub struct LocalFollowerTransport { - member: NodeId, - store: FollowerStore, -} - -impl LocalFollowerTransport { - /// Binds the local transport to one member and follower store. - #[must_use] - pub const fn new(member: NodeId, store: FollowerStore) -> Self { - Self { member, store } - } - - fn validate_member(&self, member: NodeId) -> Result<()> { - if member != self.member { - return Err(Error::PeerAuthorization( - "node-log transport selected a different follower", - )); - } - Ok(()) - } -} - -impl NodeLogTransport for LocalFollowerTransport { - fn append<'a>( - &'a self, - member: NodeId, - request: AppendRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - self.validate_member(member)?; - self.store - .append( - request.leader_session, - request.log_epoch, - request.frames, - request.covered_through, - ) - .await - }) - } - - fn seal<'a>( - &'a self, - member: NodeId, - request: SealRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - self.validate_member(member)?; - self.store - .seal(request.leader_session, request.log_epoch) - .await - }) - } - - fn retire<'a>( - &'a self, - member: NodeId, - request: RetireRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - self.validate_member(member)?; - self.store - .retire( - request.leader_session, - request.log_epoch, - request.covered_through, - ) - .await - }) - } - - fn tail<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> BoxFuture<'a, Result>> { - Box::pin(async move { - self.validate_member(member)?; - self.store - .read_tail( - request.leader_session, - request.log_epoch, - request.first_sequence, - ) - .await - }) - } - - fn tail_page<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async move { - self.validate_member(member)?; - self.store - .read_tail_page( - request.leader_session, - request.log_epoch, - request.first_sequence, - ) - .await - }) - } -} diff --git a/crates/crab-cell-runtime/src/node/tests.rs b/crates/crab-cell-runtime/src/node/tests.rs deleted file mode 100644 index 9775fd27a..000000000 --- a/crates/crab-cell-runtime/src/node/tests.rs +++ /dev/null @@ -1,173 +0,0 @@ -use crate::node::directory::RecoveryCandidateRecord; -use crate::node::directory::RecoveryCandidateWindow; -use std::sync::Arc; - -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::Store; -use ed25519_dalek::SigningKey; -use futures_util::future::BoxFuture; -use object_store::{memory::InMemory, path::Path}; - -use super::*; -use crate::fleet::placement::PlacementPlanner; -use crate::identity::{ApplicationId, CellTarget, NamespaceId, TenantId}; -use crate::node::log_transport::{ - AppendRequest, NodeLogTransport, RetireRequest, SealRequest, TailRequest, -}; -use crate::peer::{PeerOperation, PeerPrincipal, PeerSigner, wire as peer_wire}; - -// Capability modules keep the fixture-heavy suite navigable; the shared -// fixtures stay here. -mod candidates; -mod log; -mod placement; -mod records; -mod sessions; - -const NOW_MS: i64 = 1_000_000; - -fn node(session: SessionId) -> NodeId { - NodeId::from_bytes(*session.as_bytes()) -} - -struct UnavailableFollowerTransport; - -impl NodeLogTransport for UnavailableFollowerTransport { - fn append<'a>( - &'a self, - _member: NodeId, - _request: AppendRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("injected unavailable follower")) }) - } - - fn seal<'a>( - &'a self, - _member: NodeId, - _request: SealRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("injected unavailable follower")) }) - } - - fn retire<'a>( - &'a self, - _member: NodeId, - _request: RetireRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("injected unavailable follower")) }) - } - - fn tail<'a>( - &'a self, - _member: NodeId, - _request: TailRequest, - ) -> BoxFuture<'a, Result>> { - Box::pin(async { Err(Error::Node("injected unavailable follower")) }) - } -} - -fn advertisement(key: &SigningKey, progress: u64, issued_at_ms: i64) -> NodeAdvertisement { - advertisement_for(SessionId::from_bytes([1; 16]), key, progress, issued_at_ms) -} - -fn advertisement_for( - session: SessionId, - key: &SigningKey, - progress: u64, - issued_at_ms: i64, -) -> NodeAdvertisement { - advertisement_for_node_capacity( - node(session), - session, - key, - progress, - issued_at_ms, - NodeCapacity { - free_memory_bytes: 1_000, - free_disk_bytes: 2_000, - follower_free_bytes: 2_000, - follower_retained_bytes: 800, - job_credits: 3, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - }, - ) -} - -fn advertisement_for_capacity( - session: SessionId, - key: &SigningKey, - progress: u64, - issued_at_ms: i64, - capacity: NodeCapacity, -) -> NodeAdvertisement { - advertisement_for_node_capacity( - node(session), - session, - key, - progress, - issued_at_ms, - capacity, - ) -} - -fn advertisement_for_node_capacity( - node: NodeId, - session: SessionId, - key: &SigningKey, - progress: u64, - issued_at_ms: i64, - capacity: NodeCapacity, -) -> NodeAdvertisement { - advertisement_for_node_capacity_in_domain( - node, - session, - key, - progress, - issued_at_ms, - NodeFailureDomain::default(), - capacity, - ) -} - -fn advertisement_for_node_capacity_in_domain( - node: NodeId, - session: SessionId, - key: &SigningKey, - progress: u64, - issued_at_ms: i64, - failure_domain: NodeFailureDomain, - capacity: NodeCapacity, -) -> NodeAdvertisement { - NodeAdvertisement::sign( - node, - session, - "https://node-1.internal:8789".into(), - Digest::from_bytes([2; 32]), - Digest::from_bytes([3; 32]), - Digest::from_bytes([4; 32]), - Digest::from_bytes([5; 32]), - key, - progress, - issued_at_ms, - issued_at_ms + 10_000, - vec![Digest::from_bytes([6; 32]), Digest::from_bytes([7; 32])], - vec![1], - failure_domain, - capacity, - ) - .unwrap() -} - -fn directory() -> NodeDirectory { - NodeDirectory::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("root"), - [9; 16], - ), - Digest::from_bytes([2; 32]), - Digest::from_bytes([4; 32]), - Digest::from_bytes([5; 32]), - ) -} diff --git a/crates/crab-cell-runtime/src/node/tests/candidates.rs b/crates/crab-cell-runtime/src/node/tests/candidates.rs deleted file mode 100644 index 3872a3ce2..000000000 --- a/crates/crab-cell-runtime/src/node/tests/candidates.rs +++ /dev/null @@ -1,463 +0,0 @@ -//! Recovery-candidate windows and scan snapshots. - -use super::*; - -#[test] -fn recovery_candidate_window_rotates_without_growing_with_directory_size() { - let sessions = [ - SessionId::from_bytes([1; 16]), - SessionId::from_bytes([2; 16]), - SessionId::from_bytes([3; 16]), - SessionId::from_bytes([4; 16]), - ]; - let mut first = RecoveryCandidateWindow::with_start([2; 16], 2); - for session in sessions { - first.push(session); - } - assert_eq!( - first.finish(), - [ - SessionId::from_bytes([2; 16]), - SessionId::from_bytes([3; 16]) - ] - ); - - let mut wrapped = RecoveryCandidateWindow::with_start([4; 16], 2); - for session in sessions { - wrapped.push(session); - } - assert_eq!( - wrapped.finish(), - [ - SessionId::from_bytes([4; 16]), - SessionId::from_bytes([3; 16]) - ] - ); -} - -#[tokio::test] -async fn cloned_directories_share_only_a_fresh_recovery_scan_snapshot() { - let directory = directory(); - let clone = directory.clone(); - let first = directory - .recovery_scan_snapshot(NOW_MS, false) - .await - .unwrap(); - let reused = clone - .recovery_scan_snapshot(NOW_MS + 1, false) - .await - .unwrap(); - assert!(Arc::ptr_eq(&first, &reused)); - - let refreshed = clone - .recovery_scan_snapshot(NOW_MS + RECOVERY_SCAN_CACHE_TTL_MS, false) - .await - .unwrap(); - assert!(!Arc::ptr_eq(&first, &refreshed)); -} - -#[tokio::test] -async fn empty_live_node_snapshot_is_reused_within_its_ttl() { - let directory = directory(); - let first = directory - .recovery_scan_snapshot(NOW_MS, true) - .await - .unwrap(); - assert!(first.live_nodes.is_empty()); - let reused = directory - .recovery_scan_snapshot(NOW_MS + 1, true) - .await - .unwrap(); - assert!(Arc::ptr_eq(&first, &reused)); -} - -#[test] -fn recovery_candidate_snapshot_rechecks_claim_and_expiry() { - let record = RecoveryCandidateRecord { - session: SessionId::from_bytes([1; 16]), - expires_at_ms: NOW_MS + 10, - claimant: Some(SessionId::from_bytes([2; 16])), - claim_expires_at_ms: Some(NOW_MS + 20), - active: true, - phase: NodeLogPhase::Recovering, - members: Vec::new(), - }; - let other = SessionId::from_bytes([3; 16]); - assert!(!record.eligible_for(other, NOW_MS + 15)); - assert!(record.eligible_for(other, NOW_MS + 20)); - assert!(record.eligible_for(record.claimant.unwrap(), NOW_MS + 15)); -} - -#[tokio::test] -async fn recovery_claim_rechecks_signed_admission_before_fencing() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let leader = SessionId::from_bytes([1; 16]); - let claimant = SessionId::from_bytes([2; 16]); - directory - .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - directory - .create( - advertisement_for_capacity(claimant, &key, 1, NOW_MS + 1, NodeCapacity::default()), - NOW_MS + 1, - ) - .await - .unwrap(); - - assert!(matches!( - directory - .claim_expired_for_recovery(leader, claimant, NOW_MS + 10_000) - .await, - Err(Error::Capacity("node recovery claimant is not eligible")) - )); - assert!( - directory - .takeover_proof(leader, claimant, NOW_MS + 10_000) - .await - .unwrap() - .is_none() - ); -} - -#[tokio::test] -async fn expired_enrolled_log_becomes_a_renewable_recovery_claim() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let leader = SessionId::from_bytes([1; 16]); - let member = SessionId::from_bytes([2; 16]); - let claimant = SessionId::from_bytes([3; 16]); - let created = directory - .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let member_record = directory - .create(advertisement_for(member, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let enrolled = directory - .recruit_log(&created, 7, 1, 2, NOW_MS + 1) - .await - .unwrap(); - directory.activate_log(&enrolled, NOW_MS + 2).await.unwrap(); - let claimant_record = directory - .create(advertisement_for(claimant, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - directory - .refresh( - &member_record, - advertisement_for(member, &key, 2, NOW_MS + 9_000), - NOW_MS + 9_000, - ) - .await - .unwrap(); - directory - .refresh( - &claimant_record, - advertisement_for(claimant, &key, 2, NOW_MS + 9_000), - NOW_MS + 9_000, - ) - .await - .unwrap(); - - assert_eq!( - directory - .recovery_candidates(claimant, NOW_MS + 10_000, 2) - .await - .unwrap(), - [leader] - ); - let fenced = directory - .claim_expired(leader, claimant, NOW_MS + 10_000) - .await - .unwrap(); - let log = fenced.log().unwrap(); - assert_eq!(log.phase(), NodeLogPhase::Recovering); - assert_eq!(log.recovery().unwrap().claimant(), claimant); - assert!(matches!( - fenced.direct_takeover(), - Err(Error::PendingPublication) - )); - directory - .authorize_log_recovery(leader, claimant, node(member), 7, NOW_MS + 10_001) - .await - .unwrap(); - assert!( - directory - .authorize_log_append(leader, node(member), 7, 0, NOW_MS + 10_001) - .await - .is_err() - ); - - let renewed = directory - .refresh_recovery_claim(&fenced, NOW_MS + 15_000) - .await - .unwrap(); - assert_eq!(renewed.claim_generation(), fenced.claim_generation()); - assert!(renewed.claim_expires_at_ms() > fenced.claim_expires_at_ms()); - let sealed = directory - .seal_recovery(&renewed, None, NOW_MS + 15_001) - .await - .unwrap(); - assert_eq!(sealed.log().phase(), NodeLogPhase::Sealed); - assert!( - directory - .recovery_candidates(claimant, NOW_MS + 15_002, 2) - .await - .unwrap() - .is_empty() - ); - assert_eq!( - directory - .takeover_proof(leader, claimant, NOW_MS + 15_002) - .await - .unwrap() - .unwrap() - .session(), - leader - ); -} - -#[tokio::test] -async fn live_original_follower_is_the_only_affine_recovery_candidate() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let leader = SessionId::from_bytes([1; 16]); - let follower = SessionId::from_bytes([2; 16]); - let follower_node = NodeId::from_bytes([9; 16]); - let leader_record = directory - .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let follower_record = directory - .create( - advertisement_for_node_capacity( - follower_node, - follower, - &key, - 1, - NOW_MS, - NodeCapacity { - free_memory_bytes: 1_000, - free_disk_bytes: 2_000, - follower_free_bytes: 2_000, - follower_retained_bytes: 0, - job_credits: 3, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - }, - ), - NOW_MS, - ) - .await - .unwrap(); - let enrolled = directory - .recruit_log(&leader_record, 7, 1, 2, NOW_MS + 1) - .await - .unwrap(); - directory.activate_log(&enrolled, NOW_MS + 2).await.unwrap(); - let follower_record = directory - .refresh( - &follower_record, - advertisement_for_node_capacity( - follower_node, - follower, - &key, - 2, - NOW_MS + 9_000, - NodeCapacity { - free_memory_bytes: 1_000, - free_disk_bytes: 2_000, - follower_free_bytes: 2_000, - follower_retained_bytes: 0, - job_credits: 3, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - }, - ), - NOW_MS + 9_000, - ) - .await - .unwrap(); - - assert_eq!( - directory - .recovery_candidates_for_node(follower, follower_node, NOW_MS + 10_000, 2) - .await - .unwrap(), - [leader] - ); - assert!( - directory - .recovery_candidates_without_live_followers(follower, NOW_MS + 10_000, 2) - .await - .unwrap() - .is_empty() - ); - let fenced = directory - .claim_expired(leader, follower, NOW_MS + 10_000) - .await - .unwrap(); - assert_eq!(fenced.claimant(), follower); - directory - .seal_recovery(&fenced, None, NOW_MS + 10_001) - .await - .unwrap(); - assert_eq!( - directory - .preferred_recovery_node(leader, NOW_MS + 10_002) - .await - .unwrap() - .unwrap() - .session(), - follower - ); - let drained = directory - .refresh( - &follower_record, - advertisement_for_node_capacity( - follower_node, - follower, - &key, - 3, - NOW_MS + 10_003, - NodeCapacity { - free_memory_bytes: 0, - free_disk_bytes: 2_000, - follower_free_bytes: 2_000, - follower_retained_bytes: 0, - job_credits: 3, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - }, - ), - NOW_MS + 10_003, - ) - .await - .unwrap(); - assert!( - directory - .preferred_recovery_node(leader, NOW_MS + 10_004) - .await - .unwrap() - .is_none() - ); - drop(drained); -} - -#[tokio::test] -async fn non_member_recovery_candidate_is_allowed_when_all_followers_are_expired() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let leader = SessionId::from_bytes([1; 16]); - let first_member = SessionId::from_bytes([2; 16]); - let second_member = SessionId::from_bytes([3; 16]); - let fallback = SessionId::from_bytes([4; 16]); - let leader_record = directory - .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let first_record = directory - .create(advertisement_for(first_member, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let second_record = directory - .create(advertisement_for(second_member, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let fallback_record = directory - .create(advertisement_for(fallback, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let enrolled = directory - .recruit_log(&leader_record, 7, 1, 8, NOW_MS + 1) - .await - .unwrap(); - directory.activate_log(&enrolled, NOW_MS + 2).await.unwrap(); - let members = enrolled.advertisement().log().unwrap().members().to_vec(); - let fallback_record = [first_record, second_record, fallback_record] - .into_iter() - .find(|record| !members.contains(&record.advertisement().node())) - .expect("the bounded two-member log leaves one non-member"); - let fallback_session = fallback_record.advertisement().session(); - directory - .refresh( - &fallback_record, - advertisement_for(fallback_session, &key, 2, NOW_MS + 10_000), - NOW_MS + 10_000, - ) - .await - .unwrap(); - - assert_eq!( - directory - .recovery_candidates_without_live_followers(fallback_session, NOW_MS + 10_000, 2) - .await - .unwrap(), - [leader] - ); -} - -#[tokio::test] -async fn expired_recovery_claim_moves_to_a_new_live_claimant_and_fences_the_old_one() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let leader = SessionId::from_bytes([1; 16]); - let member = SessionId::from_bytes([2; 16]); - let first_claimant = SessionId::from_bytes([3; 16]); - let second_claimant = SessionId::from_bytes([4; 16]); - let created = directory - .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - directory - .create(advertisement_for(member, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let enrolled = directory - .recruit_log(&created, 7, 1, 2, NOW_MS + 1) - .await - .unwrap(); - directory.activate_log(&enrolled, NOW_MS + 2).await.unwrap(); - directory - .create( - advertisement_for(first_claimant, &key, 1, NOW_MS + 9_000), - NOW_MS + 9_000, - ) - .await - .unwrap(); - let first = directory - .claim_expired(leader, first_claimant, NOW_MS + 10_000) - .await - .unwrap(); - directory - .create( - advertisement_for(second_claimant, &key, 1, NOW_MS + 39_000), - NOW_MS + 39_000, - ) - .await - .unwrap(); - - assert!( - directory - .claim_expired(leader, second_claimant, NOW_MS + 39_999) - .await - .is_err() - ); - let second = directory - .claim_expired(leader, second_claimant, NOW_MS + 40_000) - .await - .unwrap(); - - assert_eq!(second.claimant(), second_claimant); - assert_eq!(second.claim_generation(), first.claim_generation() + 1); - assert_eq!( - second.log().unwrap().recovery().unwrap().generation(), - second.claim_generation() - ); - assert!( - directory - .refresh_recovery_claim(&first, NOW_MS + 40_001) - .await - .is_err() - ); -} diff --git a/crates/crab-cell-runtime/src/node/tests/log.rs b/crates/crab-cell-runtime/src/node/tests/log.rs deleted file mode 100644 index 1776e3741..000000000 --- a/crates/crab-cell-runtime/src/node/tests/log.rs +++ /dev/null @@ -1,345 +0,0 @@ -//! Node-log enrollment, authority, and takeover. - -use super::*; - -#[tokio::test] -async fn node_log_enrollment_activation_and_coverage_are_authoritative() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let leader = SessionId::from_bytes([1; 16]); - let first = SessionId::from_bytes([2; 16]); - let second = SessionId::from_bytes([3; 16]); - let created = directory - .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - for member in [first, second] { - directory - .create(advertisement_for(member, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - } - - let enrolled = directory - .recruit_log(&created, 4, 1, 3, NOW_MS + 1) - .await - .unwrap(); - let log = enrolled.advertisement().log().unwrap(); - assert_eq!(enrolled.advertisement().generation(), 2); - assert_eq!(log.phase(), NodeLogPhase::Open); - assert!(!log.active()); - assert_eq!(log.members(), [node(first), node(second)]); - directory - .authorize_log_append(leader, node(first), 4, 0, NOW_MS + 2) - .await - .unwrap(); - assert!( - directory - .authorize_log_append(leader, NodeId::from_bytes([8; 16]), 4, 0, NOW_MS + 2) - .await - .is_err() - ); - - let refreshed = directory - .refresh( - &created, - advertisement_for(leader, &key, 2, NOW_MS + 1_000), - NOW_MS + 1_000, - ) - .await - .unwrap(); - assert_eq!(refreshed.advertisement().generation(), 3); - assert_eq!(refreshed.advertisement().log(), Some(log)); - let active = directory - .activate_log(&refreshed, NOW_MS + 1_001) - .await - .unwrap(); - assert!(active.advertisement().log().unwrap().active()); - let covered = directory - .advance_log_coverage(&active, 27, NOW_MS + 1_002) - .await - .unwrap(); - assert!(directory.log_epoch_referenced(leader, 4).await.unwrap()); - assert_eq!(covered.advertisement().log().unwrap().tiered_through(), 27); - // An append may have been queued before coverage advanced, but its wire - // watermark must never authorize deletion beyond the persisted prefix. - for watermark in [0, 26, 27] { - directory - .authorize_log_append(leader, node(first), 4, watermark, NOW_MS + 1_003) - .await - .unwrap(); - } - assert!(matches!( - directory - .authorize_log_append(leader, node(first), 4, 28, NOW_MS + 1_003) - .await, - Err(Error::PeerAuthorization(_)) - )); - directory - .authorize_log_retire(leader, node(first), 4, 27, NOW_MS + 1_003) - .await - .unwrap(); - let gate = - crate::node::log::DurabilityGate::new(leader, node(leader), 4, [node(first), node(second)]) - .unwrap(); - let ticket = gate.issue(27).unwrap(); - gate.prove_object(ticket).unwrap(); - let rotated = crate::node::log::rotate_node_log( - &directory, - Arc::new(UnavailableFollowerTransport), - &covered, - &gate, - 1, - 3, - NOW_MS + 1_003, - ) - .await - .unwrap(); - let rotated_log = rotated.enrollment.advertisement().log().unwrap(); - assert_eq!(rotated_log.epoch(), 5); - assert_eq!(rotated_log.phase(), NodeLogPhase::Open); - assert!(!rotated_log.active()); - assert_eq!(rotated_log.tiered_through(), 0); - assert_eq!(rotated.gate.issue(1).unwrap().first_sequence(), 1); - assert!( - directory - .authorize_log_retire(leader, node(first), 4, 27, NOW_MS + 1_004) - .await - .is_err() - ); - assert!( - directory - .withdraw(&rotated.enrollment, NOW_MS + 1_005) - .await - .is_err() - ); - assert!( - directory - .withdraw_after_drain(&created, NOW_MS + 1_005) - .await - .is_err() - ); -} - -#[tokio::test] -async fn clean_node_log_close_clears_authority_before_session_withdrawal() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let leader = SessionId::from_bytes([1; 16]); - let member = SessionId::from_bytes([2; 16]); - let created = directory - .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - directory - .create(advertisement_for(member, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let enrolled = directory - .recruit_log(&created, 4, 1, 2, NOW_MS + 1) - .await - .unwrap(); - let active = directory.activate_log(&enrolled, NOW_MS + 2).await.unwrap(); - let covered = directory - .advance_log_coverage(&active, 2, NOW_MS + 3) - .await - .unwrap(); - let gate = - crate::node::log::DurabilityGate::new(leader, node(leader), 4, [node(member)]).unwrap(); - let ticket = gate.issue(2).unwrap(); - gate.prove_object(ticket).unwrap(); - - let closed = crate::node::log::close_node_log( - &directory, - Arc::new(UnavailableFollowerTransport), - &covered, - &gate, - NOW_MS + 4, - ) - .await - .unwrap(); - - assert!(closed.advertisement().log().is_none()); - assert!(!directory.log_epoch_referenced(leader, 4).await.unwrap()); - assert!( - directory - .authorize_log_append(leader, node(member), 4, 0, NOW_MS + 5) - .await - .is_err() - ); - directory.withdraw(&closed, NOW_MS + 5).await.unwrap(); - assert!(directory.load(leader, NOW_MS + 6).await.unwrap().is_none()); - assert!(!directory.log_epoch_referenced(leader, 4).await.unwrap()); -} - -#[tokio::test] -async fn request_takeover_does_not_claim_active_node_log() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let leader = SessionId::from_bytes([1; 16]); - let member = SessionId::from_bytes([2; 16]); - let claimant = SessionId::from_bytes([3; 16]); - let created = directory - .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let member_record = directory - .create(advertisement_for(member, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let enrolled = directory - .recruit_log(&created, 7, 1, 2, NOW_MS + 1) - .await - .unwrap(); - directory.activate_log(&enrolled, NOW_MS + 2).await.unwrap(); - let claimant_record = directory - .create(advertisement_for(claimant, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - for (record, session) in [(&member_record, member), (&claimant_record, claimant)] { - directory - .refresh( - record, - advertisement_for(session, &key, 2, NOW_MS + 9_000), - NOW_MS + 9_000, - ) - .await - .unwrap(); - } - - assert!(matches!( - directory - .claim_expired_for_takeover(leader, claimant, NOW_MS + 10_000) - .await, - Err(Error::PendingPublication) - )); - assert_eq!( - directory - .recovery_candidates(claimant, NOW_MS + 10_000, 2) - .await - .unwrap(), - [leader] - ); - assert!( - directory - .takeover_proof(leader, claimant, NOW_MS + 10_000) - .await - .unwrap() - .is_none() - ); -} - -#[tokio::test] -async fn request_requires_the_live_sessions_mtls_certificate_and_signing_key() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - directory - .create(advertisement(&key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let target = CellTarget::new( - TenantId::from_bytes([8; 16]), - ApplicationId::from_bytes([9; 16]), - NamespaceId::from_bytes([10; 16]), - b"partition", - ) - .unwrap(); - let signer = PeerSigner::new( - SessionId::from_bytes([1; 16]), - Digest::from_bytes([5; 32]), - key, - ); - let certificate_key = signer.verifying_key().to_bytes(); - let request = signer - .sign( - PeerPrincipal { - issuer: "https://identity.example".into(), - subject: "user".into(), - actions: vec!["repository.read".into()], - }, - NOW_MS, - NOW_MS + 20_000, - 5_000, - PeerOperation::Read(peer_wire::ReadRequest { - expected: None, - target: Some(peer_wire::Target { - tenant_id: target.tenant().as_bytes().to_vec(), - application_id: target.application().as_bytes().to_vec(), - namespace_id: target.namespace().as_bytes().to_vec(), - partition: target.partition().to_vec(), - }), - timeout_ms: 5_000, - minimum: None, - operation: Some(peer_wire::read_request::Operation::Describe(true)), - }), - ) - .unwrap(); - - assert!( - directory - .peer_verifier( - SessionId::from_bytes([1; 16]), - Digest::from_bytes([3; 32]), - certificate_key, - NOW_MS + 1, - ) - .await - .unwrap() - .verify( - crate::peer::UnverifiedPeerRequest::decode(&request).unwrap(), - NOW_MS + 1 - ) - .is_ok() - ); - assert!(matches!( - directory - .peer_verifier( - SessionId::from_bytes([1; 16]), - Digest::from_bytes([11; 32]), - certificate_key, - NOW_MS + 1, - ) - .await, - Err(Error::PeerAuthorization(_)) - )); - assert!(matches!( - directory - .peer_verifier( - SessionId::from_bytes([1; 16]), - Digest::from_bytes([3; 32]), - [12; 32], - NOW_MS + 1, - ) - .await, - Err(Error::PeerAuthorization(_)) - )); - let enrolled = directory - .peer_verifier( - SessionId::from_bytes([1; 16]), - Digest::from_bytes([3; 32]), - certificate_key, - NOW_MS + 1, - ) - .await - .unwrap(); - let later = NOW_MS + 10_001; - // The request remains valid after the enrolled node lease expires. A CPU - // queue must not turn the earlier enrollment observation into fresh proof. - assert!( - crate::peer::PeerVerifier::new( - SessionId::from_bytes([1; 16]), - Digest::from_bytes([5; 32]), - signer.verifying_key(), - ) - .verify(&request, later) - .is_ok() - ); - assert!( - enrolled - .verify( - crate::peer::UnverifiedPeerRequest::decode(&request).unwrap(), - later - ) - .is_err() - ); -} diff --git a/crates/crab-cell-runtime/src/node/tests/placement.rs b/crates/crab-cell-runtime/src/node/tests/placement.rs deleted file mode 100644 index 7c1deabd7..000000000 --- a/crates/crab-cell-runtime/src/node/tests/placement.rs +++ /dev/null @@ -1,349 +0,0 @@ -//! Signed-capacity placement and follower selection. - -use super::*; - -#[tokio::test] -async fn placement_uses_signed_capacity_only() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let session = SessionId::from_bytes([1; 16]); - directory - .create( - advertisement_for(session, &key, 1, NOW_MS) - .with_placement_capacity( - NodePlacementCapacity { - memory_capacity_bytes: 2_000, - disk_capacity_bytes: 4_000, - active_cells: 1, - max_active_cells: 8, - running_jobs: 0, - job_capacity: 3, - publication_backlog: 0, - hydration_backlog: 0, - primitive_backlog: 0, - } - .validated() - .unwrap(), - &key, - ) - .unwrap(), - NOW_MS, - ) - .await - .unwrap(); - let planner = PlacementPlanner::default(); - let cell = crate::CellId::from_bytes([9; 32]); - let advertised = directory - .choose_advertised_placement(&planner, cell, NOW_MS + 1, 4) - .await - .unwrap() - .unwrap(); - assert_eq!(advertised.node, node(session)); -} - -#[tokio::test] -async fn cold_placement_does_not_reward_the_requesting_node() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let local = SessionId::from_bytes([1; 16]); - let remote = SessionId::from_bytes([2; 16]); - for (session, free_memory_bytes) in [(local, 100), (remote, 1_000)] { - let advertisement = advertisement_for_capacity( - session, - &key, - 1, - NOW_MS, - NodeCapacity { - free_memory_bytes, - free_disk_bytes: 1_000, - job_credits: 4, - ..NodeCapacity::default() - }, - ) - .with_placement_capacity( - NodePlacementCapacity { - memory_capacity_bytes: 1_000, - disk_capacity_bytes: 1_000, - active_cells: 0, - max_active_cells: 10, - running_jobs: 0, - job_capacity: 4, - publication_backlog: 0, - hydration_backlog: 0, - primitive_backlog: 0, - } - .validated() - .unwrap(), - &key, - ) - .unwrap(); - directory.create(advertisement, NOW_MS).await.unwrap(); - } - let chosen = directory - .choose_advertised_placement( - &PlacementPlanner::default(), - crate::CellId::from_bytes([9; 32]), - NOW_MS + 1, - 4, - ) - .await - .unwrap() - .unwrap(); - assert_eq!(chosen.session, remote); -} - -#[tokio::test] -async fn idle_placement_prefers_peer_with_more_active_cell_headroom() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let requester = SessionId::from_bytes([1; 16]); - let peer = SessionId::from_bytes([2; 16]); - for (session, active_cells) in [(requester, 4), (peer, 3)] { - let placement = NodePlacementCapacity { - memory_capacity_bytes: 1_000, - disk_capacity_bytes: 2_000, - active_cells, - max_active_cells: 10, - running_jobs: 0, - job_capacity: 3, - publication_backlog: 0, - hydration_backlog: 0, - primitive_backlog: 0, - } - .validated() - .unwrap(); - let advertisement = advertisement_for(session, &key, 1, NOW_MS) - .with_placement_capacity(placement, &key) - .unwrap(); - directory.create(advertisement, NOW_MS).await.unwrap(); - } - - let chosen = directory - .choose_advertised_placement( - &PlacementPlanner::default(), - crate::CellId::from_bytes([9; 32]), - NOW_MS + 1, - 4, - ) - .await - .unwrap() - .unwrap(); - assert_eq!(chosen.session, peer); -} - -#[tokio::test] -async fn follower_selection_is_capacity_aware_deterministic_and_requires_full_shape() { - let key = SigningKey::from_bytes(&[7; 32]); - let leader = SessionId::from_bytes([1; 16]); - let first = SessionId::from_bytes([2; 16]); - let second = SessionId::from_bytes([3; 16]); - let pressured = SessionId::from_bytes([4; 16]); - let directory = directory(); - for (session, follower_free_bytes) in [ - (leader, 2_000), - (first, 2_000), - (second, 1_000), - (pressured, 9), - ] { - directory - .create( - advertisement_for_capacity( - session, - &key, - 1, - NOW_MS, - NodeCapacity { - free_memory_bytes: 1_000, - free_disk_bytes: 2_000, - follower_free_bytes, - follower_retained_bytes: 0, - job_credits: 3, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - }, - ), - NOW_MS, - ) - .await - .unwrap(); - } - - assert_eq!( - directory - .select_log_members(leader, 1_000, NOW_MS + 1, 4) - .await - .unwrap(), - [node(first), node(second)] - ); - assert!( - directory - .select_log_members(leader, 1_001, NOW_MS + 1, 4) - .await - .unwrap() - .is_empty() - ); -} - -#[tokio::test] -async fn follower_selection_prefers_proven_zone_then_host_separation() { - let key = SigningKey::from_bytes(&[7; 32]); - let leader = SessionId::from_bytes([1; 16]); - let same_zone = SessionId::from_bytes([2; 16]); - let remote_zone = SessionId::from_bytes([3; 16]); - let third_zone_same_host = SessionId::from_bytes([4; 16]); - let directory = directory(); - let capacity = NodeCapacity { - free_memory_bytes: 1_000, - free_disk_bytes: 2_000, - follower_free_bytes: 2_000, - follower_retained_bytes: 0, - job_credits: 3, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - }; - for (session, zone, host) in [ - (leader, "zone-a", "host-a"), - (same_zone, "zone-a", "host-b"), - (remote_zone, "zone-b", "host-c"), - (third_zone_same_host, "zone-c", "host-a"), - ] { - directory - .create( - advertisement_for_node_capacity_in_domain( - node(session), - session, - &key, - 1, - NOW_MS, - NodeFailureDomain::new(Some(zone.into()), Some(host.into())).unwrap(), - capacity, - ), - NOW_MS, - ) - .await - .unwrap(); - } - - assert_eq!( - directory - .select_log_members(leader, 1_000, NOW_MS + 1, 4) - .await - .unwrap(), - [node(remote_zone), node(third_zone_same_host)] - ); -} - -#[test] -fn failure_domain_rejects_unknown_or_ambiguous_labels() { - for invalid in [ - String::new(), - " zone-a".into(), - "zone a".into(), - "zone-a\n".into(), - "a".repeat(254), - ] { - assert!(NodeFailureDomain::new(Some(invalid), None).is_err()); - } - assert!(NodeFailureDomain::new(None, None).is_ok()); -} - -#[test] -fn placement_schema_is_mixed_version_safe_and_fail_closed() { - let key = SigningKey::from_bytes(&[7; 32]); - let current = advertisement(&key, 1, NOW_MS) - .with_placement_capacity( - NodePlacementCapacity { - memory_capacity_bytes: 8_192, - disk_capacity_bytes: 16_384, - active_cells: 3, - max_active_cells: 16, - running_jobs: 2, - job_capacity: 8, - publication_backlog: 4, - hydration_backlog: 5, - primitive_backlog: 6, - } - .validated() - .unwrap(), - &key, - ) - .unwrap(); - assert!(current.has_signed_placement()); - let decoded_current = NodeAdvertisement::decode_canonical(¤t.encode().unwrap()).unwrap(); - assert_eq!(decoded_current.placement_version, 2); - assert_eq!( - decoded_current.placement_capacity(), - current.placement_capacity() - ); - let observation = - PlacementObservation::from_signed_advertisement(&decoded_current, NOW_MS + 1, true) - .unwrap(); - assert_eq!(observation.memory_capacity_bytes, 8_192); - assert_eq!(observation.active_cells, 3); - assert_eq!(observation.running_jobs, 2); - assert_eq!(observation.publication_backlog, 4); - assert_eq!(observation.hydration_backlog, 5); - assert_eq!(observation.primitive_backlog, 6); - - let mut legacy = current.clone(); - legacy.placement_version = 0; - legacy.placement_signature = [0; 64]; - let decoded_legacy = NodeAdvertisement::decode_canonical(&legacy.encode().unwrap()).unwrap(); - assert!(!decoded_legacy.has_signed_placement()); - - let mut previous = current.clone(); - previous.placement_version = 1; - let decoded_previous = - NodeAdvertisement::decode_canonical(&previous.encode().unwrap()).unwrap(); - assert!(!decoded_previous.has_signed_placement()); - assert_eq!( - decoded_previous - .placement_capacity() - .unwrap() - .publication_backlog, - 0 - ); - - let mut future = current; - future.placement_version = 3; - future.placement_signature = [0; 64]; - let decoded_future = NodeAdvertisement::decode_canonical(&future.encode().unwrap()).unwrap(); - assert!(!decoded_future.has_signed_placement()); - assert!( - PlacementObservation::from_signed_advertisement(&decoded_future, NOW_MS + 1, false) - .is_err() - ); -} - -#[test] -fn placement_upgrade_sets_schema_when_legacy_capacity_was_unusable() { - let key = SigningKey::from_bytes(&[7; 32]); - let legacy = advertisement_for_capacity( - SessionId::from_bytes([1; 16]), - &key, - 1, - NOW_MS, - NodeCapacity::default(), - ); - assert!(!legacy.has_signed_placement()); - let upgraded = legacy - .with_placement_capacity( - NodePlacementCapacity { - memory_capacity_bytes: 8_192, - disk_capacity_bytes: 16_384, - active_cells: 0, - max_active_cells: 16, - running_jobs: 0, - job_capacity: 8, - publication_backlog: 0, - hydration_backlog: 0, - primitive_backlog: 0, - } - .validated() - .unwrap(), - &key, - ) - .unwrap(); - assert!(upgraded.has_signed_placement()); - let decoded = NodeAdvertisement::decode_canonical(&upgraded.encode().unwrap()).unwrap(); - assert_eq!(decoded.placement_version, 2); - assert_eq!(decoded.placement_capacity(), upgraded.placement_capacity()); -} diff --git a/crates/crab-cell-runtime/src/node/tests/records.rs b/crates/crab-cell-runtime/src/node/tests/records.rs deleted file mode 100644 index f9bdc43b4..000000000 --- a/crates/crab-cell-runtime/src/node/tests/records.rs +++ /dev/null @@ -1,187 +0,0 @@ -//! Advertisement records: create, load, refresh, and validation. - -use super::*; - -#[tokio::test] -async fn create_load_and_refresh_preserve_signed_boot_identity() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let created = directory - .create(advertisement(&key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let adopted = directory - .create(advertisement(&key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - assert_eq!(adopted.advertisement(), created.advertisement()); - let loaded = directory - .load(SessionId::from_bytes([1; 16]), NOW_MS + 1) - .await - .unwrap() - .unwrap(); - assert_eq!(loaded.advertisement(), created.advertisement()); - assert!( - directory - .is_live(SessionId::from_bytes([1; 16]), NOW_MS + 1) - .await - .unwrap() - ); - assert!( - !directory - .is_live(SessionId::from_bytes([9; 16]), NOW_MS + 1) - .await - .unwrap() - ); - - let refreshed = directory - .refresh( - &created, - advertisement(&key, 1, NOW_MS + 1_000), - NOW_MS + 1_000, - ) - .await - .unwrap(); - assert_eq!(refreshed.advertisement().progress(), 1); - assert_eq!( - refreshed.advertisement().signature, - created.advertisement().signature - ); - let refreshed = directory - .refresh( - &refreshed, - advertisement(&key, 2, NOW_MS + 2_000), - NOW_MS + 2_000, - ) - .await - .unwrap(); - assert_eq!(refreshed.advertisement().progress(), 2); -} - -#[tokio::test] -async fn invalid_signature_expiry_and_identity_change_fail_closed() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - assert!( - NodeAdvertisement::sign( - NodeId::from_bytes([1; 16]), - SessionId::from_bytes([1; 16]), - "https:///not-an-authority".into(), - Digest::from_bytes([2; 32]), - Digest::from_bytes([3; 32]), - Digest::from_bytes([4; 32]), - Digest::from_bytes([5; 32]), - &key, - 1, - NOW_MS, - NOW_MS + 10_000, - vec![Digest::from_bytes([6; 32])], - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 1, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .is_err() - ); - let original = advertisement(&key, 1, NOW_MS); - let mut tampered_capacity = original.clone(); - tampered_capacity.capacity.free_memory_bytes = tampered_capacity - .capacity - .free_memory_bytes - .saturating_add(1); - // Capacity is part of the signed heartbeat; changing it without a new - // signature must fail closed. The optional placement block is authenticated - // independently because it drives ownership placement. - assert!(tampered_capacity.verify_signature().is_err()); - let signed = original - .clone() - .with_placement_capacity( - NodePlacementCapacity { - memory_capacity_bytes: 8_192, - disk_capacity_bytes: 16_384, - active_cells: 1, - max_active_cells: 8, - running_jobs: 1, - job_capacity: 4, - publication_backlog: 0, - hydration_backlog: 0, - primitive_backlog: 0, - } - .validated() - .unwrap(), - &key, - ) - .unwrap(); - let mut tampered_placement = signed; - tampered_placement - .placement - .as_mut() - .expect("signed placement is present") - .memory_capacity_bytes = 8_193; - assert!(tampered_placement.verify_signature().is_err()); - let mut tampered = original.encode().unwrap(); - let endpoint_byte = tampered - .windows(b"node-1".len()) - .position(|window| window == b"node-1") - .unwrap() - + b"node-".len(); - tampered[endpoint_byte] = b'2'; - assert!(NodeAdvertisement::decode_canonical(&tampered).is_err()); - - let created = directory.create(original, NOW_MS).await.unwrap(); - assert!( - directory - .load(SessionId::from_bytes([1; 16]), NOW_MS + 10_000) - .await - .is_err() - ); - assert!( - !directory - .is_live(SessionId::from_bytes([1; 16]), NOW_MS + 10_000) - .await - .unwrap() - ); - - let other_key = SigningKey::from_bytes(&[8; 32]); - assert!( - directory - .refresh( - &created, - advertisement(&other_key, 2, NOW_MS + 1_000), - NOW_MS + 1_000, - ) - .await - .is_err() - ); -} - -#[tokio::test] -async fn refresh_accepts_new_signed_capacity_for_same_boot_session() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let session = SessionId::from_bytes([1; 16]); - let created = directory - .create(advertisement_for(session, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let next_capacity = NodeCapacity { - free_memory_bytes: 900, - free_disk_bytes: 1_800, - follower_free_bytes: 1_700, - follower_retained_bytes: 700, - job_credits: 2, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - }; - let next = advertisement_for_capacity(session, &key, 2, NOW_MS + 1_000, next_capacity); - assert_ne!(next.signature, created.advertisement().signature); - let refreshed = directory - .refresh(&created, next, NOW_MS + 1_000) - .await - .unwrap(); - assert_eq!(refreshed.advertisement().capacity(), next_capacity); - assert!(refreshed.advertisement().verify_signature().is_ok()); -} diff --git a/crates/crab-cell-runtime/src/node/tests/sessions.rs b/crates/crab-cell-runtime/src/node/tests/sessions.rs deleted file mode 100644 index 6c992c998..000000000 --- a/crates/crab-cell-runtime/src/node/tests/sessions.rs +++ /dev/null @@ -1,505 +0,0 @@ -//! Session listing, claims, collection, and withdrawal. - -use super::*; - -#[tokio::test] -async fn stable_follower_node_resolves_a_new_session_for_old_log_recovery() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let leader = SessionId::from_bytes([1; 16]); - let old_follower = SessionId::from_bytes([2; 16]); - let restarted_follower = SessionId::from_bytes([3; 16]); - let claimant = SessionId::from_bytes([4; 16]); - let follower_node = NodeId::from_bytes([9; 16]); - let capacity = NodeCapacity { - free_memory_bytes: 1_000, - free_disk_bytes: 2_000, - follower_free_bytes: 2_000, - follower_retained_bytes: 0, - job_credits: 3, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - }; - let leader_record = directory - .create(advertisement_for(leader, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - directory - .create( - advertisement_for_node_capacity(follower_node, old_follower, &key, 1, NOW_MS, capacity), - NOW_MS, - ) - .await - .unwrap(); - let enrolled = directory - .recruit_log(&leader_record, 7, 1, 2, NOW_MS + 1) - .await - .unwrap(); - directory.activate_log(&enrolled, NOW_MS + 2).await.unwrap(); - - directory - .create( - advertisement_for_node_capacity( - follower_node, - restarted_follower, - &key, - 1, - NOW_MS + 10_000, - capacity, - ), - NOW_MS + 10_000, - ) - .await - .unwrap(); - directory - .create( - advertisement_for(claimant, &key, 1, NOW_MS + 10_000), - NOW_MS + 10_000, - ) - .await - .unwrap(); - let fenced = directory - .claim_expired(leader, claimant, NOW_MS + 10_000) - .await - .unwrap(); - - assert_eq!(fenced.log().unwrap().members(), [follower_node]); - assert_eq!( - directory - .resolve_node(follower_node, NOW_MS + 10_001) - .await - .unwrap() - .unwrap() - .session(), - restarted_follower - ); - directory - .authorize_log_recovery(leader, claimant, follower_node, 7, NOW_MS + 10_001) - .await - .unwrap(); -} - -#[tokio::test] -async fn operational_inspection_preserves_expired_lease_evidence_without_reviving_it() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let session = SessionId::from_bytes([8; 16]); - directory - .create(advertisement_for(session, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - - let expired_at = NOW_MS + 10_000; - assert!(!directory.is_live(session, expired_at).await.unwrap()); - let inspected = directory - .inspect_advertisement(session, expired_at) - .await - .unwrap() - .unwrap(); - assert_eq!(inspected.session(), session); - assert_eq!(inspected.expires_at_ms(), expired_at); - assert!(directory.load(session, expired_at).await.is_err()); -} - -#[tokio::test] -async fn overlapping_live_sessions_for_one_node_fail_closed() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let stable = NodeId::from_bytes([8; 16]); - let capacity = NodeCapacity { - free_memory_bytes: 1_000, - free_disk_bytes: 2_000, - follower_free_bytes: 2_000, - follower_retained_bytes: 0, - job_credits: 3, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - }; - for session in [ - SessionId::from_bytes([1; 16]), - SessionId::from_bytes([2; 16]), - ] { - directory - .create( - advertisement_for_node_capacity(stable, session, &key, 1, NOW_MS, capacity), - NOW_MS, - ) - .await - .unwrap(); - } - - assert!(matches!( - directory.live(NOW_MS + 1, 2).await, - Err(Error::Node(_)) - )); - assert!(matches!( - directory.resolve_node(stable, NOW_MS + 1).await, - Err(Error::Node(_)) - )); -} - -#[tokio::test] -async fn live_listing_is_sorted_bounded_and_ignores_expired_sessions() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - for session in [ - SessionId::from_bytes([8; 16]), - SessionId::from_bytes([1; 16]), - ] { - directory - .create(advertisement_for(session, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - } - - let live = directory.live(NOW_MS + 1, 2).await.unwrap(); - assert_eq!( - live.iter() - .map(NodeAdvertisement::session) - .collect::>(), - [ - SessionId::from_bytes([1; 16]), - SessionId::from_bytes([8; 16]) - ] - ); - assert!(matches!( - directory.live(NOW_MS + 1, 1).await, - Err(Error::Node(_)) - )); - assert!(directory.live(NOW_MS + 10_000, 1).await.unwrap().is_empty()); -} - -#[tokio::test] -async fn maintenance_inventory_retains_expired_session_until_withdrawal() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let session = SessionId::from_bytes([1; 16]); - let observed = directory - .create(advertisement(&key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - assert!(directory.live(NOW_MS + 10_000, 1).await.unwrap().is_empty()); - assert_eq!( - directory - .advertised_sessions(NOW_MS + 10_000, 1) - .await - .unwrap(), - [session] - ); - directory - .withdraw(&observed, NOW_MS + 10_001) - .await - .unwrap(); - assert!( - directory - .advertised_sessions(NOW_MS + 10_001, 1) - .await - .unwrap() - .is_empty() - ); -} - -#[tokio::test] -async fn expired_session_claim_is_atomic_idempotent_and_blocks_refresh() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let session = SessionId::from_bytes([1; 16]); - let claimant = SessionId::from_bytes([8; 16]); - let observed = directory - .create(advertisement(&key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - directory - .create( - advertisement_for(claimant, &key, 1, NOW_MS + 9_000), - NOW_MS + 9_000, - ) - .await - .unwrap(); - assert!( - directory - .claim_expired(session, claimant, NOW_MS + 9_999) - .await - .is_err() - ); - - let fenced = directory - .claim_expired(session, claimant, NOW_MS + 10_000) - .await - .unwrap(); - assert_eq!(fenced.session(), session); - assert_eq!( - directory - .claim_expired(session, claimant, NOW_MS + 10_001) - .await - .unwrap(), - fenced - ); - assert!( - directory - .claim_expired(session, SessionId::from_bytes([9; 16]), NOW_MS + 10_001,) - .await - .is_err() - ); - assert!( - directory - .refresh( - &observed, - advertisement(&key, 2, NOW_MS + 10_001), - NOW_MS + 10_001, - ) - .await - .is_err() - ); - assert!( - directory - .withdraw(&observed, NOW_MS + 10_001) - .await - .is_err() - ); - assert!(matches!( - directory - .withdraw_after_drain(&observed, NOW_MS + 10_001) - .await, - Err(Error::Fenced) - )); - assert_eq!( - directory - .claim_expired(session, claimant, NOW_MS + 10_002) - .await - .unwrap(), - fenced - ); -} - -#[tokio::test] -async fn stale_collection_fences_records_after_the_clock_skew_horizon() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let stale = SessionId::from_bytes([1; 16]); - let current = SessionId::from_bytes([8; 16]); - directory - .create(advertisement_for(stale, &key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let collection_ms = NOW_MS + 10_000 + STALE_ADVERTISEMENT_RETENTION_MS; - directory - .create( - advertisement_for(current, &key, 1, collection_ms), - collection_ms, - ) - .await - .unwrap(); - - assert_eq!( - directory.collect_stale(collection_ms - 1, 1).await.unwrap(), - 0 - ); - assert_eq!(directory.collect_stale(collection_ms, 1).await.unwrap(), 1); - assert!(directory.load_canonical(stale).await.unwrap().is_none()); - assert!( - !directory - .advertised_sessions(collection_ms, 2) - .await - .unwrap() - .contains(&stale) - ); - assert!(directory.is_live(current, collection_ms + 1).await.unwrap()); -} - -#[tokio::test] -async fn stale_collection_preserves_a_session_refreshed_before_fencing() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let created = directory - .create(advertisement(&key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let collection_ms = NOW_MS + 10_000 + STALE_ADVERTISEMENT_RETENTION_MS; - directory - .refresh( - &created, - advertisement(&key, 2, collection_ms), - collection_ms, - ) - .await - .unwrap(); - - assert_eq!(directory.collect_stale(collection_ms, 1).await.unwrap(), 0); - assert!( - directory - .is_live(SessionId::from_bytes([1; 16]), collection_ms + 1) - .await - .unwrap() - ); -} - -#[tokio::test] -async fn drained_withdrawal_reconciles_an_unobserved_heartbeat() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let session = SessionId::from_bytes([1; 16]); - let created = directory - .create(advertisement(&key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let refreshed = directory - .refresh(&created, advertisement(&key, 2, NOW_MS + 1), NOW_MS + 1) - .await - .unwrap(); - directory - .withdraw_after_drain(&created, NOW_MS + 2) - .await - .unwrap(); - assert!(directory.is_retired(session).await.unwrap()); - assert!( - directory - .refresh(&refreshed, advertisement(&key, 3, NOW_MS + 3), NOW_MS + 3) - .await - .is_err() - ); -} - -#[tokio::test] -async fn drained_withdrawal_rejects_a_changed_boot_identity() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let created = directory - .create(advertisement(&key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - let replacement_key = SigningKey::from_bytes(&[8; 32]); - let mut changed = advertisement(&replacement_key, 2, NOW_MS + 1); - changed.generation = 2; - let path = directory.layout.node_path(changed.session.as_bytes()); - directory - .layout - .store() - .update( - &path, - Bytes::from(changed.encode().unwrap()), - created.token.clone(), - ) - .await - .unwrap(); - assert!( - directory - .withdraw_after_drain(&created, NOW_MS + 2) - .await - .is_err() - ); - assert_eq!( - directory - .load_canonical(changed.session) - .await - .unwrap() - .unwrap() - .0, - changed - ); -} - -#[tokio::test] -async fn withdrawal_removes_only_the_exact_observed_advertisement() { - let key = SigningKey::from_bytes(&[7; 32]); - let directory = directory(); - let session = SessionId::from_bytes([1; 16]); - let created = directory - .create(advertisement(&key, 1, NOW_MS), NOW_MS) - .await - .unwrap(); - directory.withdraw(&created, NOW_MS + 1).await.unwrap(); - assert!(directory.load_canonical(session).await.unwrap().is_none()); - - let replacement = SessionId::from_bytes([2; 16]); - let created = directory - .create( - advertisement_for(replacement, &key, 1, NOW_MS + 20_000), - NOW_MS + 20_000, - ) - .await - .unwrap(); - let refreshed = directory - .refresh( - &created, - advertisement_for(replacement, &key, 2, NOW_MS + 21_000), - NOW_MS + 21_000, - ) - .await - .unwrap(); - assert!(directory.withdraw(&created, NOW_MS + 21_001).await.is_err()); - assert_eq!( - directory - .load(replacement, NOW_MS + 21_001) - .await - .unwrap() - .unwrap() - .advertisement(), - refreshed.advertisement() - ); -} - -#[tokio::test] -async fn live_listing_rejects_misplaced_or_foreign_active_records() { - let key = SigningKey::from_bytes(&[7; 32]); - let subject = directory(); - let advertisement = advertisement(&key, 1, NOW_MS); - subject - .layout - .store() - .put_overwrite( - &subject - .layout - .node_directory_path() - .join("ffffffffffffffffffffffffffffffff.json"), - advertisement.encode().unwrap().into(), - ) - .await - .unwrap(); - assert!(matches!( - subject.live(NOW_MS + 1, 2).await, - Err(Error::Node(_)) - )); - - let foreign = NodeAdvertisement::sign( - NodeId::from_bytes([9; 16]), - SessionId::from_bytes([9; 16]), - "https://node-1.internal:8789".into(), - Digest::from_bytes([10; 32]), - Digest::from_bytes([3; 32]), - Digest::from_bytes([4; 32]), - Digest::from_bytes([5; 32]), - &key, - 1, - NOW_MS, - NOW_MS + 10_000, - vec![Digest::from_bytes([6; 32])], - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 1, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .unwrap(); - let clean = directory(); - clean - .layout - .store() - .put_overwrite( - &clean.layout.node_path(foreign.session().as_bytes()), - foreign.encode().unwrap().into(), - ) - .await - .unwrap(); - assert!( - clean - .is_live(SessionId::from_bytes([9; 16]), NOW_MS + 1) - .await - .is_err() - ); - assert!(matches!( - clean.live(NOW_MS + 1, 2).await, - Err(Error::Node(_)) - )); -} diff --git a/crates/crab-cell-runtime/src/peer.rs b/crates/crab-cell-runtime/src/peer.rs deleted file mode 100644 index 365aeddcb..000000000 --- a/crates/crab-cell-runtime/src/peer.rs +++ /dev/null @@ -1,511 +0,0 @@ -//! Peer operation envelopes, signing, and transport authorization. -use ed25519_dalek::{Signature, Signer, SigningKey, VerifyingKey}; -use prost::Message; - -use crate::identity::IncarnationId; -use crate::identity::{ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId}; -use crate::{Error, Result}; - -mod dispatch; -mod protobuf; -mod transport; -mod validation; - -pub use dispatch::{ - PeerAuthorizer, PeerCellResolver, PeerDispatcher, PeerReplicaControl, PeerReplicaResolver, -}; -pub(crate) use transport::PeerClientTransport; -pub use transport::PeerRoundTrip; -pub use transport::{EffectPeerClient, MigrationPeerClient, ReplicaPeerClient}; - -use protobuf::{ - MessageKind, field_payload, oneof_payload, require_fields, validate_message, validate_operation, -}; - -use validation::*; - -const PROTOCOL_VERSION: u32 = 1; -const MAX_AUTHORIZATION_BYTES: usize = 16 * 1024; -const MAX_OPERATION_BYTES: usize = crate::codec::MAX_WIRE_BYTES; -/// Largest peer request the runtime accepts: authorization envelope, operation -/// bytes, and framing. -pub const MAX_PEER_REQUEST_BYTES: usize = MAX_AUTHORIZATION_BYTES + MAX_OPERATION_BYTES + 128; -const MAX_ACTIONS: usize = 128; -const MAX_PRINCIPAL_BYTES: usize = 512; -const MAX_AUTH_LIFETIME_MS: i64 = 60_000; -const MAX_CLOCK_SKEW_MS: i64 = 5 * 60_000; -const MAX_MUTATION_LIFETIME_MS: i64 = 24 * 60 * 60_000; -const MAX_EFFECT_LIFETIME_MS: i64 = 7 * 24 * 60 * 60_000; - -/// Generated private peer messages. They are not a public service or -/// application API. -/// -/// The schema is documented once in `docs/contracts/peer.proto`; the generated -/// types and fields deliberately carry no Rust doc comments of their own. -#[allow(missing_docs)] -pub mod wire { - include!(concat!(env!("OUT_DIR"), "/crab.cell.peer.v1.rs")); -} - -fn wire_description(value: crate::client::CellDescription) -> wire::CellDescription { - wire::CellDescription { - cell_id: value.cell.as_bytes().to_vec(), - incarnation: value.incarnation.as_bytes().to_vec(), - code: value.code.as_bytes().to_vec(), - schema: value.schema, - } -} - -/// One peer operation currently executable by the typed Cell client. -#[derive(Clone)] -pub enum PeerOperation { - /// Runs one mutation on the target Cell. - Mutate(wire::MutationRequest), - /// Runs one read on the target Cell. - Read(wire::ReadRequest), - /// Resolves one request's outcome. - Resolve(wire::ResolveRequest), - /// Delivers one effect to the target Cell. - DeliverEffect(wire::EffectRequest), - /// Resolves one effect's outcome. - ResolveEffect(wire::EffectResolveRequest), - /// Moves one Cell to this session. - Migrate(wire::MigrationRequest), -} - -impl PeerOperation { - fn tag(&self) -> u32 { - match self { - Self::Mutate(_) => 10, - Self::Read(_) => 11, - Self::Resolve(_) => 12, - Self::DeliverEffect(_) => 13, - Self::ResolveEffect(_) => 14, - Self::Migrate(_) => 15, - } - } - - fn encode(&self) -> Vec { - match self { - Self::Mutate(value) => value.encode_to_vec(), - Self::Read(value) => value.encode_to_vec(), - Self::Resolve(value) => value.encode_to_vec(), - Self::DeliverEffect(value) => value.encode_to_vec(), - Self::ResolveEffect(value) => value.encode_to_vec(), - Self::Migrate(value) => value.encode_to_vec(), - } - } - - fn validate(&self, now_ms: i64) -> Result<()> { - match self { - Self::Mutate(value) => validate_mutation(value, now_ms), - Self::Read(value) => validate_read(value), - Self::Resolve(value) => validate_resolve(value, now_ms), - Self::DeliverEffect(value) => validate_effect(value, now_ms), - Self::ResolveEffect(value) => validate_effect_resolve(value, now_ms), - Self::Migrate(value) => validate_migration(value), - } - } -} - -/// Original authorized principal delegated across one private peer hop. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct PeerPrincipal { - /// Issuer that vouched for the principal. - pub issuer: String, - /// Authenticated principal name. - pub subject: String, - /// Actions the principal may perform. - pub actions: Vec, -} - -/// Boot-session signer used only after node enrollment has bound its public key. -pub struct PeerSigner { - session: SessionId, - release: Digest, - key: SigningKey, -} - -impl PeerSigner { - /// Creates a signer for one session and release. - #[must_use] - pub fn new(session: SessionId, release: Digest, key: SigningKey) -> Self { - Self { - session, - release, - key, - } - } - - /// Returns the key peers use to verify this signer. - #[must_use] - pub fn verifying_key(&self) -> VerifyingKey { - self.key.verifying_key() - } - - /// Encodes and signs one bounded private request without changing its operation bytes. - pub fn sign( - &self, - principal: PeerPrincipal, - issued_at_ms: i64, - expires_at_ms: i64, - remaining_ms: u32, - operation: PeerOperation, - ) -> Result> { - validate_principal(&principal)?; - validate_time_bounds(issued_at_ms, expires_at_ms, issued_at_ms, remaining_ms)?; - operation.validate(issued_at_ms)?; - let tag = operation.tag(); - let payload = operation.encode(); - if payload.len() > MAX_OPERATION_BYTES { - return Err(Error::Peer("operation exceeds wire limit")); - } - validate_operation(tag, &payload)?; - let payload_digest = blake3::hash(&payload); - let mut authorization = wire::PeerAuthorization { - origin_session: self.session.as_bytes().to_vec(), - principal_issuer: principal.issuer, - principal_subject: principal.subject, - actions: principal.actions, - release_digest: self.release.as_bytes().to_vec(), - issued_at_ms, - expires_at_ms, - payload_digest: payload_digest.as_bytes().to_vec(), - signature: Vec::new(), - }; - let signing_bytes = signing_bytes(tag, &authorization)?; - authorization.signature = self.key.sign(&signing_bytes).to_bytes().to_vec(); - encode_request(authorization, 1, remaining_ms, tag, &payload) - } -} - -/// Enrollment-bound verifier for one currently advertised node session. -pub struct PeerVerifier { - session: SessionId, - release: Digest, - key: VerifyingKey, -} - -impl PeerVerifier { - #[must_use] - /// Creates a verifier for one session and release. - pub fn new(session: SessionId, release: Digest, key: VerifyingKey) -> Self { - Self { - session, - release, - key, - } - } - - /// Strictly decodes and authenticates one request before actor admission. - pub fn verify(&self, input: &[u8], now_ms: i64) -> Result { - self.verify_decoded(UnverifiedPeerRequest::decode(input)?, now_ms) - } - - /// Authenticates a structurally checked request without decoding its payload again. - pub fn verify_decoded( - &self, - decoded: UnverifiedPeerRequest, - now_ms: i64, - ) -> Result { - let UnverifiedPeerRequest { - request, - operation_tag, - operation_bytes, - .. - } = decoded; - validate_decoded_operation(request.operation.as_ref(), now_ms)?; - let authorization = request - .authorization - .as_ref() - .ok_or(Error::Peer("authorization is missing"))?; - validate_authorization(authorization, now_ms, request.remaining_ms)?; - if authorization.origin_session.as_slice() != self.session.as_bytes() - || authorization.release_digest.as_slice() != self.release.as_bytes() - { - return Err(Error::Peer("peer enrollment or release does not match")); - } - let expected_digest = blake3::hash(&operation_bytes); - if authorization.payload_digest.as_slice() != expected_digest.as_bytes() { - return Err(Error::Peer("peer payload digest does not match")); - } - let signature = - Signature::from_slice(&authorization.signature).map_err(Error::PeerSignature)?; - self.key - .verify_strict(&signing_bytes(operation_tag, authorization)?, &signature) - .map_err(Error::PeerSignature)?; - let target = operation_target(request.operation.as_ref())?; - let principal = PeerPrincipal { - issuer: authorization.principal_issuer.clone(), - subject: authorization.principal_subject.clone(), - actions: authorization.actions.clone(), - }; - Ok(VerifiedPeerRequest { - request, - target, - principal, - origin_session: self.session, - operation_tag, - operation_bytes, - }) - } -} - -/// Structurally checked peer envelope whose claims are not authenticated. -/// -/// Use its session only to locate enrollment and its timeout only to bound -/// waiting. Actor admission requires [`PeerVerifier::verify_decoded`] first. -pub struct UnverifiedPeerRequest { - request: wire::PeerRequest, - session: SessionId, - operation_tag: u32, - operation_bytes: Vec, -} - -impl UnverifiedPeerRequest { - /// Strictly decodes one bounded request, retaining its exact signed payload. - pub fn decode(input: &[u8]) -> Result { - if input.len() > MAX_PEER_REQUEST_BYTES { - return Err(Error::Peer("request exceeds peer byte limit")); - } - let fields = validate_message(input, MessageKind::PeerRequest)?; - require_fields(&fields, &[1, 2, 3, 4])?; - let authorization_range = field_payload(&fields, 2)?; - if authorization_range.len() > MAX_AUTHORIZATION_BYTES { - return Err(Error::Peer("authorization exceeds 16 KiB")); - } - let authorization_fields = - validate_message(&input[authorization_range], MessageKind::Authorization)?; - require_fields(&authorization_fields, &[1, 2, 3, 5, 6, 7, 8, 9])?; - let (tag, payload) = oneof_payload(input, &fields, &[10, 11, 12, 13, 14, 15])?; - if payload.len() > MAX_OPERATION_BYTES { - return Err(Error::Peer("operation exceeds wire limit")); - } - validate_operation(tag, payload)?; - let request = wire::PeerRequest::decode(input)?; - if request.version != PROTOCOL_VERSION - || !(1..=2).contains(&request.hop_count) - || !(1..=60_000).contains(&request.remaining_ms) - { - return Err(Error::Peer("unsupported peer version, hop, or deadline")); - } - let authorization = request - .authorization - .as_ref() - .ok_or(Error::Peer("authorization is missing"))?; - let session = SessionId::try_from(authorization.origin_session.as_slice())?; - Ok(Self { - request, - session, - operation_tag: tag, - operation_bytes: payload.to_vec(), - }) - } - - /// Returns the untrusted session used solely for enrollment lookup. - #[must_use] - pub const fn session(&self) -> SessionId { - self.session - } - - /// Returns the bounded, untrusted transport wait budget in milliseconds. - #[must_use] - pub const fn remaining_ms(&self) -> u32 { - self.request.remaining_ms - } -} - -/// Authenticated request retaining the exact signed nested operation bytes. -pub struct VerifiedPeerRequest { - request: wire::PeerRequest, - target: CellTarget, - principal: PeerPrincipal, - origin_session: SessionId, - operation_tag: u32, - operation_bytes: Vec, -} - -impl VerifiedPeerRequest { - /// Returns the Cell the request targets. - #[must_use] - pub const fn target(&self) -> &CellTarget { - &self.target - } - - /// Returns the authenticated principal. - #[must_use] - pub const fn principal(&self) -> &PeerPrincipal { - &self.principal - } - - /// Returns the session the request originated from. - #[must_use] - pub const fn origin_session(&self) -> SessionId { - self.origin_session - } - - /// Returns how many peers forwarded the request. - #[must_use] - pub const fn hop_count(&self) -> u32 { - self.request.hop_count - } - - /// Returns the deadline remaining for the request. - #[must_use] - pub const fn remaining_ms(&self) -> u32 { - self.request.remaining_ms - } - - /// Returns the wire tag of the requested operation. - #[must_use] - pub const fn operation_tag(&self) -> u32 { - self.operation_tag - } - - /// Returns the decoded operation. - #[must_use] - pub fn operation(&self) -> Option<&wire::peer_request::Operation> { - self.request.operation.as_ref() - } - - /// Reports whether the principal may perform an action. - #[must_use] - pub fn permits(&self, action: &str) -> bool { - self.principal - .actions - .binary_search_by(|candidate| candidate.as_str().cmp(action)) - .is_ok() - } - - /// Preserves signed payload bytes while reducing the deadline for one final hop. - pub fn forward(&self, remaining_ms: u32) -> Result> { - if self.request.hop_count >= 2 { - return Err(Error::Peer("peer hop limit reached")); - } - if remaining_ms == 0 || remaining_ms > self.request.remaining_ms { - return Err(Error::Peer("forwarding extended or exhausted the deadline")); - } - let authorization = self - .request - .authorization - .clone() - .ok_or(Error::Peer("authorization is missing"))?; - encode_request( - authorization, - self.request.hop_count + 1, - remaining_ms, - self.operation_tag, - &self.operation_bytes, - ) - } -} - -/// Encodes one bounded canonical reply produced by the private dispatcher. -pub fn encode_peer_reply(reply: &wire::PeerReply) -> Result> { - validate_reply(reply)?; - let encoded = reply.encode_to_vec(); - if encoded.len() > MAX_PEER_REQUEST_BYTES { - return Err(Error::Peer("reply exceeds peer byte limit")); - } - Ok(encoded) -} - -/// Strictly decodes a reply without allowing Prost to discard unknown fields. -pub fn decode_peer_reply(input: &[u8]) -> Result { - if input.len() > MAX_PEER_REQUEST_BYTES { - return Err(Error::Peer("reply exceeds peer byte limit")); - } - let fields = validate_message(input, MessageKind::PeerReply)?; - if !fields.iter().any(|field| matches!(field.tag(), 1..=5)) { - return Err(Error::Peer("peer reply outcome is missing")); - } - let reply = wire::PeerReply::decode(input)?; - validate_reply(&reply)?; - Ok(reply) -} - -fn signing_bytes(tag: u32, authorization: &wire::PeerAuthorization) -> Result> { - let tag = u16::try_from(tag).map_err(|_| Error::Peer("operation tag overflow"))?; - let mut output = Vec::with_capacity(256); - output.extend_from_slice(b"crab.peer.v1\0"); - output.extend_from_slice(&tag.to_be_bytes()); - append_bytes(&mut output, &authorization.origin_session)?; - append_bytes(&mut output, authorization.principal_issuer.as_bytes())?; - append_bytes(&mut output, authorization.principal_subject.as_bytes())?; - append_u32(&mut output, authorization.actions.len())?; - for action in &authorization.actions { - append_bytes(&mut output, action.as_bytes())?; - } - append_bytes(&mut output, &authorization.release_digest)?; - output.extend_from_slice(&authorization.issued_at_ms.to_be_bytes()); - output.extend_from_slice(&authorization.expires_at_ms.to_be_bytes()); - append_bytes(&mut output, &authorization.payload_digest)?; - if output.len() > MAX_AUTHORIZATION_BYTES { - return Err(Error::Peer("canonical authorization exceeds 16 KiB")); - } - Ok(output) -} - -fn append_u32(output: &mut Vec, value: usize) -> Result<()> { - output.extend_from_slice( - &u32::try_from(value) - .map_err(|_| Error::Peer("canonical value exceeds u32"))? - .to_be_bytes(), - ); - Ok(()) -} - -fn append_bytes(output: &mut Vec, value: &[u8]) -> Result<()> { - append_u32(output, value.len())?; - output.extend_from_slice(value); - Ok(()) -} - -fn encode_request( - authorization: wire::PeerAuthorization, - hop_count: u32, - remaining_ms: u32, - operation_tag: u32, - operation: &[u8], -) -> Result> { - let authorization = authorization.encode_to_vec(); - if authorization.len() > MAX_AUTHORIZATION_BYTES || operation.len() > MAX_OPERATION_BYTES { - return Err(Error::Peer("peer request component exceeds limit")); - } - let mut output = Vec::with_capacity(authorization.len() + operation.len() + 32); - encode_varint_field(&mut output, 1, u64::from(PROTOCOL_VERSION)); - encode_bytes_field(&mut output, 2, &authorization)?; - encode_varint_field(&mut output, 3, u64::from(hop_count)); - encode_varint_field(&mut output, 4, u64::from(remaining_ms)); - encode_bytes_field(&mut output, operation_tag, operation)?; - if output.len() > MAX_PEER_REQUEST_BYTES { - return Err(Error::Peer("request exceeds peer byte limit")); - } - Ok(output) -} - -fn encode_varint_field(output: &mut Vec, field: u32, value: u64) { - encode_varint(output, u64::from(field) << 3); - encode_varint(output, value); -} - -fn encode_bytes_field(output: &mut Vec, field: u32, value: &[u8]) -> Result<()> { - encode_varint(output, (u64::from(field) << 3) | 2); - encode_varint( - output, - u64::try_from(value.len()).map_err(|_| Error::Peer("peer field length overflow"))?, - ); - output.extend_from_slice(value); - Ok(()) -} - -fn encode_varint(output: &mut Vec, mut value: u64) { - while value >= 0x80 { - output.push((value as u8) | 0x80); - value >>= 7; - } - output.push(value as u8); -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/peer/dispatch.rs b/crates/crab-cell-runtime/src/peer/dispatch.rs deleted file mode 100644 index 8d09d7b81..000000000 --- a/crates/crab-cell-runtime/src/peer/dispatch.rs +++ /dev/null @@ -1,705 +0,0 @@ -use std::{collections::HashMap, future::Future, pin::Pin, sync::Arc, time::Instant}; - -use futures_util::future::BoxFuture; -use prost::Message; - -use crate::cell::actor::CellHandle; -use crate::cell::executor::{MutationIdentity, Resolution, StoredOutcome}; -use crate::client::{CellDescription, CellReadReplica, Receipt}; -use crate::client::{ - CellTransport, EncodedCommand, EncodedObservation, EncodedQuery, EncodedResolve, - LocalCellTransport, encoded_command_operation_digest, local_description, next_metadata, - receipt, -}; -use crate::fleet::telemetry::{ - CellTelemetryHandle, PrimitiveOperationKind, PrimitiveOperationOutcome, -}; -use crate::identity::{CellTarget, Digest, IncarnationId, RequestId, SessionId}; -use crate::primitives::effects::InboxDelivery; -use crate::registry::{CommandInvocation, Registry}; -use crate::{Error, Result}; - -use super::{VerifiedPeerRequest, wire, wire_description}; - -const MAX_RESULT_BYTES: usize = 1024 * 1024; - -/// Resolves only a currently active owner on the receiving node. -pub trait PeerCellResolver: Send + Sync + 'static { - /// Resolves one target to an active local Cell handle. - fn resolve( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>>; -} - -/// Resolves only an admitted local read-only snapshot on the receiving node. -pub trait PeerReplicaResolver: Send + Sync + 'static { - /// Returns a ready snapshot for the exact target or an unavailable error. - fn resolve( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>>; -} - -/// Controls admitted snapshots after the receiver authorizes the peer operation. -pub trait PeerReplicaControl: Send + Sync + 'static { - /// Admits a selected snapshot only on a current owner's authenticated hint. - fn activate( - &self, - target: CellTarget, - origin: SessionId, - ) -> BoxFuture<'static, Result>; - - /// Reports the exact admitted position and current owner readiness. - fn status(&self, target: CellTarget) -> BoxFuture<'static, Result<(Receipt, bool)>>; -} - -/// Rechecks current product authorization after peer authentication. -pub trait PeerAuthorizer: Send + Sync + 'static { - /// Rechecks product authorization for one verified request. - fn authorize(&self, request: &VerifiedPeerRequest) -> Result<()>; -} - -/// Executes authenticated peer work through the canonical local Cell transport. -pub struct PeerDispatcher { - registry: Arc, - resolver: Arc, - replicas: Option>, - replica_control: Option>, - authorizer: Arc, - telemetry: CellTelemetryHandle, -} - -impl PeerDispatcher { - /// Creates a dispatcher over the compiled registry and its resolver and - /// authorizer. - #[must_use] - pub fn new( - registry: Arc, - resolver: Arc, - authorizer: Arc, - ) -> Self { - Self { - registry, - resolver, - replicas: None, - replica_control: None, - authorizer, - telemetry: CellTelemetryHandle::default(), - } - } - - /// Reports primitive operations executed for peer requests. - #[must_use] - pub fn with_telemetry(mut self, telemetry: CellTelemetryHandle) -> Self { - self.telemetry = telemetry; - self - } - - /// Binds an admitted snapshot resolver for explicit peer replica queries. - #[must_use] - pub fn with_replica_resolver(mut self, replicas: Arc) -> Self { - self.replicas = Some(replicas); - self - } - - /// Binds owner-hinted activation and readiness to the node's reader lifecycle. - #[must_use] - pub fn with_replica_control(mut self, control: Arc) -> Self { - self.replica_control = Some(control); - self - } - - /// Authorizes, resolves and dispatches one verified request without a second SQL path. - pub async fn dispatch(&self, request: &VerifiedPeerRequest, now_ms: i64) -> wire::PeerReply { - if let Err(error) = self.authorizer.authorize(request) { - return error_reply(error); - } - let resolved = if matches!( - request.operation(), - Some(wire::peer_request::Operation::Read(wire::ReadRequest { - operation: Some( - wire::read_request::Operation::ReplicaQuery(_) - | wire::read_request::Operation::ReplicaActivate(_) - | wire::read_request::Operation::ReplicaStatus(_) - ), - .. - })) - ) { - // Explicit replica queries use snapshot admission, never owner activation. - Err(Error::CellNotActive) - } else { - self.resolver.resolve(request.target().clone()).await - }; - self.dispatch_authorized(request, now_ms, resolved).await - } - - /// Dispatches with a receiver-resolved local handle after rechecking authorization. - /// - /// The handle must still match the exact target; actor admission fences a - /// handle whose owner changed after resolution. Replica queries ignore the - /// owner result and use the admitted snapshot resolver. - pub async fn dispatch_resolved( - &self, - request: &VerifiedPeerRequest, - now_ms: i64, - resolved: Result, - ) -> wire::PeerReply { - if let Err(error) = self.authorizer.authorize(request) { - return error_reply(error); - } - self.dispatch_authorized(request, now_ms, resolved).await - } - - async fn dispatch_authorized( - &self, - request: &VerifiedPeerRequest, - now_ms: i64, - resolved: Result, - ) -> wire::PeerReply { - if let Some(wire::peer_request::Operation::Read(read)) = request.operation() - && matches!( - read.operation, - Some( - wire::read_request::Operation::ReplicaActivate(true) - | wire::read_request::Operation::ReplicaStatus(true) - ) - ) - { - let result = async { - let control = self - .replica_control - .as_ref() - .ok_or(Error::ReplicaUnavailable)?; - if matches!( - read.operation, - Some(wire::read_request::Operation::ReplicaActivate(true)) - ) { - control - .activate(request.target().clone(), request.origin_session()) - .await - .map(|receipt| (receipt, true)) - } else { - control.status(request.target().clone()).await - } - } - .await; - return match result { - Ok((receipt, ready)) => wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - receipt: Some(wire_receipt(receipt)), - result: Some(wire::read_reply::Result::ReplicaReady(ready)), - })), - }, - Err(error) => error_reply(error), - }; - } - if let Some(wire::peer_request::Operation::Read(read)) = request.operation() - && let Some(wire::read_request::Operation::ReplicaQuery(query)) = - read.operation.as_ref() - { - return self - .replica_query(request.target().clone(), read, query, now_ms) - .await; - } - let handle = match resolved { - Ok(handle) => handle, - Err(error) => return error_reply(error), - }; - let entry = handle.catalog().entry(); - if handle.cell_id() != request.target().cell_id() - || entry.namespace() != request.target().namespace() - || entry.partition() != request.target().partition() - { - return error_reply(Error::CatalogCollision); - } - let transport = LocalCellTransport { - registry: Arc::clone(&self.registry), - handles: Arc::new(HashMap::from([(handle.cell_id(), handle.clone())])), - handle, - telemetry: self.telemetry.clone(), - }; - match request.operation() { - Some(wire::peer_request::Operation::Mutate(mutation)) => { - self.mutate(&transport, mutation, now_ms).await - } - Some(wire::peer_request::Operation::Read(read)) => { - self.read(&transport, read, now_ms).await - } - Some(wire::peer_request::Operation::Resolve(resolve)) => { - self.resolve(&transport, resolve, now_ms).await - } - Some(wire::peer_request::Operation::DeliverEffect(effect)) => { - self.deliver_effect(&transport, effect, now_ms).await - } - Some(wire::peer_request::Operation::ResolveEffect(resolve)) => { - self.resolve_effect(&transport, resolve, now_ms).await - } - Some(wire::peer_request::Operation::Migrate(migration)) => { - self.migrate(&transport, migration, now_ms).await - } - None => error_reply(Error::Peer("peer operation is missing")), - } - } - - async fn replica_query( - &self, - target: CellTarget, - read: &wire::ReadRequest, - query: &wire::CellQuery, - now_ms: i64, - ) -> wire::PeerReply { - let result = async { - let resolver = self.replicas.as_ref().ok_or(Error::ReplicaUnavailable)?; - let replica = resolver.resolve(target.clone()).await.map_err(|error| { - if matches!(error, Error::CellNotActive) { - Error::ReplicaUnavailable - } else { - error - } - })?; - let expected = replica.description(); - validate_expected(read.expected.as_ref(), expected)?; - let (module, operation) = self.registry.routed_query_contract( - target.namespace(), - query.query_id, - query.codec_version, - )?; - replica - .query_encoded(EncodedQuery { - target, - expected, - minimum: read.minimum.as_ref().map(runtime_receipt).transpose()?, - now_ms, - module, - operation_id: query.query_id, - codec_version: query.codec_version, - input: query.input.clone(), - input_limit: operation.input_limit, - output_limit: operation.output_limit, - }) - .await - } - .await; - match result { - Ok(observation) => wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - receipt: Some(wire_receipt(observation.receipt)), - result: Some(wire::read_reply::Result::CommandOutput(observation.output)), - })), - }, - Err(error) => error_reply(error), - } - } - - /// Encodes the validated reply for the management HTTP response body. - pub async fn dispatch_bytes( - &self, - request: &VerifiedPeerRequest, - now_ms: i64, - ) -> Result> { - let reply = self.dispatch(request, now_ms).await; - super::encode_peer_reply(&reply) - } - - async fn mutate( - &self, - transport: &LocalCellTransport, - request: &wire::MutationRequest, - now_ms: i64, - ) -> wire::PeerReply { - let result = self.mutate_inner(transport, request, now_ms).await; - let outcome = match result { - Ok(outcome) => mutation_reply(transport, outcome), - // Command execution wraps fenced commit/publication failures as - // OutcomeUnknown. A direct FULL here therefore proves rollback; - // migration and other peer operations retain their own contract. - Err(Error::Sqlite(error)) - if error.sqlite_error_code() == Some(rusqlite::ErrorCode::DiskFull) => - { - return error_reply(Error::Capacity("SQLite database full")); - } - Err(error) => return error_reply(error), - }; - wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Mutation(outcome)), - } - } - - async fn mutate_inner( - &self, - transport: &LocalCellTransport, - request: &wire::MutationRequest, - now_ms: i64, - ) -> Result { - let expected = local_description(&transport.handle); - validate_expected(request.expected.as_ref(), expected)?; - let identity = mutation_identity( - request - .identity - .as_ref() - .ok_or(Error::Peer("mutation identity is missing"))?, - expected.incarnation, - )?; - let command = match request.operation.as_ref() { - Some(wire::mutation_request::Operation::CellCommand(command)) => command, - _ => return Err(Error::Peer("typed mutation operation is not implemented")), - }; - let (module, descriptor) = self.registry.routed_command_contract( - request_target(request.target.as_ref())?.namespace(), - command.command_id, - command.codec_version, - )?; - validate_description( - &self.registry, - module, - expected, - descriptor.schema_min, - descriptor.schema_max, - )?; - let operation_digest = encoded_command_operation_digest( - expected, - identity, - command.command_id, - command.codec_version, - &command.input, - )?; - transport - .command(EncodedCommand { - target: request_target(request.target.as_ref())?, - expected, - identity, - operation_digest, - now_ms, - module, - operation_id: command.command_id, - codec_version: command.codec_version, - input: command.input.clone(), - input_limit: descriptor.input_limit, - output_limit: descriptor.output_limit, - }) - .await - } - - async fn read( - &self, - transport: &LocalCellTransport, - request: &wire::ReadRequest, - now_ms: i64, - ) -> wire::PeerReply { - let expected = local_description(&transport.handle); - let result = match request.operation.as_ref() { - Some(wire::read_request::Operation::Describe(true)) => { - wire::read_reply::Result::Description(wire_description(expected)) - } - Some(wire::read_request::Operation::CellQuery(query)) => { - match self - .query(transport, request, query, expected, now_ms) - .await - { - Ok(observation) => { - return wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - receipt: Some(wire_receipt(observation.receipt)), - result: Some(wire::read_reply::Result::CommandOutput( - observation.output, - )), - })), - }; - } - Err(error) => return error_reply(error), - } - } - _ => return error_reply(Error::Peer("typed read operation is not implemented")), - }; - wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - receipt: None, - result: Some(result), - })), - } - } - - async fn query( - &self, - transport: &LocalCellTransport, - request: &wire::ReadRequest, - query: &wire::CellQuery, - expected: CellDescription, - now_ms: i64, - ) -> Result { - validate_expected(request.expected.as_ref(), expected)?; - let target = request_target(request.target.as_ref())?; - let (module, descriptor) = self.registry.routed_query_contract( - target.namespace(), - query.query_id, - query.codec_version, - )?; - validate_description( - &self.registry, - module, - expected, - descriptor.schema_min, - descriptor.schema_max, - )?; - transport - .query(EncodedQuery { - target, - expected, - minimum: request.minimum.as_ref().map(runtime_receipt).transpose()?, - now_ms, - module, - operation_id: query.query_id, - codec_version: query.codec_version, - input: query.input.clone(), - input_limit: descriptor.input_limit, - output_limit: descriptor.output_limit, - }) - .await - } - - async fn resolve( - &self, - transport: &LocalCellTransport, - request: &wire::ResolveRequest, - now_ms: i64, - ) -> wire::PeerReply { - let expected = local_description(&transport.handle); - let result = async { - validate_expected(request.expected.as_ref(), expected)?; - let identity = mutation_identity( - request - .identity - .as_ref() - .ok_or(Error::Peer("resolve identity is missing"))?, - expected.incarnation, - )?; - let operation_digest = Digest::try_from(request.operation_digest.as_slice())?; - transport - .resolve(EncodedResolve { - target: request_target(request.target.as_ref())?, - expected, - identity, - operation_digest, - now_ms, - max_result_bytes: MAX_RESULT_BYTES, - }) - .await - } - .await; - let resolution = match result { - Ok(value) => value, - Err(error) => return error_reply(error), - }; - wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Resolve(resolve_reply( - transport, resolution, - ))), - } - } - - async fn deliver_effect( - &self, - transport: &LocalCellTransport, - request: &wire::EffectRequest, - now_ms: i64, - ) -> wire::PeerReply { - let result = self.deliver_effect_inner(transport, request, now_ms).await; - let outcome = match result { - Ok(outcome) => mutation_reply(transport, outcome), - Err(error) => return error_reply(error), - }; - wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Mutation(outcome)), - } - } - - async fn deliver_effect_inner( - &self, - transport: &LocalCellTransport, - request: &wire::EffectRequest, - now_ms: i64, - ) -> Result { - let expected = local_description(&transport.handle); - validate_effect_incarnation(request.destination_incarnation.as_slice(), expected)?; - let target = request_target(request.target.as_ref())?; - let identity = request - .identity - .as_ref() - .ok_or(Error::Peer("effect identity is missing"))?; - let command = match request.operation.as_ref() { - Some(wire::effect_request::Operation::CellCommand(command)) => command, - _ => return Err(Error::Peer("typed effect operation is not implemented")), - }; - let (module, descriptor) = self.registry.routed_command_contract( - target.namespace(), - command.command_id, - command.codec_version, - )?; - validate_description( - &self.registry, - module, - expected, - descriptor.schema_min, - descriptor.schema_max, - )?; - if command.input.len() > descriptor.input_limit as usize { - return Err(Error::Command("encoded effect input exceeds limit")); - } - let effect_id = exact_effect_id(identity)?; - let mut canonical_operation = request.clone(); - canonical_operation.destination_incarnation.clear(); - let encoded_operation = canonical_operation.encode_to_vec(); - let encoded_request = request.encode_to_vec(); - let delivery = InboxDelivery { - effect_id, - operation_digest: crate::primitives::effects::effect_operation_digest( - target.cell_id(), - effect_id, - &encoded_operation, - ), - expires_at_ms: identity.expires_at_ms, - }; - let registry = Arc::clone(&self.registry); - let telemetry = self.telemetry.clone(); - let schema = transport.handle.schema(); - let input = command.input.clone(); - let operation_id = command.command_id; - let codec_version = command.codec_version; - transport - .handle - .deliver_effect( - delivery, - now_ms, - encoded_request.len(), - descriptor.output_limit as usize, - move |transaction| { - let (sequence, logical_time_ms) = next_metadata(transaction, now_ms)?; - let started = Instant::now(); - let result = registry.execute_command_with_issue_time( - transaction, - CommandInvocation { - module, - operation_id, - codec_version, - schema, - target: target.clone(), - sequence, - now_ms: logical_time_ms, - input: &input, - }, - now_ms, - ); - telemetry.primitive_operation( - module, - PrimitiveOperationKind::Command, - PrimitiveOperationOutcome::from(&result), - started.elapsed(), - ); - result - }, - ) - .await - } - - async fn resolve_effect( - &self, - transport: &LocalCellTransport, - request: &wire::EffectResolveRequest, - now_ms: i64, - ) -> wire::PeerReply { - let result = async { - let expected = local_description(&transport.handle); - validate_effect_incarnation(request.destination_incarnation.as_slice(), expected)?; - let identity = request - .identity - .as_ref() - .ok_or(Error::Peer("effect Resolve identity is missing"))?; - let delivery = InboxDelivery { - effect_id: exact_effect_id(identity)?, - operation_digest: Digest::try_from(request.operation_digest.as_slice())?, - expires_at_ms: identity.expires_at_ms, - }; - transport - .handle - .resolve_effect(delivery, now_ms, MAX_RESULT_BYTES) - .await - } - .await; - let resolution = match result { - Ok(value) => value, - Err(error) => return error_reply(error), - }; - wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Resolve(resolve_reply( - transport, resolution, - ))), - } - } - - async fn migrate( - &self, - transport: &LocalCellTransport, - request: &wire::MigrationRequest, - now_ms: i64, - ) -> wire::PeerReply { - let result = self.migrate_inner(transport, request, now_ms).await; - match result { - Ok(current) => wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Migration(wire::MigrationReply { - description: Some(wire_description(current)), - })), - }, - Err(error) => error_reply(error), - } - } - - async fn migrate_inner( - &self, - transport: &LocalCellTransport, - request: &wire::MigrationRequest, - now_ms: i64, - ) -> Result { - let target = request_target(request.target.as_ref())?; - let from_code = Digest::try_from(request.from_code.as_slice())?; - let to_code = Digest::try_from(request.to_code.as_slice())?; - let plan = self - .registry - .next_migration(target.namespace(), from_code, request.from_schema)? - .ok_or(Error::Registry("requested Cell migration has no successor"))?; - if plan.to_code() != to_code || plan.to_schema() != request.to_schema { - return Err(Error::Registry( - "requested Cell migration differs from the compiled successor", - )); - } - - let current = local_description(&transport.handle); - if current.cell != target.cell_id() - || current.incarnation != IncarnationId::try_from(request.incarnation.as_slice())? - { - return Err(Error::Fenced); - } - if current.code == plan.to_code() && current.schema >= plan.to_schema() { - return Ok(current); - } - if current.code != plan.from_code() || current.schema != plan.from_schema() { - return Err(Error::Fenced); - } - - let migrated = transport.handle.migrate(plan, now_ms).await?; - let description = local_description(&migrated.handle); - if description.code != migrated.outcome.code - || description.schema != migrated.outcome.schema - { - return Err(Error::Control( - "published migration capability and outcome differ", - )); - } - Ok(description) - } -} - -mod convert; - -pub(crate) use convert::error_reply; -use convert::*; diff --git a/crates/crab-cell-runtime/src/peer/dispatch/convert.rs b/crates/crab-cell-runtime/src/peer/dispatch/convert.rs deleted file mode 100644 index c0246313a..000000000 --- a/crates/crab-cell-runtime/src/peer/dispatch/convert.rs +++ /dev/null @@ -1,293 +0,0 @@ -//! Request validation and reply conversion for dispatched peer operations. - -use super::*; - -pub(super) fn validate_effect_incarnation(value: &[u8], expected: CellDescription) -> Result<()> { - if IncarnationId::try_from(value)? != expected.incarnation { - return Err(Error::Fenced); - } - Ok(()) -} - -pub(super) fn validate_expected( - observed: Option<&wire::CellDescription>, - current: CellDescription, -) -> Result<()> { - let observed = observed.ok_or(Error::Peer("expected Cell description is missing"))?; - if observed.cell_id.as_slice() != current.cell.as_bytes() - || observed.incarnation.as_slice() != current.incarnation.as_bytes() - || observed.code.as_slice() != current.code.as_bytes() - || observed.schema != current.schema - { - return Err(Error::Fenced); - } - Ok(()) -} - -pub(super) fn exact_effect_id(identity: &wire::EffectIdentity) -> Result<[u8; 32]> { - let effect_id = <[u8; 32]>::try_from(identity.effect_id.as_slice()) - .map_err(|_| Error::Peer("invalid effect ID length"))?; - let expected = crate::primitives::effects::effect_id( - crate::CellId::try_from(identity.source_cell.as_slice())?, - IncarnationId::try_from(identity.source_incarnation.as_slice())?, - identity.source_sequence, - identity.ordinal, - ); - if effect_id != expected { - return Err(Error::Peer("effect identity derivation does not match")); - } - Ok(effect_id) -} - -pub(super) fn mutation_identity( - identity: &wire::MutationIdentity, - expected_incarnation: IncarnationId, -) -> Result { - if IncarnationId::try_from(identity.incarnation.as_slice())? != expected_incarnation { - return Err(Error::Fenced); - } - Ok(MutationIdentity { - request_id: RequestId::try_from(identity.request_id.as_slice())?, - issued_at_ms: identity.issued_at_ms, - expires_at_ms: identity.expires_at_ms, - }) -} - -pub(super) fn request_target(target: Option<&wire::Target>) -> Result { - let target = target.ok_or(Error::Peer("peer target is missing"))?; - CellTarget::new( - crate::TenantId::try_from(target.tenant_id.as_slice())?, - crate::ApplicationId::try_from(target.application_id.as_slice())?, - crate::NamespaceId::try_from(target.namespace_id.as_slice())?, - &target.partition, - ) -} - -pub(super) fn validate_description( - registry: &Registry, - module: &str, - description: CellDescription, - schema_min: u32, - schema_max: u32, -) -> Result<()> { - if !registry.supports_module_code(module, description.code, description.schema) - || !(schema_min..=schema_max).contains(&description.schema) - { - return Err(Error::Fenced); - } - Ok(()) -} - -pub(super) fn mutation_reply( - transport: &LocalCellTransport, - outcome: StoredOutcome, -) -> wire::MutationReply { - let description = local_description(&transport.handle); - match outcome { - StoredOutcome::Success { - result, - commit_sequence, - } => wire::MutationReply { - receipt: Some(wire_receipt(receipt(description, commit_sequence))), - outcome: Some(wire::mutation_reply::Outcome::Result( - wire::MutationResult { - result: Some(wire::mutation_result::Result::CommandOutput(result)), - }, - )), - }, - StoredOutcome::Rejected { - result, - commit_sequence, - } => wire::MutationReply { - receipt: Some(wire_receipt(receipt(description, commit_sequence))), - outcome: Some(wire::mutation_reply::Outcome::Error(wire::Error { - code: wire::error::Code::PreconditionFailed as i32, - outcome: wire::error::Outcome::Rejected as i32, - message: "Cell command was durably rejected".into(), - retry_after_ms: 0, - application_details: result, - })), - }, - } -} - -pub(super) fn resolve_reply( - transport: &LocalCellTransport, - resolution: Resolution, -) -> wire::ResolveReply { - match resolution { - Resolution::Committed(outcome) => wire::ResolveReply { - state: match &outcome { - StoredOutcome::Success { .. } => wire::resolve_reply::State::Committed as i32, - StoredOutcome::Rejected { .. } => wire::resolve_reply::State::Rejected as i32, - }, - reply: Some(mutation_reply(transport, outcome)), - }, - Resolution::Absent => wire::ResolveReply { - state: wire::resolve_reply::State::Absent as i32, - reply: None, - }, - Resolution::Unknown => wire::ResolveReply { - state: wire::resolve_reply::State::Unknown as i32, - reply: None, - }, - Resolution::Expired => wire::ResolveReply { - state: wire::resolve_reply::State::Expired as i32, - reply: None, - }, - } -} - -pub(super) fn runtime_receipt(value: &wire::Receipt) -> Result { - Ok(Receipt { - cell: crate::CellId::try_from(value.cell_id.as_slice())?, - incarnation: IncarnationId::try_from(value.incarnation.as_slice())?, - commit_sequence: value.commit_sequence, - }) -} - -pub(super) fn wire_receipt(value: Receipt) -> wire::Receipt { - wire::Receipt { - cell_id: value.cell.as_bytes().to_vec(), - incarnation: value.incarnation.as_bytes().to_vec(), - commit_sequence: value.commit_sequence, - } -} - -pub(crate) fn error_reply(error: Error) -> wire::PeerReply { - let application_details = match &error { - Error::ReplicaBehind { - observed_sequence, - minimum_sequence, - } => [ - observed_sequence.to_be_bytes(), - minimum_sequence.to_be_bytes(), - ] - .concat(), - _ => Vec::new(), - }; - let (code, outcome, message, retry_after_ms) = match error { - Error::RequestConflict => ( - wire::error::Code::RequestIdConflict, - wire::error::Outcome::Rejected, - "request identity conflicts with a stored operation", - 0, - ), - Error::OutcomeUnknown { .. } | Error::EffectOutcomeUnknown { .. } => ( - wire::error::Code::OutcomeUnknown, - wire::error::Outcome::Unknown, - "accepted command outcome requires resolution", - 100, - ), - Error::EffectExpired => ( - wire::error::Code::RequestExpired, - wire::error::Outcome::Rejected, - "Cell effect expired before delivery", - 0, - ), - Error::Capacity(_) => ( - wire::error::Code::ResourceExhausted, - wire::error::Outcome::NotStarted, - "Cell runtime capacity is exhausted", - 100, - ), - Error::ReplicaBehind { .. } => ( - wire::error::Code::ReplicaBehind, - wire::error::Outcome::NotStarted, - "Cell read replica is behind the requested receipt", - 100, - ), - Error::ReplicaUnavailable => ( - wire::error::Code::ReplicaUnavailable, - wire::error::Outcome::NotStarted, - "Cell read replica is unavailable", - 100, - ), - Error::Fenced - | Error::CellNotActive - | Error::CellDraining - | Error::RuntimeClosed - | Error::StreamCancelled - | Error::PendingPublication => ( - wire::error::Code::Unavailable, - wire::error::Outcome::NotStarted, - "Cell owner is unavailable", - 100, - ), - Error::Registry(_) => ( - wire::error::Code::SchemaIncompatible, - wire::error::Outcome::Rejected, - "compiled Cell operation is unavailable", - 0, - ), - Error::Identity(_) | Error::Command(_) | Error::Peer(_) | Error::PeerDecode(_) => ( - wire::error::Code::InvalidArgument, - wire::error::Outcome::Rejected, - "invalid Cell peer request", - 0, - ), - Error::Deadline => ( - wire::error::Code::Unavailable, - wire::error::Outcome::Unknown, - "Cell operation deadline expired", - 100, - ), - Error::PeerSignature(_) => ( - wire::error::Code::PermissionDenied, - wire::error::Outcome::Rejected, - "Cell peer authorization failed", - 0, - ), - Error::PeerAuthorization(_) => ( - wire::error::Code::PermissionDenied, - wire::error::Outcome::Rejected, - "Cell peer principal is not authorized", - 0, - ), - Error::Control(_) - | Error::Catalog(_) - | Error::Node(_) - | Error::Release(_) - | Error::Backup(_) - | Error::Retention(_) - | Error::RetentionIo(_) - | Error::RetentionWorkerJoin(_) - | Error::FollowerIo(_) - | Error::FollowerWorkerJoin(_) - | Error::PeerTransport { .. } - | Error::PeerTransportUnknown { .. } - | Error::CatalogCollision - | Error::CatalogFull - | Error::Json(_) - | Error::Codec(_) - | Error::Sqlite(_) - | Error::Utf8(_) - | Error::Storage(_) - | Error::Ltx(_) - | Error::WorkerStart(_) - | Error::WorkerJoin(_) - | Error::WorkerPanic - | Error::ActivityWorkerStart(_) - | Error::ActivityWorkerJoin(_) - | Error::ActivityWorkerPanic - | Error::ActivityPanic - | Error::NativePanic - | Error::RuntimeStart(_) - | Error::Facility { .. } - | Error::CellAlreadyActive => ( - wire::error::Code::Internal, - wire::error::Outcome::Unknown, - "Cell runtime failed", - 100, - ), - }; - wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Error(wire::Error { - code: code as i32, - outcome: outcome as i32, - message: message.into(), - retry_after_ms, - application_details, - })), - } -} diff --git a/crates/crab-cell-runtime/src/peer/protobuf.rs b/crates/crab-cell-runtime/src/peer/protobuf.rs deleted file mode 100644 index 64bfbe7f8..000000000 --- a/crates/crab-cell-runtime/src/peer/protobuf.rs +++ /dev/null @@ -1,362 +0,0 @@ -use std::{collections::HashSet, ops::Range}; - -use crate::{Error, Result}; - -#[derive(Clone, Copy)] -pub(super) enum MessageKind { - PeerRequest, - Authorization, - MutationRequest, - ReadRequest, - ResolveRequest, - EffectRequest, - EffectResolveRequest, - MigrationRequest, - EffectIdentity, - Target, - MutationIdentity, - CellCommand, - CellQuery, - Receipt, - PeerReply, - MutationReply, - MutationResult, - ReadReply, - ResolveReply, - Error, - CellDescription, - MigrationReply, -} - -#[derive(Clone, Copy)] -struct FieldRule { - tag: u32, - wire: u8, - nested: Option, - repeated: bool, - oneof: u8, -} - -const fn scalar(tag: u32, wire: u8) -> FieldRule { - FieldRule { - tag, - wire, - nested: None, - repeated: false, - oneof: 0, - } -} - -const fn message(tag: u32, nested: MessageKind) -> FieldRule { - FieldRule { - tag, - wire: 2, - nested: Some(nested), - repeated: false, - oneof: 0, - } -} - -const fn oneof(tag: u32, nested: Option, group: u8) -> FieldRule { - FieldRule { - tag, - wire: 2, - nested, - repeated: false, - oneof: group, - } -} - -const fn scalar_oneof(tag: u32, wire: u8, group: u8) -> FieldRule { - FieldRule { - tag, - wire, - nested: None, - repeated: false, - oneof: group, - } -} - -const fn repeated(tag: u32, wire: u8) -> FieldRule { - FieldRule { - tag, - wire, - nested: None, - repeated: true, - oneof: 0, - } -} - -fn rules(kind: MessageKind) -> Vec { - match kind { - MessageKind::PeerRequest => vec![ - scalar(1, 0), - message(2, MessageKind::Authorization), - scalar(3, 0), - scalar(4, 0), - oneof(10, Some(MessageKind::MutationRequest), 1), - oneof(11, Some(MessageKind::ReadRequest), 1), - oneof(12, Some(MessageKind::ResolveRequest), 1), - oneof(13, Some(MessageKind::EffectRequest), 1), - oneof(14, Some(MessageKind::EffectResolveRequest), 1), - oneof(15, Some(MessageKind::MigrationRequest), 1), - ], - MessageKind::Authorization => vec![ - scalar(1, 2), - scalar(2, 2), - scalar(3, 2), - repeated(4, 2), - scalar(5, 2), - scalar(6, 0), - scalar(7, 0), - scalar(8, 2), - scalar(9, 2), - ], - MessageKind::MutationRequest => vec![ - message(1, MessageKind::Target), - message(2, MessageKind::MutationIdentity), - scalar(3, 0), - message(4, MessageKind::CellDescription), - oneof(10, Some(MessageKind::CellCommand), 1), - ], - MessageKind::ReadRequest => vec![ - message(1, MessageKind::Target), - scalar(2, 0), - message(3, MessageKind::Receipt), - message(4, MessageKind::CellDescription), - scalar_oneof(10, 0, 1), - oneof(15, Some(MessageKind::CellQuery), 1), - scalar_oneof(16, 0, 1), - oneof(17, Some(MessageKind::CellQuery), 1), - scalar_oneof(18, 0, 1), - scalar_oneof(19, 0, 1), - ], - MessageKind::ResolveRequest => vec![ - message(1, MessageKind::Target), - message(2, MessageKind::MutationIdentity), - scalar(3, 2), - message(4, MessageKind::CellDescription), - ], - MessageKind::EffectRequest => vec![ - message(1, MessageKind::Target), - scalar(2, 2), - message(3, MessageKind::EffectIdentity), - oneof(10, Some(MessageKind::CellCommand), 1), - ], - MessageKind::EffectResolveRequest => vec![ - message(1, MessageKind::Target), - scalar(2, 2), - message(3, MessageKind::EffectIdentity), - scalar(4, 2), - ], - MessageKind::MigrationRequest => vec![ - message(1, MessageKind::Target), - scalar(2, 2), - scalar(3, 2), - scalar(4, 0), - scalar(5, 2), - scalar(6, 0), - ], - MessageKind::EffectIdentity => vec![ - scalar(1, 2), - scalar(2, 2), - scalar(3, 2), - scalar(4, 0), - scalar(5, 0), - scalar(6, 0), - ], - MessageKind::Target => vec![scalar(1, 2), scalar(2, 2), scalar(3, 2), scalar(4, 2)], - MessageKind::MutationIdentity => { - vec![scalar(1, 2), scalar(2, 2), scalar(3, 0), scalar(4, 0)] - } - MessageKind::CellCommand | MessageKind::CellQuery => { - vec![scalar(1, 0), scalar(2, 0), scalar(3, 2)] - } - MessageKind::Receipt => vec![scalar(1, 2), scalar(2, 2), scalar(3, 0)], - MessageKind::PeerReply => vec![ - oneof(1, Some(MessageKind::MutationReply), 1), - oneof(2, Some(MessageKind::ReadReply), 1), - oneof(3, Some(MessageKind::ResolveReply), 1), - oneof(4, Some(MessageKind::Error), 1), - oneof(5, Some(MessageKind::MigrationReply), 1), - ], - MessageKind::MutationReply => vec![ - message(1, MessageKind::Receipt), - oneof(2, Some(MessageKind::MutationResult), 1), - oneof(3, Some(MessageKind::Error), 1), - ], - MessageKind::MutationResult => vec![scalar_oneof(1, 2, 1)], - MessageKind::ReadReply => vec![ - message(1, MessageKind::Receipt), - oneof(2, Some(MessageKind::CellDescription), 1), - scalar_oneof(6, 2, 1), - oneof(7, Some(MessageKind::Error), 1), - scalar_oneof(8, 0, 1), - scalar_oneof(9, 0, 1), - ], - MessageKind::ResolveReply => vec![scalar(1, 0), message(2, MessageKind::MutationReply)], - MessageKind::Error => vec![ - scalar(1, 0), - scalar(2, 0), - scalar(3, 2), - scalar(4, 0), - scalar(5, 2), - ], - MessageKind::CellDescription => { - vec![scalar(1, 2), scalar(2, 2), scalar(3, 2), scalar(4, 0)] - } - MessageKind::MigrationReply => vec![message(1, MessageKind::CellDescription)], - } -} - -pub(super) struct FieldOccurrence { - tag: u32, - payload: Option>, -} - -impl FieldOccurrence { - pub(super) const fn tag(&self) -> u32 { - self.tag - } -} - -pub(super) fn validate_message(input: &[u8], kind: MessageKind) -> Result> { - let rules = rules(kind); - let mut position = 0; - let mut fields = Vec::new(); - let mut singular = HashSet::new(); - let mut oneofs = HashSet::new(); - while position < input.len() { - let key = decode_varint(input, &mut position)?; - let tag = u32::try_from(key >> 3).map_err(|_| Error::Peer("field tag overflow"))?; - let wire = (key & 7) as u8; - if tag == 0 { - return Err(Error::Peer("zero Protobuf field tag")); - } - let rule = rules - .iter() - .find(|rule| rule.tag == tag) - .ok_or(Error::Peer("unknown Protobuf field"))?; - if wire != rule.wire { - return Err(Error::Peer("wrong Protobuf wire type")); - } - if !rule.repeated && !singular.insert(tag) { - return Err(Error::Peer("duplicate singular Protobuf field")); - } - if rule.oneof != 0 && !oneofs.insert(rule.oneof) { - return Err(Error::Peer("duplicate Protobuf oneof")); - } - let payload = match wire { - 0 => { - decode_varint(input, &mut position)?; - None - } - 1 => { - take(input, &mut position, 8)?; - None - } - 2 => { - let length = usize::try_from(decode_varint(input, &mut position)?) - .map_err(|_| Error::Peer("Protobuf length overflow"))?; - let start = position; - take(input, &mut position, length)?; - let range = start..position; - if let Some(nested) = rule.nested { - validate_message(&input[range.clone()], nested)?; - } - Some(range) - } - 5 => { - take(input, &mut position, 4)?; - None - } - _ => return Err(Error::Peer("unsupported Protobuf wire type")), - }; - fields.push(FieldOccurrence { tag, payload }); - } - Ok(fields) -} - -fn decode_varint(input: &[u8], position: &mut usize) -> Result { - let mut value = 0_u64; - for shift in (0..=63).step_by(7) { - let byte = *input - .get(*position) - .ok_or(Error::Peer("truncated Protobuf varint"))?; - *position += 1; - if shift == 63 && byte > 1 { - return Err(Error::Peer("Protobuf varint overflow")); - } - value |= u64::from(byte & 0x7f) << shift; - if byte & 0x80 == 0 { - return Ok(value); - } - } - Err(Error::Peer("Protobuf varint overflow")) -} - -fn take<'a>(input: &'a [u8], position: &mut usize, length: usize) -> Result<&'a [u8]> { - let end = position - .checked_add(length) - .filter(|end| *end <= input.len()) - .ok_or(Error::Peer("truncated Protobuf field"))?; - let value = &input[*position..end]; - *position = end; - Ok(value) -} - -pub(super) fn require_fields(fields: &[FieldOccurrence], required: &[u32]) -> Result<()> { - if required - .iter() - .any(|required| !fields.iter().any(|field| field.tag == *required)) - { - return Err(Error::Peer("required Protobuf field is missing")); - } - Ok(()) -} - -pub(super) fn field_payload(fields: &[FieldOccurrence], tag: u32) -> Result> { - fields - .iter() - .find(|field| field.tag == tag) - .and_then(|field| field.payload.clone()) - .ok_or(Error::Peer("required Protobuf payload is missing")) -} - -pub(super) fn oneof_payload<'a>( - input: &'a [u8], - fields: &[FieldOccurrence], - tags: &[u32], -) -> Result<(u32, &'a [u8])> { - let field = fields - .iter() - .find(|field| tags.contains(&field.tag)) - .ok_or(Error::Peer("peer operation is missing"))?; - let range = field - .payload - .clone() - .ok_or(Error::Peer("peer operation payload is missing"))?; - Ok((field.tag, &input[range])) -} - -pub(super) fn validate_operation(tag: u32, payload: &[u8]) -> Result<()> { - let (kind, required, operation_tags): (MessageKind, &[u32], &[u32]) = match tag { - 10 => (MessageKind::MutationRequest, &[1, 2, 4], &[10]), - 11 => (MessageKind::ReadRequest, &[1], &[10, 15, 16, 17, 18, 19]), - 12 => (MessageKind::ResolveRequest, &[1, 2, 3, 4], &[]), - 13 => (MessageKind::EffectRequest, &[1, 2, 3], &[10]), - 14 => (MessageKind::EffectResolveRequest, &[1, 2, 3, 4], &[]), - 15 => (MessageKind::MigrationRequest, &[1, 2, 3, 4, 5, 6], &[]), - _ => return Err(Error::Peer("peer operation is not implemented")), - }; - let fields = validate_message(payload, kind)?; - require_fields(&fields, required)?; - if !operation_tags.is_empty() - && !fields - .iter() - .any(|field| operation_tags.contains(&field.tag)) - { - return Err(Error::Peer("typed peer operation is not implemented")); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/peer/tests.rs b/crates/crab-cell-runtime/src/peer/tests.rs deleted file mode 100644 index bcbb1504c..000000000 --- a/crates/crab-cell-runtime/src/peer/tests.rs +++ /dev/null @@ -1,536 +0,0 @@ -use super::*; - -const NOW_MS: i64 = 1_000_000; - -fn signer() -> PeerSigner { - PeerSigner::new( - SessionId::from_bytes([1; 16]), - Digest::from_bytes([2; 32]), - SigningKey::from_bytes(&[3; 32]), - ) -} - -#[test] -fn signed_peer_request_carries_maximum_kv_value() { - let input = crate::primitives::kv::KvAtomicRequest { - scope: b"scope".to_vec(), - checks: Vec::new(), - mutations: vec![crate::primitives::kv::KvMutation::Put { - key: b"key".to_vec(), - value: vec![7; 4 * 1024 * 1024], - expires_at_ms: None, - }], - }; - let mut encoder = - crate::codec::BoundedEncoder::new(crate::codec::MAX_WIRE_BYTES as u32).unwrap(); - crate::codec::WireValue::encode(&input, &mut encoder).unwrap(); - let mut request = mutation(); - let Some(wire::mutation_request::Operation::CellCommand(command)) = &mut request.operation - else { - panic!("expected Cell command"); - }; - command.input = encoder.finish(); - let signing = signer(); - let encoded = signing - .sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::Mutate(request), - ) - .unwrap(); - let verified = verifier(&signing).verify(&encoded, NOW_MS + 1_000).unwrap(); - assert_eq!(verified.operation_tag(), 10); -} - -#[test] -fn short_operation_deadline_survives_peer_transit_without_extending_identity() { - let expires_at_ms = NOW_MS + 5_000; - let (authorization_expires_at_ms, remaining_ms) = - transport::peer_time_budget(NOW_MS, expires_at_ms).unwrap(); - assert_eq!(remaining_ms, 5_000); - let mut request = mutation(); - request.identity.as_mut().unwrap().expires_at_ms = expires_at_ms; - request.timeout_ms = remaining_ms; - let signer = signer(); - let encoded = signer - .sign( - principal(), - NOW_MS, - authorization_expires_at_ms, - remaining_ms, - PeerOperation::Mutate(request), - ) - .unwrap(); - let verifier = verifier(&signer); - assert!(verifier.verify(&encoded, NOW_MS + 1_000).is_ok()); - assert!(verifier.verify(&encoded, expires_at_ms).is_err()); -} - -fn principal() -> PeerPrincipal { - PeerPrincipal { - issuer: "https://identity.example".into(), - subject: "alice".into(), - actions: vec!["repository.issue.create".into()], - } -} - -fn target() -> wire::Target { - wire::Target { - tenant_id: vec![4; 16], - application_id: vec![5; 16], - namespace_id: vec![6; 16], - partition: b"repository-42".to_vec(), - } -} - -fn mutation() -> wire::MutationRequest { - wire::MutationRequest { - expected: Some(wire::CellDescription { - cell_id: vec![9; 32], - incarnation: vec![8; 16], - code: vec![10; 32], - schema: 1, - }), - target: Some(target()), - identity: Some(wire::MutationIdentity { - request_id: vec![7; 16], - incarnation: vec![8; 16], - issued_at_ms: NOW_MS, - expires_at_ms: NOW_MS + 60_000, - }), - timeout_ms: 30_000, - operation: Some(wire::mutation_request::Operation::CellCommand( - wire::CellCommand { - command_id: 7, - codec_version: 1, - input: b"command-input".to_vec(), - }, - )), - } -} - -#[test] -fn expected_contract_is_required_and_covered_by_the_signature() { - let signer = signer(); - let sign = |request| { - signer.sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::Mutate(request), - ) - }; - let mut missing = mutation(); - missing.expected = None; - assert!(sign(missing).is_err()); - - let encoded = sign(mutation()).unwrap(); - let mut decoded = wire::PeerRequest::decode(encoded.as_slice()).unwrap(); - assert!( - verifier(&signer) - .verify(&decoded.encode_to_vec(), NOW_MS) - .is_ok() - ); - let Some(wire::peer_request::Operation::Mutate(request)) = &mut decoded.operation else { - panic!("expected a mutation"); - }; - request.expected.as_mut().unwrap().schema += 1; - assert!( - verifier(&signer) - .verify(&decoded.encode_to_vec(), NOW_MS) - .is_err() - ); -} - -fn effect_identity() -> wire::EffectIdentity { - let source_cell = crate::CellId::from_bytes([9; 32]); - let source_incarnation = IncarnationId::from_bytes([10; 16]); - let source_sequence = 3; - let ordinal = 2; - wire::EffectIdentity { - effect_id: crate::primitives::effects::effect_id( - source_cell, - source_incarnation, - source_sequence, - ordinal, - ) - .to_vec(), - source_cell: source_cell.as_bytes().to_vec(), - source_incarnation: source_incarnation.as_bytes().to_vec(), - source_sequence, - ordinal, - expires_at_ms: NOW_MS + 5 * 60_000, - } -} - -fn effect() -> wire::EffectRequest { - wire::EffectRequest { - target: Some(target()), - destination_incarnation: vec![8; 16], - identity: Some(effect_identity()), - operation: Some(wire::effect_request::Operation::CellCommand( - wire::CellCommand { - command_id: 7, - codec_version: 1, - input: b"effect-input".to_vec(), - }, - )), - } -} - -fn migration() -> wire::MigrationRequest { - wire::MigrationRequest { - target: Some(target()), - incarnation: vec![8; 16], - from_code: vec![9; 32], - from_schema: 1, - to_code: vec![10; 32], - to_schema: 2, - } -} - -fn verifier(signer: &PeerSigner) -> PeerVerifier { - PeerVerifier::new( - SessionId::from_bytes([1; 16]), - Digest::from_bytes([2; 32]), - signer.verifying_key(), - ) -} - -#[test] -fn signed_request_verifies_and_forward_preserves_payload() { - let signer = signer(); - let encoded = signer - .sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::Mutate(mutation()), - ) - .unwrap(); - assert_eq!( - UnverifiedPeerRequest::decode(&encoded).unwrap().session(), - SessionId::from_bytes([1; 16]) - ); - let verifier = verifier(&signer); - let verified = verifier.verify(&encoded, NOW_MS + 1_000).unwrap(); - assert_eq!(verified.hop_count(), 1); - assert_eq!(verified.remaining_ms(), 30_000); - assert!(verified.permits("repository.issue.create")); - assert_eq!(verified.target().partition(), b"repository-42"); - - let forwarded = verified.forward(20_000).unwrap(); - let forwarded = verifier.verify(&forwarded, NOW_MS + 2_000).unwrap(); - assert_eq!(forwarded.hop_count(), 2); - assert_eq!(forwarded.remaining_ms(), 20_000); - assert!(forwarded.forward(10_000).is_err()); -} - -#[test] -fn signed_effect_delivery_and_resolve_bind_derived_identity() { - let signer = signer(); - let delivery = effect(); - let encoded = signer - .sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::DeliverEffect(delivery.clone()), - ) - .unwrap(); - let verified = verifier(&signer).verify(&encoded, NOW_MS + 1_000).unwrap(); - assert_eq!(verified.operation_tag(), 13); - assert!(matches!( - verified.operation(), - Some(wire::peer_request::Operation::DeliverEffect(_)) - )); - - let digest = crate::primitives::effects::effect_operation_digest( - verified.target().cell_id(), - <[u8; 32]>::try_from(effect_identity().effect_id).unwrap(), - &delivery.encode_to_vec(), - ); - let resolve = wire::EffectResolveRequest { - target: delivery.target, - destination_incarnation: delivery.destination_incarnation, - identity: delivery.identity, - operation_digest: digest.as_bytes().to_vec(), - }; - let encoded = signer - .sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::ResolveEffect(resolve), - ) - .unwrap(); - assert_eq!( - verifier(&signer) - .verify(&encoded, NOW_MS + 1_000) - .unwrap() - .operation_tag(), - 14 - ); - - let mut invalid = effect(); - invalid.identity.as_mut().unwrap().effect_id[0] ^= 1; - assert!( - signer - .sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::DeliverEffect(invalid), - ) - .is_err() - ); -} - -#[test] -fn signed_migration_binds_source_and_successor_versions() { - let signer = signer(); - let encoded = signer - .sign( - PeerPrincipal { - issuer: "crab-runtime:test".into(), - subject: "release-operator".into(), - actions: vec!["cell.release.migrate".into()], - }, - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::Migrate(migration()), - ) - .unwrap(); - let verified = verifier(&signer).verify(&encoded, NOW_MS + 1_000).unwrap(); - assert_eq!(verified.operation_tag(), 15); - assert!(matches!( - verified.operation(), - Some(wire::peer_request::Operation::Migrate(request)) - if request.from_schema == 1 && request.to_schema == 2 - )); - - let mut unchanged = migration(); - unchanged.to_code = unchanged.from_code.clone(); - unchanged.to_schema = unchanged.from_schema; - assert!( - signer - .sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::Migrate(unchanged), - ) - .is_err() - ); -} - -#[test] -fn payload_tampering_fails_before_dispatch() { - let signer = signer(); - let mut encoded = signer - .sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::Mutate(mutation()), - ) - .unwrap(); - let last = encoded.last_mut().unwrap(); - *last ^= 1; - assert!(verifier(&signer).verify(&encoded, NOW_MS + 1_000).is_err()); -} - -#[test] -fn unknown_and_duplicate_fields_are_rejected() { - let signer = signer(); - let encoded = signer - .sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::Mutate(mutation()), - ) - .unwrap(); - let verifier = verifier(&signer); - - let mut unknown = encoded.clone(); - encode_varint_field(&mut unknown, 99, 1); - assert!(verifier.verify(&unknown, NOW_MS + 1_000).is_err()); - - let mut duplicate = encoded; - encode_varint_field(&mut duplicate, 1, 1); - assert!(verifier.verify(&duplicate, NOW_MS + 1_000).is_err()); -} - -#[test] -fn primitive_specific_wire_fields_are_rejected() { - let mut payload = mutation().encode_to_vec(); - encode_bytes_field(&mut payload, 11, b"unsupported-primitive").unwrap(); - assert!(super::protobuf::validate_operation(10, &payload).is_err()); -} - -#[test] -fn reordered_protobuf_payload_retains_exact_signature_binding() { - let signer = signer(); - let mutation = mutation(); - let mut payload = Vec::new(); - let command = match mutation.operation.as_ref() { - Some(wire::mutation_request::Operation::CellCommand(command)) => command.encode_to_vec(), - None => unreachable!(), - }; - encode_bytes_field(&mut payload, 10, &command).unwrap(); - encode_bytes_field( - &mut payload, - 4, - &mutation.expected.as_ref().unwrap().encode_to_vec(), - ) - .unwrap(); - encode_bytes_field( - &mut payload, - 2, - &mutation.identity.as_ref().unwrap().encode_to_vec(), - ) - .unwrap(); - encode_bytes_field( - &mut payload, - 1, - &mutation.target.as_ref().unwrap().encode_to_vec(), - ) - .unwrap(); - encode_varint_field(&mut payload, 3, u64::from(mutation.timeout_ms)); - - let digest = blake3::hash(&payload); - let principal = principal(); - let mut authorization = wire::PeerAuthorization { - origin_session: vec![1; 16], - principal_issuer: principal.issuer, - principal_subject: principal.subject, - actions: principal.actions, - release_digest: vec![2; 32], - issued_at_ms: NOW_MS, - expires_at_ms: NOW_MS + 60_000, - payload_digest: digest.as_bytes().to_vec(), - signature: Vec::new(), - }; - authorization.signature = signer - .key - .sign(&signing_bytes(10, &authorization).unwrap()) - .to_bytes() - .to_vec(); - let encoded = encode_request(authorization, 1, 30_000, 10, &payload).unwrap(); - let verified = verifier(&signer).verify(&encoded, NOW_MS + 1_000).unwrap(); - assert_eq!(verified.operation_tag(), 10); -} - -#[test] -fn unsorted_actions_and_expired_authorization_are_rejected() { - let signer = signer(); - let mut invalid = principal(); - invalid.actions = vec!["z".into(), "a".into()]; - assert!( - signer - .sign( - invalid, - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::Mutate(mutation()), - ) - .is_err() - ); - - let encoded = signer - .sign( - principal(), - NOW_MS, - NOW_MS + 60_000, - 30_000, - PeerOperation::Mutate(mutation()), - ) - .unwrap(); - assert!(verifier(&signer).verify(&encoded, NOW_MS + 60_000).is_err()); -} - -#[test] -fn reply_codec_rejects_unknown_fields_and_invalid_enums() { - let reply = wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - receipt: None, - result: Some(wire::read_reply::Result::Description( - wire::CellDescription { - cell_id: vec![1; 32], - incarnation: vec![2; 16], - code: vec![3; 32], - schema: 1, - }, - )), - })), - }; - let encoded = encode_peer_reply(&reply).unwrap(); - assert!(matches!( - decode_peer_reply(&encoded).unwrap().outcome, - Some(wire::peer_reply::Outcome::Read(_)) - )); - - let mut unknown = encoded; - encode_varint_field(&mut unknown, 99, 1); - assert!(decode_peer_reply(&unknown).is_err()); - - let invalid = wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Error(wire::Error { - code: 99, - outcome: wire::error::Outcome::Rejected as i32, - message: "invalid".into(), - retry_after_ms: 0, - application_details: Vec::new(), - })), - }; - assert!(encode_peer_reply(&invalid).is_err()); - - let migration = wire::PeerReply { - outcome: Some(wire::peer_reply::Outcome::Migration(wire::MigrationReply { - description: Some(wire::CellDescription { - cell_id: vec![1; 32], - incarnation: vec![2; 16], - code: vec![3; 32], - schema: 2, - }), - })), - }; - assert!(matches!( - decode_peer_reply(&encode_peer_reply(&migration).unwrap()) - .unwrap() - .outcome, - Some(wire::peer_reply::Outcome::Migration(_)) - )); -} - -#[test] -fn replica_behind_error_preserves_both_positions_over_peer_wire() { - let reply = dispatch::error_reply(Error::ReplicaBehind { - observed_sequence: 17, - minimum_sequence: 23, - }); - let decoded = decode_peer_reply(&encode_peer_reply(&reply).unwrap()).unwrap(); - let Some(wire::peer_reply::Outcome::Error(error)) = decoded.outcome else { - panic!("expected replica position error"); - }; - assert!(matches!( - transport::runtime_error(error), - Error::ReplicaBehind { - observed_sequence: 17, - minimum_sequence: 23 - } - )); -} diff --git a/crates/crab-cell-runtime/src/peer/transport.rs b/crates/crab-cell-runtime/src/peer/transport.rs deleted file mode 100644 index ceb9c4110..000000000 --- a/crates/crab-cell-runtime/src/peer/transport.rs +++ /dev/null @@ -1,488 +0,0 @@ -use std::{ - future::Future, - pin::Pin, - sync::Arc, - time::{SystemTime, UNIX_EPOCH}, -}; - -use prost::Message; - -use crate::cell::executor::{MutationIdentity, Resolution, StoredOutcome}; -use crate::client::{CellDescription, Receipt}; -use crate::client::{ - CellTransport, EncodedCommand, EncodedObservation, EncodedQuery, EncodedResolve, -}; -use crate::identity::{CellId, CellTarget, Digest, IncarnationId}; -use crate::node::NodeAdvertisement; -use crate::primitives::effects::EffectClaim; -use crate::registry::MigrationPlan; -use crate::{Error, Result}; - -use super::{PeerOperation, PeerPrincipal, PeerSigner, decode_peer_reply, wire, wire_description}; - -mod convert; -mod replicas; - -pub use replicas::ReplicaPeerClient; - -pub(crate) use convert::runtime_error; -use convert::*; - -const DEFAULT_TIMEOUT_MS: u32 = 30_000; - -/// Sends one authenticated request to the current owner and returns exact reply bytes. -pub trait PeerRoundTrip: Send + Sync + 'static { - /// Sends one authenticated request and returns the exact reply bytes. - fn send( - &self, - target: CellTarget, - request: Vec, - remaining_ms: u32, - ) -> Pin>> + Send + 'static>>; - - /// Sends one already-authenticated request to a live enrolled node. - /// - /// This route carries advisory activation and explicit replica queries. - /// The receiving node must verify its authority and admission before serving. - /// Implementations that only route through the owner may fail closed. - fn send_to_node( - &self, - _target: CellTarget, - _node: NodeAdvertisement, - _request: Vec, - _remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - Box::pin(async { Err(Error::Peer("direct node routing is unavailable")) }) - } -} - -pub(crate) struct PeerClientTransport { - signer: Arc, - principal: PeerPrincipal, - round_trip: Arc, -} - -/// Authenticated private transport for one already-published source effect lease. -#[derive(Clone)] -pub struct EffectPeerClient { - transport: PeerClientTransport, -} - -/// Authenticated private control client for one registry-selected Cell migration. -#[derive(Clone)] -pub struct MigrationPeerClient { - transport: PeerClientTransport, -} - -impl MigrationPeerClient { - /// Creates a migration peer client over a signer, principal, and round trip. - #[must_use] - pub fn new( - signer: Arc, - principal: PeerPrincipal, - round_trip: Arc, - ) -> Self { - Self { - transport: PeerClientTransport::new(signer, principal, round_trip), - } - } - - /// Migrates one exact remote capability without accepting migration SQL on the wire. - pub async fn migrate( - &self, - target: CellTarget, - expected: CellDescription, - plan: MigrationPlan, - now_ms: i64, - ) -> Result { - if expected.cell != target.cell_id() - || expected.code != plan.from_code() - || expected.schema != plan.from_schema() - { - return Err(Error::Registry( - "peer migration plan does not match described Cell", - )); - } - let expires_at_ms = now_ms.saturating_add(60_000); - let operation = PeerOperation::Migrate(wire::MigrationRequest { - target: Some(wire_target(&target)), - incarnation: expected.incarnation.as_bytes().to_vec(), - from_code: plan.from_code().as_bytes().to_vec(), - from_schema: plan.from_schema(), - to_code: plan.to_code().as_bytes().to_vec(), - to_schema: plan.to_schema(), - }); - let reply = match self - .transport - .exchange(target.clone(), now_ms, expires_at_ms, operation) - .await - { - Ok(reply) => reply, - Err(source @ Error::PeerTransportUnknown { .. }) => { - let observed = self.transport.describe(target).await?; - if migrated_description(observed, expected, plan) { - return Ok(observed); - } - return Err(source); - } - Err(error) => return Err(error), - }; - match reply.outcome { - Some(wire::peer_reply::Outcome::Migration(reply)) => { - let observed = runtime_description( - reply - .description - .ok_or(Error::Peer("migration reply description is missing"))?, - )?; - if migrated_description(observed, expected, plan) { - Ok(observed) - } else { - Err(Error::Peer( - "migration reply does not match the requested successor", - )) - } - } - Some(wire::peer_reply::Outcome::Error(error)) => Err(runtime_error(error)), - _ => Err(Error::Peer("unexpected migration reply")), - } - } -} - -impl EffectPeerClient { - /// Creates an effect peer client over a signer, principal, and round trip. - #[must_use] - pub fn new( - signer: Arc, - principal: PeerPrincipal, - round_trip: Arc, - ) -> Self { - Self { - transport: PeerClientTransport::new(signer, principal, round_trip), - } - } - - /// Delivers one exact lease after the caller has validated its published token. - pub async fn deliver(&self, claim: &EffectClaim, now_ms: i64) -> Result { - if claim.lease_until_ms < now_ms.saturating_add(1_000) { - return Err(Error::Command( - "effect lease has insufficient delivery margin", - )); - } - let (mut request, target) = checked_effect_request(claim, now_ms)?; - let expected = self.transport.describe(target.clone()).await?; - if expected.cell != target.cell_id() { - return Err(Error::Fenced); - } - request.destination_incarnation = expected.incarnation.as_bytes().to_vec(); - let expires_at_ms = now_ms.saturating_add(60_000).min(claim.expires_at_ms); - let reply = match self - .transport - .exchange( - target, - now_ms, - expires_at_ms, - PeerOperation::DeliverEffect(request), - ) - .await - { - Err(source @ Error::PeerTransportUnknown { .. }) => { - return Err(Error::EffectOutcomeUnknown { - effect_id: claim.effect_id, - operation_digest: claim.operation_digest, - source: Box::new(source), - }); - } - result => result?, - }; - match reply.outcome { - Some(wire::peer_reply::Outcome::Mutation(reply)) => { - effect_mutation_outcome(reply, expected, claim) - } - Some(wire::peer_reply::Outcome::Error(error)) => { - Err(effect_error(error, claim.effect_id, claim.operation_digest)) - } - _ => Err(Error::Peer("unexpected effect delivery reply")), - } - } - - /// Resolves one ambiguous delivery against the destination inbox. - pub async fn resolve(&self, claim: &EffectClaim, now_ms: i64) -> Result { - let (mut request, target) = checked_effect_request(claim, now_ms)?; - let expected = self.transport.describe(target.clone()).await?; - if expected.cell != target.cell_id() { - return Err(Error::Fenced); - } - request.destination_incarnation = expected.incarnation.as_bytes().to_vec(); - let expires_at_ms = now_ms.saturating_add(60_000).min(claim.expires_at_ms); - let reply = self - .transport - .exchange( - target, - now_ms, - expires_at_ms, - PeerOperation::ResolveEffect(wire::EffectResolveRequest { - target: request.target, - destination_incarnation: request.destination_incarnation, - identity: request.identity, - operation_digest: claim.operation_digest.as_bytes().to_vec(), - }), - ) - .await?; - match reply.outcome { - Some(wire::peer_reply::Outcome::Resolve(reply)) => { - effect_resolution_outcome(reply, expected, claim) - } - Some(wire::peer_reply::Outcome::Error(error)) => { - Err(effect_error(error, claim.effect_id, claim.operation_digest)) - } - _ => Err(Error::Peer("unexpected effect Resolve reply")), - } - } -} - -impl PeerClientTransport { - pub(crate) fn new( - signer: Arc, - principal: PeerPrincipal, - round_trip: Arc, - ) -> Self { - Self { - signer, - principal, - round_trip, - } - } - - async fn exchange( - &self, - target: CellTarget, - now_ms: i64, - expires_at_ms: i64, - operation: PeerOperation, - ) -> Result { - let (authorization_expires_at_ms, remaining_ms) = peer_time_budget(now_ms, expires_at_ms)?; - let request = self.signer.sign( - self.principal.clone(), - now_ms, - authorization_expires_at_ms, - remaining_ms, - operation, - )?; - let reply = self.round_trip.send(target, request, remaining_ms).await?; - decode_peer_reply(&reply) - } -} - -impl CellTransport for PeerClientTransport { - fn describe( - &self, - target: CellTarget, - ) -> Pin> + Send + 'static>> { - let transport = self.clone(); - Box::pin(async move { - let now_ms = unix_time_ms()?; - let reply = transport - .exchange( - target.clone(), - now_ms, - now_ms.saturating_add(60_000), - PeerOperation::Read(wire::ReadRequest { - target: Some(wire_target(&target)), - timeout_ms: DEFAULT_TIMEOUT_MS, - minimum: None, - expected: None, - operation: Some(wire::read_request::Operation::Describe(true)), - }), - ) - .await?; - match reply.outcome { - Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - result: Some(wire::read_reply::Result::Description(description)), - .. - })) => runtime_description(description), - Some(wire::peer_reply::Outcome::Error(error)) => Err(runtime_error(error)), - _ => Err(Error::Peer("unexpected describe reply")), - } - }) - } - - fn command( - &self, - command: EncodedCommand, - ) -> Pin> + Send + 'static>> { - let transport = self.clone(); - Box::pin(async move { - let expires_at_ms = command - .now_ms - .saturating_add(60_000) - .min(command.identity.expires_at_ms); - let reply = match transport - .exchange( - command.target.clone(), - command.now_ms, - expires_at_ms, - PeerOperation::Mutate(wire::MutationRequest { - target: Some(wire_target(&command.target)), - expected: Some(wire_description(command.expected)), - identity: Some(wire_identity( - command.identity, - command.expected.incarnation, - )), - timeout_ms: remaining_ms(command.now_ms, expires_at_ms)?, - operation: Some(wire::mutation_request::Operation::CellCommand( - wire::CellCommand { - command_id: command.operation_id, - codec_version: command.codec_version, - input: command.input.clone(), - }, - )), - }), - ) - .await - { - Err(source @ Error::PeerTransportUnknown { .. }) => { - return Err(Error::OutcomeUnknown { - request_id: command.identity.request_id, - operation_digest: command.operation_digest, - source: Box::new(source), - }); - } - result => result?, - }; - match reply.outcome { - Some(wire::peer_reply::Outcome::Mutation(reply)) => mutation_outcome( - reply, - command.expected, - command.identity, - command.operation_digest, - ), - Some(wire::peer_reply::Outcome::Error(error)) => Err(command_error( - error, - command.identity, - command.operation_digest, - )), - _ => Err(Error::Peer("unexpected mutation reply")), - } - }) - } - - fn query( - &self, - query: EncodedQuery, - ) -> Pin> + Send + 'static>> { - let transport = self.clone(); - Box::pin(async move { - let expires_at_ms = query.now_ms.saturating_add(60_000); - let reply = transport - .exchange( - query.target.clone(), - query.now_ms, - expires_at_ms, - PeerOperation::Read(wire::ReadRequest { - target: Some(wire_target(&query.target)), - expected: Some(wire_description(query.expected)), - timeout_ms: DEFAULT_TIMEOUT_MS, - minimum: query.minimum.map(wire_receipt), - operation: Some(wire::read_request::Operation::CellQuery( - wire::CellQuery { - query_id: query.operation_id, - codec_version: query.codec_version, - input: query.input, - }, - )), - }), - ) - .await?; - match reply.outcome { - Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - receipt: Some(receipt), - result: Some(wire::read_reply::Result::CommandOutput(output)), - })) => Ok(EncodedObservation { - output, - receipt: checked_receipt(receipt, query.expected)?, - }), - Some(wire::peer_reply::Outcome::Error(error)) => Err(runtime_error(error)), - _ => Err(Error::Peer("unexpected query reply")), - } - }) - } - - fn resolve( - &self, - resolve: EncodedResolve, - ) -> Pin> + Send + 'static>> { - let transport = self.clone(); - Box::pin(async move { - let expires_at_ms = resolve - .now_ms - .saturating_add(60_000) - .min(resolve.identity.expires_at_ms); - let reply = transport - .exchange( - resolve.target.clone(), - resolve.now_ms, - expires_at_ms, - PeerOperation::Resolve(wire::ResolveRequest { - target: Some(wire_target(&resolve.target)), - expected: Some(wire_description(resolve.expected)), - identity: Some(wire_identity( - resolve.identity, - resolve.expected.incarnation, - )), - operation_digest: resolve.operation_digest.as_bytes().to_vec(), - }), - ) - .await?; - match reply.outcome { - Some(wire::peer_reply::Outcome::Resolve(reply)) => resolution_outcome( - reply, - resolve.expected, - resolve.identity, - resolve.operation_digest, - ), - Some(wire::peer_reply::Outcome::Error(error)) => Err(command_error( - error, - resolve.identity, - resolve.operation_digest, - )), - _ => Err(Error::Peer("unexpected resolve reply")), - } - }) - } -} - -impl Clone for PeerClientTransport { - fn clone(&self) -> Self { - Self { - signer: Arc::clone(&self.signer), - principal: self.principal.clone(), - round_trip: Arc::clone(&self.round_trip), - } - } -} - -fn remaining_ms(now_ms: i64, expires_at_ms: i64) -> Result { - let remaining = expires_at_ms - .checked_sub(now_ms) - .filter(|remaining| *remaining > 0) - .ok_or(Error::Peer("peer request already expired"))?; - u32::try_from(remaining.min(i64::from(DEFAULT_TIMEOUT_MS))) - .map_err(|_| Error::Peer("peer deadline overflow")) -} - -pub(super) fn peer_time_budget(now_ms: i64, operation_expires_at_ms: i64) -> Result<(i64, u32)> { - let remaining_ms = remaining_ms(now_ms, operation_expires_at_ms)?; - // Authorization must outlive the operation budget so transit does not - // invalidate a signed request before the peer can verify it. - let authorization_expires_at_ms = now_ms - .checked_add(60_000) - .ok_or(Error::Peer("peer authorization deadline overflow"))?; - Ok((authorization_expires_at_ms, remaining_ms)) -} - -fn unix_time_ms() -> Result { - let duration = SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_err(|_| Error::Peer("system clock precedes Unix epoch"))?; - i64::try_from(duration.as_millis()).map_err(|_| Error::Peer("system clock overflow")) -} diff --git a/crates/crab-cell-runtime/src/peer/transport/convert.rs b/crates/crab-cell-runtime/src/peer/transport/convert.rs deleted file mode 100644 index 4dad8ea85..000000000 --- a/crates/crab-cell-runtime/src/peer/transport/convert.rs +++ /dev/null @@ -1,312 +0,0 @@ -//! Reply, receipt, and description conversion between runtime and wire types. - -use super::*; - -pub(super) fn migrated_description( - observed: CellDescription, - expected: CellDescription, - plan: MigrationPlan, -) -> bool { - observed.cell == expected.cell - && observed.incarnation == expected.incarnation - && observed.code == plan.to_code() - && observed.schema >= plan.to_schema() -} - -pub(super) fn checked_effect_request( - claim: &EffectClaim, - now_ms: i64, -) -> Result<(wire::EffectRequest, CellTarget)> { - if claim.attempt == 0 - || claim.token.iter().all(|byte| *byte == 0) - || claim.operation.is_empty() - || claim.expires_at_ms <= now_ms - { - return Err(Error::Command("invalid effect claim for delivery")); - } - let request = wire::EffectRequest::decode(claim.operation.as_slice())?; - if request.encode_to_vec() != claim.operation { - return Err(Error::Peer("stored effect request is not canonical")); - } - if !request.destination_incarnation.is_empty() { - return Err(Error::Peer( - "stored effect request pins a destination incarnation", - )); - } - let mut validated = request.clone(); - validated.destination_incarnation = vec![1; 16]; - PeerOperation::DeliverEffect(validated).validate(now_ms)?; - let target = runtime_target( - request - .target - .as_ref() - .ok_or(Error::Peer("effect target is missing"))?, - )?; - if target.cell_id() != claim.destination - || crate::primitives::effects::effect_operation_digest( - target.cell_id(), - claim.effect_id, - &claim.operation, - ) != claim.operation_digest - { - return Err(Error::Command("effect claim target or digest changed")); - } - let identity = request - .identity - .as_ref() - .ok_or(Error::Peer("effect identity is missing"))?; - if identity.effect_id.as_slice() != claim.effect_id - || identity.source_sequence != claim.created_sequence - || identity.expires_at_ms != claim.expires_at_ms - { - return Err(Error::Command("effect claim source identity changed")); - } - Ok((request, target)) -} - -pub(super) fn runtime_target(value: &wire::Target) -> Result { - CellTarget::new( - crate::TenantId::try_from(value.tenant_id.as_slice())?, - crate::ApplicationId::try_from(value.application_id.as_slice())?, - crate::NamespaceId::try_from(value.namespace_id.as_slice())?, - &value.partition, - ) -} - -pub(super) fn effect_mutation_outcome( - reply: wire::MutationReply, - expected: CellDescription, - claim: &EffectClaim, -) -> Result { - let receipt = checked_receipt( - reply - .receipt - .ok_or(Error::Peer("effect reply receipt is missing"))?, - expected, - )?; - match reply.outcome { - Some(wire::mutation_reply::Outcome::Result(wire::MutationResult { - result: Some(wire::mutation_result::Result::CommandOutput(result)), - })) => Ok(StoredOutcome::Success { - result, - commit_sequence: receipt.commit_sequence, - }), - Some(wire::mutation_reply::Outcome::Error(error)) - if error.code == wire::error::Code::PreconditionFailed as i32 - && error.outcome == wire::error::Outcome::Rejected as i32 => - { - Ok(StoredOutcome::Rejected { - result: error.application_details, - commit_sequence: receipt.commit_sequence, - }) - } - Some(wire::mutation_reply::Outcome::Error(error)) => { - Err(effect_error(error, claim.effect_id, claim.operation_digest)) - } - _ => Err(Error::Peer("unexpected effect mutation result")), - } -} - -pub(super) fn effect_resolution_outcome( - reply: wire::ResolveReply, - expected: CellDescription, - claim: &EffectClaim, -) -> Result { - match wire::resolve_reply::State::try_from(reply.state) - .map_err(|_| Error::Peer("unknown effect Resolve state"))? - { - wire::resolve_reply::State::Committed | wire::resolve_reply::State::Rejected => { - Ok(Resolution::Committed(effect_mutation_outcome( - reply - .reply - .ok_or(Error::Peer("resolved effect reply is missing"))?, - expected, - claim, - )?)) - } - wire::resolve_reply::State::Absent => Ok(Resolution::Absent), - wire::resolve_reply::State::Unknown => Ok(Resolution::Unknown), - wire::resolve_reply::State::Expired => Ok(Resolution::Expired), - wire::resolve_reply::State::Invalid => Err(Error::Peer("invalid effect Resolve state")), - } -} - -pub(super) fn effect_error(error: wire::Error, effect_id: [u8; 32], digest: Digest) -> Error { - if error.code == wire::error::Code::OutcomeUnknown as i32 - || error.outcome == wire::error::Outcome::Unknown as i32 - { - return Error::EffectOutcomeUnknown { - effect_id, - operation_digest: digest, - source: Box::new(runtime_error(error)), - }; - } - if error.code == wire::error::Code::RequestExpired as i32 { - return Error::EffectExpired; - } - runtime_error(error) -} - -pub(super) fn mutation_outcome( - reply: wire::MutationReply, - expected: CellDescription, - identity: MutationIdentity, - digest: Digest, -) -> Result { - let receipt = checked_receipt( - reply - .receipt - .ok_or(Error::Peer("mutation reply receipt is missing"))?, - expected, - )?; - match reply.outcome { - Some(wire::mutation_reply::Outcome::Result(wire::MutationResult { - result: Some(wire::mutation_result::Result::CommandOutput(result)), - })) => Ok(StoredOutcome::Success { - result, - commit_sequence: receipt.commit_sequence, - }), - Some(wire::mutation_reply::Outcome::Error(error)) - if error.code == wire::error::Code::PreconditionFailed as i32 - && error.outcome == wire::error::Outcome::Rejected as i32 => - { - Ok(StoredOutcome::Rejected { - result: error.application_details, - commit_sequence: receipt.commit_sequence, - }) - } - Some(wire::mutation_reply::Outcome::Error(error)) => { - Err(command_error(error, identity, digest)) - } - _ => Err(Error::Peer("unexpected mutation result")), - } -} - -pub(super) fn resolution_outcome( - reply: wire::ResolveReply, - expected: CellDescription, - identity: MutationIdentity, - digest: Digest, -) -> Result { - match wire::resolve_reply::State::try_from(reply.state) - .map_err(|_| Error::Peer("unknown resolve state"))? - { - wire::resolve_reply::State::Committed | wire::resolve_reply::State::Rejected => { - Ok(Resolution::Committed(mutation_outcome( - reply - .reply - .ok_or(Error::Peer("resolved mutation reply is missing"))?, - expected, - identity, - digest, - )?)) - } - wire::resolve_reply::State::Absent => Ok(Resolution::Absent), - wire::resolve_reply::State::Unknown => Ok(Resolution::Unknown), - wire::resolve_reply::State::Expired => Ok(Resolution::Expired), - wire::resolve_reply::State::Invalid => Err(Error::Peer("invalid resolve state")), - } -} - -pub(super) fn command_error( - error: wire::Error, - identity: MutationIdentity, - digest: Digest, -) -> Error { - if error.code == wire::error::Code::OutcomeUnknown as i32 - || error.outcome == wire::error::Outcome::Unknown as i32 - { - return Error::OutcomeUnknown { - request_id: identity.request_id, - operation_digest: digest, - source: Box::new(runtime_error(error)), - }; - } - runtime_error(error) -} - -pub(crate) fn runtime_error(error: wire::Error) -> Error { - match wire::error::Code::try_from(error.code) { - Ok(wire::error::Code::PermissionDenied) => { - Error::PeerAuthorization("remote peer denied the principal") - } - Ok(wire::error::Code::RequestIdConflict) => Error::RequestConflict, - Ok(wire::error::Code::ResourceExhausted) => Error::Capacity("remote Cell owner"), - Ok(wire::error::Code::SchemaIncompatible) => { - Error::Registry("remote owner rejected the compiled operation") - } - Ok(wire::error::Code::ReplicaBehind) => match error.application_details.as_slice() { - [a, b, c, d, e, f, g, h, i, j, k, l, m, n, o, p] => Error::ReplicaBehind { - observed_sequence: u64::from_be_bytes([*a, *b, *c, *d, *e, *f, *g, *h]), - minimum_sequence: u64::from_be_bytes([*i, *j, *k, *l, *m, *n, *o, *p]), - }, - _ => Error::Peer("remote read replica returned invalid position details"), - }, - Ok(wire::error::Code::ReplicaUnavailable) => Error::ReplicaUnavailable, - Ok(wire::error::Code::Unavailable | wire::error::Code::OutcomeUnknown) => { - Error::CellNotActive - } - Ok( - wire::error::Code::PreconditionFailed - | wire::error::Code::LeaseLost - | wire::error::Code::RequestExpired, - ) => Error::Fenced, - Ok( - wire::error::Code::InvalidArgument - | wire::error::Code::NotFound - | wire::error::Code::Internal - | wire::error::Code::Invalid, - ) - | Err(_) => Error::Peer("remote peer rejected the request"), - } -} - -pub(super) fn runtime_description(value: wire::CellDescription) -> Result { - Ok(CellDescription { - cell: CellId::try_from(value.cell_id.as_slice())?, - incarnation: IncarnationId::try_from(value.incarnation.as_slice())?, - code: Digest::try_from(value.code.as_slice())?, - schema: value.schema, - }) -} - -pub(super) fn checked_receipt(value: wire::Receipt, expected: CellDescription) -> Result { - let receipt = Receipt { - cell: CellId::try_from(value.cell_id.as_slice())?, - incarnation: IncarnationId::try_from(value.incarnation.as_slice())?, - commit_sequence: value.commit_sequence, - }; - if receipt.cell != expected.cell || receipt.incarnation != expected.incarnation { - return Err(Error::Peer("peer receipt does not match described Cell")); - } - Ok(receipt) -} - -pub(super) fn wire_target(target: &CellTarget) -> wire::Target { - wire::Target { - tenant_id: target.tenant().as_bytes().to_vec(), - application_id: target.application().as_bytes().to_vec(), - namespace_id: target.namespace().as_bytes().to_vec(), - partition: target.partition().to_vec(), - } -} - -pub(super) fn wire_identity( - identity: MutationIdentity, - incarnation: IncarnationId, -) -> wire::MutationIdentity { - wire::MutationIdentity { - request_id: identity.request_id.as_bytes().to_vec(), - incarnation: incarnation.as_bytes().to_vec(), - issued_at_ms: identity.issued_at_ms, - expires_at_ms: identity.expires_at_ms, - } -} - -pub(super) fn wire_receipt(receipt: Receipt) -> wire::Receipt { - wire::Receipt { - cell_id: receipt.cell.as_bytes().to_vec(), - incarnation: receipt.incarnation.as_bytes().to_vec(), - commit_sequence: receipt.commit_sequence, - } -} diff --git a/crates/crab-cell-runtime/src/peer/transport/replicas.rs b/crates/crab-cell-runtime/src/peer/transport/replicas.rs deleted file mode 100644 index f65513fdf..000000000 --- a/crates/crab-cell-runtime/src/peer/transport/replicas.rs +++ /dev/null @@ -1,252 +0,0 @@ -//! Typed queries and lifecycle hints for selected read-only replicas. - -use super::*; -use crate::client::Observed; -use crate::codec::{decode_wire, encode_wire}; -use crate::node::NodeDirectory; -use crate::registry::{Query, Registry}; - -const REPLICA_QUERY_TIMEOUT_MS: u32 = 5_000; - -/// Explicit, typed read client for one selected read-only Cell snapshot. -#[derive(Clone)] -pub struct ReplicaPeerClient { - registry: Arc, - transport: PeerClientTransport, -} - -impl ReplicaPeerClient { - /// Binds typed replica reads to the compiled registry and authenticated peer transport. - #[must_use] - pub fn new( - registry: Arc, - signer: Arc, - principal: PeerPrincipal, - round_trip: Arc, - ) -> Self { - Self { - registry, - transport: PeerClientTransport::new(signer, principal, round_trip), - } - } - - /// Queries one selected node without falling back to the current owner. - pub async fn query( - &self, - target: &CellTarget, - node: NodeAdvertisement, - expected: CellDescription, - minimum: Option, - input: Q::Input, - ) -> Result> { - if expected.cell != target.cell_id() - || minimum.is_some_and(|receipt| { - receipt.cell != expected.cell || receipt.incarnation != expected.incarnation - }) - { - return Err(Error::Fenced); - } - let operation = self.registry.query_contract::(target.namespace())?; - crate::client::validate_description(&self.registry, Q::MODULE, expected, operation)?; - let input = encode_wire(&input, operation.input_limit)?; - let observation = self - .query_encoded( - node, - EncodedQuery { - target: target.clone(), - expected, - minimum, - now_ms: unix_time_ms()?, - module: Q::MODULE, - operation_id: Q::ID, - codec_version: Q::CODEC_VERSION, - input, - input_limit: operation.input_limit, - output_limit: operation.output_limit, - }, - tokio::time::Instant::now() - + std::time::Duration::from_millis(u64::from(REPLICA_QUERY_TIMEOUT_MS)), - ) - .await?; - Ok(Observed { - output: decode_wire(&observation.output, operation.output_limit)?, - receipt: observation.receipt, - }) - } - - pub(crate) fn registry(&self) -> &Registry { - &self.registry - } - - /// Requests admission on a selected boot session and checks its exact Cell lifetime. - /// - /// The caller supplies an activation-authorized principal. Discovery is - /// revalidated just before dispatch so queued hints cannot target a later boot. - /// A pending hint rechecks the session at its observed expiry and ends if - /// it is no longer live. Renewals retain the original request and budget. - pub async fn activate( - &self, - target: &CellTarget, - directory: &NodeDirectory, - node: NodeAdvertisement, - expected: CellDescription, - ) -> Result { - if expected.cell != target.cell_id() { - return Err(Error::Fenced); - } - let observed = directory - .load(node.session(), unix_time_ms()?) - .await? - .ok_or(Error::CellNotActive)?; - let node = observed.advertisement().clone(); - let session = node.session(); - let mut expires_at_ms = node.expires_at_ms(); - let expired = async { - loop { - let remaining_ms = expires_at_ms.saturating_sub(unix_time_ms()?).max(0) as u64; - tokio::time::sleep(std::time::Duration::from_millis(remaining_ms)).await; - let current = directory - .load(session, unix_time_ms()?) - .await? - .ok_or(Error::CellNotActive)?; - expires_at_ms = current.advertisement().expires_at_ms(); - } - }; - // Recheck a pending peer at lease expiry so a dead connection does not - // hold recruitment until the transport deadline. Renewals retain slow - // valid opens; a ready reply can interrupt a stalled directory read. - let (receipt, ready) = tokio::select! { - result = self.readiness( - target, - node, - expected, - wire::read_request::Operation::ReplicaActivate(true), - DEFAULT_TIMEOUT_MS, - ) => result?, - result = expired => return result, - }; - if !ready { - return Err(Error::ReplicaUnavailable); - } - Ok(receipt) - } - - /// Probes a selected snapshot without activating it or granting ownership. - pub async fn status( - &self, - target: &CellTarget, - node: NodeAdvertisement, - expected: CellDescription, - ) -> Result<(Receipt, bool)> { - self.readiness( - target, - node, - expected, - wire::read_request::Operation::ReplicaStatus(true), - REPLICA_QUERY_TIMEOUT_MS, - ) - .await - } - - async fn readiness( - &self, - target: &CellTarget, - node: NodeAdvertisement, - expected: CellDescription, - operation: wire::read_request::Operation, - timeout_ms: u32, - ) -> Result<(Receipt, bool)> { - if expected.cell != target.cell_id() { - return Err(Error::Fenced); - } - let now_ms = unix_time_ms()?; - let (expires_at_ms, remaining_ms) = - peer_time_budget(now_ms, now_ms.saturating_add(i64::from(timeout_ms)))?; - let request = self.transport.signer.sign( - self.transport.principal.clone(), - now_ms, - expires_at_ms, - remaining_ms, - PeerOperation::Read(wire::ReadRequest { - target: Some(wire_target(target)), - timeout_ms: remaining_ms, - minimum: None, - expected: None, - operation: Some(operation), - }), - )?; - let reply = self - .transport - .round_trip - .send_to_node(target.clone(), node, request, remaining_ms) - .await?; - match decode_peer_reply(&reply)?.outcome { - Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - receipt: Some(receipt), - result: Some(wire::read_reply::Result::ReplicaReady(ready)), - })) => Ok((checked_receipt(receipt, expected)?, ready)), - Some(wire::peer_reply::Outcome::Error(error)) => Err(runtime_error(error)), - _ => Err(Error::Peer("unexpected read replica readiness reply")), - } - } - - pub(crate) async fn query_encoded( - &self, - node: NodeAdvertisement, - query: EncodedQuery, - deadline: tokio::time::Instant, - ) -> Result { - let now_ms = unix_time_ms()?; - let remaining = deadline - .checked_duration_since(tokio::time::Instant::now()) - .ok_or(Error::Deadline)?; - let budget_ms = u32::try_from(remaining.as_millis()).unwrap_or(u32::MAX); - if budget_ms == 0 { - return Err(Error::Deadline); - } - let expires_at_ms = now_ms.saturating_add(i64::from(budget_ms)); - let (authorization_expires_at_ms, remaining_ms) = peer_time_budget(now_ms, expires_at_ms)?; - let request = self.transport.signer.sign( - self.transport.principal.clone(), - now_ms, - authorization_expires_at_ms, - remaining_ms, - PeerOperation::Read(wire::ReadRequest { - target: Some(wire_target(&query.target)), - timeout_ms: remaining_ms, - minimum: query.minimum.map(wire_receipt), - expected: Some(wire_description(query.expected)), - operation: Some(wire::read_request::Operation::ReplicaQuery( - wire::CellQuery { - query_id: query.operation_id, - codec_version: query.codec_version, - input: query.input, - }, - )), - }), - )?; - let bytes = self - .transport - .round_trip - .send_to_node(query.target, node, request, remaining_ms) - .await?; - let reply = decode_peer_reply(&bytes)?; - match reply.outcome { - Some(wire::peer_reply::Outcome::Read(wire::ReadReply { - receipt: Some(receipt), - result: Some(wire::read_reply::Result::CommandOutput(output)), - })) => { - let receipt = checked_receipt(receipt, query.expected)?; - if query - .minimum - .is_some_and(|minimum| receipt.commit_sequence < minimum.commit_sequence) - { - return Err(Error::Peer("read replica returned an older receipt")); - } - Ok(EncodedObservation { output, receipt }) - } - Some(wire::peer_reply::Outcome::Error(error)) => Err(runtime_error(error)), - _ => Err(Error::Peer("unexpected read replica reply")), - } - } -} diff --git a/crates/crab-cell-runtime/src/peer/validation.rs b/crates/crab-cell-runtime/src/peer/validation.rs deleted file mode 100644 index 9fbe71ead..000000000 --- a/crates/crab-cell-runtime/src/peer/validation.rs +++ /dev/null @@ -1,472 +0,0 @@ -//! Reply and authorization validation for peer requests. - -use super::*; - -pub(super) fn validate_reply(reply: &wire::PeerReply) -> Result<()> { - match reply.outcome.as_ref() { - Some(wire::peer_reply::Outcome::Mutation(reply)) => validate_mutation_reply(reply), - Some(wire::peer_reply::Outcome::Read(reply)) => validate_read_reply(reply), - Some(wire::peer_reply::Outcome::Resolve(reply)) => validate_resolve_reply(reply), - Some(wire::peer_reply::Outcome::Error(error)) => validate_error(error), - Some(wire::peer_reply::Outcome::Migration(reply)) => validate_description_wire( - reply - .description - .as_ref() - .ok_or(Error::Peer("migration reply description is missing"))?, - ), - None => Err(Error::Peer("peer reply outcome is missing")), - } -} - -pub(super) fn validate_mutation_reply(reply: &wire::MutationReply) -> Result<()> { - validate_receipt( - reply - .receipt - .as_ref() - .ok_or(Error::Peer("mutation reply receipt is missing"))?, - )?; - match reply.outcome.as_ref() { - Some(wire::mutation_reply::Outcome::Result(result)) => match result.result.as_ref() { - Some(wire::mutation_result::Result::CommandOutput(output)) - if output.len() <= MAX_OPERATION_BYTES => - { - Ok(()) - } - Some(wire::mutation_result::Result::CommandOutput(_)) => { - Err(Error::Peer("mutation result exceeds wire limit")) - } - None => Err(Error::Peer("mutation result is missing")), - }, - Some(wire::mutation_reply::Outcome::Error(error)) => validate_error(error), - None => Err(Error::Peer("mutation reply outcome is missing")), - } -} - -pub(super) fn validate_read_reply(reply: &wire::ReadReply) -> Result<()> { - if let Some(receipt) = &reply.receipt { - validate_receipt(receipt)?; - } - match reply.result.as_ref() { - Some(wire::read_reply::Result::Description(description)) => { - if reply.receipt.is_some() - || description.cell_id.len() != 32 - || description.incarnation.len() != 16 - || description.code.len() != 32 - || description.schema == 0 - { - return Err(Error::Peer("invalid peer Cell description")); - } - Ok(()) - } - Some(wire::read_reply::Result::CommandOutput(output)) => { - validate_receipt( - reply - .receipt - .as_ref() - .ok_or(Error::Peer("read reply receipt is missing"))?, - )?; - if output.len() > MAX_OPERATION_BYTES { - return Err(Error::Peer("read result exceeds wire limit")); - } - Ok(()) - } - Some(wire::read_reply::Result::ReplicaReady(_)) => validate_receipt( - reply - .receipt - .as_ref() - .ok_or(Error::Peer("replica-ready receipt is missing"))?, - ), - Some(wire::read_reply::Result::ReplicaReconciled(true)) if reply.receipt.is_none() => { - Ok(()) - } - Some(wire::read_reply::Result::ReplicaReconciled(_)) => { - Err(Error::Peer("invalid replica reconciliation reply")) - } - Some(wire::read_reply::Result::Error(error)) => validate_error(error), - None => Err(Error::Peer("read reply result is missing")), - } -} - -pub(super) fn validate_resolve_reply(reply: &wire::ResolveReply) -> Result<()> { - let state = wire::resolve_reply::State::try_from(reply.state) - .map_err(|_| Error::Peer("unknown resolve state"))?; - match state { - wire::resolve_reply::State::Committed | wire::resolve_reply::State::Rejected => { - validate_mutation_reply( - reply - .reply - .as_ref() - .ok_or(Error::Peer("resolved mutation reply is missing"))?, - ) - } - wire::resolve_reply::State::Absent - | wire::resolve_reply::State::Unknown - | wire::resolve_reply::State::Expired - if reply.reply.is_none() => - { - Ok(()) - } - wire::resolve_reply::State::Invalid => Err(Error::Peer("invalid resolve state")), - _ => Err(Error::Peer("resolve state and reply disagree")), - } -} - -pub(super) fn validate_receipt(receipt: &wire::Receipt) -> Result<()> { - if receipt.cell_id.len() != 32 || receipt.incarnation.len() != 16 { - return Err(Error::Peer("invalid peer receipt identity")); - } - Ok(()) -} - -pub(super) fn validate_error(error: &wire::Error) -> Result<()> { - let code = wire::error::Code::try_from(error.code) - .map_err(|_| Error::Peer("unknown peer error code"))?; - let outcome = wire::error::Outcome::try_from(error.outcome) - .map_err(|_| Error::Peer("unknown peer error outcome"))?; - if code == wire::error::Code::Invalid - || outcome == wire::error::Outcome::Unspecified - || error.message.is_empty() - || error.message.len() > 2_048 - || error.retry_after_ms > 60_000 - || error.application_details.len() > MAX_OPERATION_BYTES - { - return Err(Error::Peer("invalid peer error bounds")); - } - if code == wire::error::Code::ReplicaBehind && error.application_details.len() != 16 { - return Err(Error::Peer("invalid read replica position details")); - } - Ok(()) -} - -pub(super) fn validate_principal(principal: &PeerPrincipal) -> Result<()> { - if principal.issuer.is_empty() - || principal.subject.is_empty() - || principal.issuer.len() > MAX_PRINCIPAL_BYTES - || principal.subject.len() > MAX_PRINCIPAL_BYTES - || principal.actions.is_empty() - || principal.actions.len() > MAX_ACTIONS - { - return Err(Error::Peer("invalid peer principal bounds")); - } - let mut previous: Option<&str> = None; - for action in &principal.actions { - if action.is_empty() - || action.len() > 128 - || !action.bytes().all(|byte| { - byte.is_ascii_alphanumeric() || matches!(byte, b'.' | b':' | b'_' | b'-') - }) - || previous.is_some_and(|previous| previous >= action.as_str()) - { - return Err(Error::Peer( - "peer actions must be sorted unique identifiers", - )); - } - previous = Some(action); - } - Ok(()) -} - -pub(super) fn validate_authorization( - authorization: &wire::PeerAuthorization, - now_ms: i64, - remaining_ms: u32, -) -> Result<()> { - let principal = PeerPrincipal { - issuer: authorization.principal_issuer.clone(), - subject: authorization.principal_subject.clone(), - actions: authorization.actions.clone(), - }; - validate_principal(&principal)?; - if authorization.origin_session.len() != 16 - || authorization.release_digest.len() != 32 - || authorization.payload_digest.len() != 32 - || authorization.signature.len() != 64 - { - return Err(Error::Peer("invalid peer authorization identity length")); - } - validate_time_bounds( - authorization.issued_at_ms, - authorization.expires_at_ms, - now_ms, - remaining_ms, - ) -} - -pub(super) fn validate_time_bounds( - issued_at_ms: i64, - expires_at_ms: i64, - now_ms: i64, - remaining_ms: u32, -) -> Result<()> { - if issued_at_ms < 0 - || expires_at_ms <= issued_at_ms - || expires_at_ms - issued_at_ms > MAX_AUTH_LIFETIME_MS - || issued_at_ms > now_ms.saturating_add(MAX_CLOCK_SKEW_MS) - || expires_at_ms <= now_ms - || !(1..=60_000).contains(&remaining_ms) - || i64::from(remaining_ms) > expires_at_ms - now_ms - { - return Err(Error::Peer("invalid or expired peer authorization time")); - } - Ok(()) -} - -pub(super) fn operation_target( - operation: Option<&wire::peer_request::Operation>, -) -> Result { - let target = match operation { - Some(wire::peer_request::Operation::Mutate(value)) => value.target.as_ref(), - Some(wire::peer_request::Operation::Read(value)) => value.target.as_ref(), - Some(wire::peer_request::Operation::Resolve(value)) => value.target.as_ref(), - Some(wire::peer_request::Operation::DeliverEffect(value)) => value.target.as_ref(), - Some(wire::peer_request::Operation::ResolveEffect(value)) => value.target.as_ref(), - Some(wire::peer_request::Operation::Migrate(value)) => value.target.as_ref(), - None => return Err(Error::Peer("peer operation is missing")), - } - .ok_or(Error::Peer("peer target is missing"))?; - CellTarget::new( - TenantId::try_from(target.tenant_id.as_slice())?, - ApplicationId::try_from(target.application_id.as_slice())?, - NamespaceId::try_from(target.namespace_id.as_slice())?, - &target.partition, - ) -} - -pub(super) fn validate_decoded_operation( - operation: Option<&wire::peer_request::Operation>, - now_ms: i64, -) -> Result<()> { - match operation { - Some(wire::peer_request::Operation::Mutate(value)) => validate_mutation(value, now_ms), - Some(wire::peer_request::Operation::Read(value)) => validate_read(value), - Some(wire::peer_request::Operation::Resolve(value)) => validate_resolve(value, now_ms), - Some(wire::peer_request::Operation::DeliverEffect(value)) => validate_effect(value, now_ms), - Some(wire::peer_request::Operation::ResolveEffect(value)) => { - validate_effect_resolve(value, now_ms) - } - Some(wire::peer_request::Operation::Migrate(value)) => validate_migration(value), - None => Err(Error::Peer("peer operation is missing")), - } -} - -pub(super) fn validate_mutation(request: &wire::MutationRequest, now_ms: i64) -> Result<()> { - validate_target_wire(request.target.as_ref())?; - validate_expected_description(request.expected.as_ref())?; - validate_timeout(request.timeout_ms)?; - let identity = request - .identity - .as_ref() - .ok_or(Error::Peer("mutation identity is missing"))?; - validate_mutation_identity(identity, now_ms)?; - match request.operation.as_ref() { - Some(wire::mutation_request::Operation::CellCommand(command)) - if command.command_id != 0 && command.codec_version != 0 => - { - Ok(()) - } - Some(wire::mutation_request::Operation::CellCommand(_)) => { - Err(Error::Peer("invalid Cell command identifier")) - } - None => Err(Error::Peer("mutation operation is missing")), - } -} - -pub(super) fn validate_read(request: &wire::ReadRequest) -> Result<()> { - validate_target_wire(request.target.as_ref())?; - validate_timeout(request.timeout_ms)?; - if let Some(receipt) = &request.minimum - && (receipt.cell_id.len() != 32 || receipt.incarnation.len() != 16) - { - return Err(Error::Peer("invalid minimum receipt identity length")); - } - match request.operation.as_ref() { - Some(wire::read_request::Operation::Describe(true)) => Ok(()), - Some(wire::read_request::Operation::Describe(false)) => { - Err(Error::Peer("describe selector must be true")) - } - Some(wire::read_request::Operation::CellQuery(query)) - if query.query_id != 0 && query.codec_version != 0 => - { - validate_expected_description(request.expected.as_ref()) - } - Some(wire::read_request::Operation::ReplicaActivate(true)) if request.minimum.is_none() => { - Ok(()) - } - Some(wire::read_request::Operation::ReplicaReconcile(true)) - if request.minimum.is_none() => - { - Ok(()) - } - Some(wire::read_request::Operation::ReplicaStatus(true)) if request.minimum.is_none() => { - Ok(()) - } - Some(wire::read_request::Operation::ReplicaQuery(query)) - if query.query_id != 0 && query.codec_version != 0 => - { - validate_expected_description(request.expected.as_ref()) - } - Some(wire::read_request::Operation::CellQuery(_)) => { - Err(Error::Peer("invalid Cell query identifier")) - } - Some(wire::read_request::Operation::ReplicaQuery(_)) => { - Err(Error::Peer("invalid replica query identifier")) - } - Some(wire::read_request::Operation::ReplicaActivate(_)) => { - Err(Error::Peer("invalid replica activation request")) - } - Some(wire::read_request::Operation::ReplicaReconcile(_)) => { - Err(Error::Peer("invalid replica reconciliation request")) - } - Some(wire::read_request::Operation::ReplicaStatus(_)) => { - Err(Error::Peer("invalid replica status request")) - } - None => Err(Error::Peer("read operation is missing")), - } -} - -pub(super) fn validate_resolve(request: &wire::ResolveRequest, now_ms: i64) -> Result<()> { - validate_target_wire(request.target.as_ref())?; - validate_expected_description(request.expected.as_ref())?; - let identity = request - .identity - .as_ref() - .ok_or(Error::Peer("resolve identity is missing"))?; - validate_mutation_identity(identity, now_ms)?; - if request.operation_digest.len() != 32 { - return Err(Error::Peer("invalid operation digest length")); - } - Ok(()) -} - -pub(super) fn validate_effect(request: &wire::EffectRequest, now_ms: i64) -> Result<()> { - validate_target_wire(request.target.as_ref())?; - if request.destination_incarnation.len() != 16 { - return Err(Error::Peer("invalid effect destination incarnation")); - } - validate_effect_identity( - request - .identity - .as_ref() - .ok_or(Error::Peer("effect identity is missing"))?, - now_ms, - )?; - match request.operation.as_ref() { - Some(wire::effect_request::Operation::CellCommand(command)) - if command.command_id != 0 && command.codec_version != 0 => - { - Ok(()) - } - Some(wire::effect_request::Operation::CellCommand(_)) => { - Err(Error::Peer("invalid effect Cell command identifier")) - } - None => Err(Error::Peer("effect operation is missing")), - } -} - -pub(super) fn validate_effect_resolve( - request: &wire::EffectResolveRequest, - now_ms: i64, -) -> Result<()> { - validate_target_wire(request.target.as_ref())?; - if request.destination_incarnation.len() != 16 || request.operation_digest.len() != 32 { - return Err(Error::Peer("invalid effect Resolve identity")); - } - validate_effect_identity( - request - .identity - .as_ref() - .ok_or(Error::Peer("effect Resolve identity is missing"))?, - now_ms, - ) -} - -pub(super) fn validate_migration(request: &wire::MigrationRequest) -> Result<()> { - validate_target_wire(request.target.as_ref())?; - if request.incarnation.len() != 16 - || request.from_code.len() != 32 - || request.to_code.len() != 32 - || request.from_schema == 0 - || request.to_schema == 0 - || request.from_code == request.to_code && request.from_schema == request.to_schema - { - return Err(Error::Peer("invalid peer migration versions")); - } - Ok(()) -} - -pub(super) fn validate_description_wire(description: &wire::CellDescription) -> Result<()> { - if description.cell_id.len() != 32 - || description.incarnation.len() != 16 - || description.code.len() != 32 - || description.schema == 0 - { - return Err(Error::Peer("invalid peer Cell description")); - } - Ok(()) -} - -fn validate_expected_description(expected: Option<&wire::CellDescription>) -> Result<()> { - validate_description_wire(expected.ok_or(Error::Peer("expected Cell description is missing"))?) -} - -pub(super) fn validate_effect_identity(identity: &wire::EffectIdentity, now_ms: i64) -> Result<()> { - let remaining_ms = identity.expires_at_ms.checked_sub(now_ms); - if identity.effect_id.len() != 32 - || identity.source_cell.len() != 32 - || identity.source_incarnation.len() != 16 - || identity.source_sequence == 0 - || now_ms < 0 - || remaining_ms.is_none_or(|remaining| remaining <= 0 || remaining > MAX_EFFECT_LIFETIME_MS) - { - return Err(Error::Peer("invalid or expired effect identity")); - } - let source_cell = crate::CellId::try_from(identity.source_cell.as_slice())?; - let source_incarnation = IncarnationId::try_from(identity.source_incarnation.as_slice())?; - if identity.effect_id.as_slice() - != crate::primitives::effects::effect_id( - source_cell, - source_incarnation, - identity.source_sequence, - identity.ordinal, - ) - { - return Err(Error::Peer("effect identity derivation does not match")); - } - Ok(()) -} - -pub(super) fn validate_target_wire(target: Option<&wire::Target>) -> Result<()> { - let target = target.ok_or(Error::Peer("peer target is missing"))?; - if target.tenant_id.len() != 16 - || target.application_id.len() != 16 - || target.namespace_id.len() != 16 - || target.partition.len() > 1_024 - { - return Err(Error::Peer("invalid peer target bounds")); - } - Ok(()) -} - -pub(super) fn validate_mutation_identity( - identity: &wire::MutationIdentity, - now_ms: i64, -) -> Result<()> { - if identity.request_id.len() != 16 - || identity.incarnation.len() != 16 - || identity.issued_at_ms < 0 - || identity.expires_at_ms <= identity.issued_at_ms - || identity.expires_at_ms - identity.issued_at_ms > MAX_MUTATION_LIFETIME_MS - || identity.issued_at_ms > now_ms.saturating_add(MAX_CLOCK_SKEW_MS) - || identity.expires_at_ms <= now_ms - { - return Err(Error::Peer("invalid or expired mutation identity")); - } - Ok(()) -} - -pub(super) fn validate_timeout(timeout_ms: u32) -> Result<()> { - if timeout_ms > 60_000 { - return Err(Error::Peer("peer operation timeout exceeds 60 seconds")); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/primitives.rs b/crates/crab-cell-runtime/src/primitives.rs deleted file mode 100644 index 1af1faa42..000000000 --- a/crates/crab-cell-runtime/src/primitives.rs +++ /dev/null @@ -1,12 +0,0 @@ -//! Application-facing distributed primitives. - -pub mod activity_pool; -pub mod blob; -pub mod capacity; -pub mod cron; -pub mod effects; -pub mod kv; -pub mod maintenance; -pub mod queue; -pub mod sql; -pub mod workflow; diff --git a/crates/crab-cell-runtime/src/primitives/activity_pool.rs b/crates/crab-cell-runtime/src/primitives/activity_pool.rs deleted file mode 100644 index 9b0dc0444..000000000 --- a/crates/crab-cell-runtime/src/primitives/activity_pool.rs +++ /dev/null @@ -1,215 +0,0 @@ -//! Bounded blocking pool for Activity work. -use std::{ - panic::{AssertUnwindSafe, catch_unwind}, - sync::{Arc, Mutex, mpsc}, - thread::JoinHandle, -}; - -use tokio::sync::{OwnedSemaphorePermit, Semaphore, oneshot}; - -use crate::primitives::workflow::ActivityExecution; -use crate::{Error, Result}; - -const MAX_ACTIVITY_WORKERS: usize = 16; - -type ActivityJob = Box Result + Send + 'static>; - -/// Fixed native blocking-activity pool with claim-before-execution admission. -#[derive(Clone)] -pub struct BlockingActivityPool { - inner: Arc, -} - -impl BlockingActivityPool { - /// Starts one bounded OS-thread pool for trusted blocking activity handlers. - pub fn new(worker_count: usize) -> Result { - if worker_count == 0 || worker_count > MAX_ACTIVITY_WORKERS { - return Err(Error::Capacity("invalid blocking activity worker count")); - } - let (sender, receiver) = mpsc::sync_channel(worker_count); - let receiver = Arc::new(Mutex::new(receiver)); - let mut threads: Vec> = Vec::with_capacity(worker_count); - for index in 0..worker_count { - let receiver = Arc::clone(&receiver); - let thread = match std::thread::Builder::new() - .name(format!("crab-cell-activity-{index}")) - .spawn(move || run_worker(&receiver)) - { - Ok(thread) => thread, - Err(error) => { - drop(sender); - for thread in threads { - let _ = thread.join(); - } - return Err(Error::ActivityWorkerStart(Box::new(error))); - } - }; - threads.push(thread); - } - Ok(Self { - inner: Arc::new(PoolInner { - lifecycle: Mutex::new(ActivityLifecycle { - sender: Some(sender), - threads, - closing: false, - }), - admission: Arc::new(Semaphore::new(worker_count)), - }), - }) - } - - /// Uses the node's CPU count, capped at sixteen blocking callbacks. - pub fn for_system() -> Result { - Self::new( - std::thread::available_parallelism() - .map(usize::from) - .unwrap_or(1) - .clamp(1, MAX_ACTIVITY_WORKERS), - ) - } - - /// Reserves one actual blocking slot without waiting or claiming durable work. - pub fn try_reserve(&self) -> Result> { - let lifecycle = self - .inner - .lifecycle - .lock() - .map_err(|_| Error::RuntimeClosed)?; - if lifecycle.closing { - return Err(Error::RuntimeClosed); - } - let permit = match Arc::clone(&self.inner.admission).try_acquire_owned() { - Ok(permit) => permit, - Err(tokio::sync::TryAcquireError::NoPermits) => return Ok(None), - Err(tokio::sync::TryAcquireError::Closed) => return Err(Error::RuntimeClosed), - }; - Ok(Some(BlockingActivityReservation { - pool: Arc::clone(&self.inner), - permit: Some(permit), - })) - } - - /// Stops admission, drains submitted callbacks and joins every worker. - pub async fn shutdown(&self) -> Result<()> { - let threads = { - let mut lifecycle = self - .inner - .lifecycle - .lock() - .map_err(|_| Error::RuntimeClosed)?; - if lifecycle.closing { - return Err(Error::RuntimeClosed); - } - lifecycle.closing = true; - self.inner.admission.close(); - lifecycle.sender.take(); - std::mem::take(&mut lifecycle.threads) - }; - tokio::task::spawn_blocking(move || join_workers(threads)) - .await - .map_err(Error::ActivityWorkerJoin)? - } -} - -/// A pre-claim blocking slot retained until its callback actually terminates. -pub struct BlockingActivityReservation { - pool: Arc, - permit: Option, -} - -impl BlockingActivityReservation { - pub(crate) async fn execute(mut self, handler: ActivityJob) -> Result { - let (reply, response) = oneshot::channel(); - let job = BlockingJob { - handler, - reply, - permit: self.permit.take().ok_or(Error::RuntimeClosed)?, - }; - let sender = { - let lifecycle = self - .pool - .lifecycle - .lock() - .map_err(|_| Error::RuntimeClosed)?; - if lifecycle.closing { - return Err(Error::RuntimeClosed); - } - lifecycle - .sender - .as_ref() - .cloned() - .ok_or(Error::RuntimeClosed)? - }; - sender.try_send(job).map_err(|error| match error { - mpsc::TrySendError::Full(_) => { - Error::Capacity("blocking activity worker queue is full") - } - mpsc::TrySendError::Disconnected(_) => Error::RuntimeClosed, - })?; - response.await.map_err(|_| Error::RuntimeClosed)? - } -} - -struct PoolInner { - lifecycle: Mutex, - admission: Arc, -} - -struct ActivityLifecycle { - sender: Option>, - threads: Vec>, - closing: bool, -} - -impl Drop for PoolInner { - fn drop(&mut self) { - let lifecycle = match self.lifecycle.get_mut() { - Ok(lifecycle) => lifecycle, - Err(poisoned) => poisoned.into_inner(), - }; - lifecycle.sender.take(); - let _ = join_workers(std::mem::take(&mut lifecycle.threads)); - } -} - -struct BlockingJob { - handler: ActivityJob, - reply: oneshot::Sender>, - permit: OwnedSemaphorePermit, -} - -fn run_worker(receiver: &Mutex>) { - loop { - let job = receiver - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .recv(); - let Ok(job) = job else { - return; - }; - let result = catch_unwind(AssertUnwindSafe(job.handler)) - .map_err(|_| Error::ActivityPanic) - .and_then(|result| result); - // Completion is observable only after admission reopens, allowing the - // caller to reserve the next worker immediately after execute returns. - drop(job.permit); - let _ = job.reply.send(result); - } -} - -fn join_workers(threads: Vec>) -> Result<()> { - let mut panicked = false; - for thread in threads { - if thread.join().is_err() { - panicked = true; - } - } - if panicked { - Err(Error::ActivityWorkerPanic) - } else { - Ok(()) - } -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/primitives/activity_pool/tests.rs b/crates/crab-cell-runtime/src/primitives/activity_pool/tests.rs deleted file mode 100644 index 1c73926f2..000000000 --- a/crates/crab-cell-runtime/src/primitives/activity_pool/tests.rs +++ /dev/null @@ -1,125 +0,0 @@ -use std::{ - sync::{Arc, Barrier}, - time::Duration, -}; - -use super::*; - -fn completed(value: u8) -> ActivityJob { - Box::new(move || Ok(ActivityExecution::Completed(vec![value]))) -} - -#[tokio::test(flavor = "multi_thread")] -async fn reservation_bounds_submitted_work_until_callback_finishes() { - let pool = BlockingActivityPool::new(1).unwrap(); - let reservation = pool.try_reserve().unwrap().unwrap(); - assert!(pool.try_reserve().unwrap().is_none()); - let entered = Arc::new(Barrier::new(2)); - let release = Arc::new(Barrier::new(2)); - let task = tokio::spawn({ - let entered = Arc::clone(&entered); - let release = Arc::clone(&release); - async move { - reservation - .execute(Box::new(move || { - entered.wait(); - release.wait(); - Ok(ActivityExecution::Completed(vec![1])) - })) - .await - } - }); - tokio::task::spawn_blocking(move || entered.wait()) - .await - .unwrap(); - task.abort(); - assert!(pool.try_reserve().unwrap().is_none()); - tokio::task::spawn_blocking(move || release.wait()) - .await - .unwrap(); - tokio::time::timeout(Duration::from_secs(1), async { - loop { - if pool.try_reserve().unwrap().is_some() { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - pool.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn handler_panic_isolated_from_next_job() { - let pool = BlockingActivityPool::new(1).unwrap(); - let panic = pool - .try_reserve() - .unwrap() - .unwrap() - .execute(Box::new(|| panic!("handler panic"))) - .await - .unwrap_err(); - assert!(matches!(panic, Error::ActivityPanic)); - let result = pool - .try_reserve() - .unwrap() - .unwrap() - .execute(completed(7)) - .await - .unwrap(); - assert_eq!(result, ActivityExecution::Completed(vec![7])); - pool.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn completed_callback_releases_capacity_before_result_is_observed() { - let pool = BlockingActivityPool::new(1).unwrap(); - let result = pool - .try_reserve() - .unwrap() - .unwrap() - .execute(completed(9)) - .await - .unwrap(); - assert_eq!(result, ActivityExecution::Completed(vec![9])); - assert!(pool.try_reserve().unwrap().is_some()); - pool.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn shutdown_drains_submitted_callbacks_and_closes_admission() { - let pool = BlockingActivityPool::new(1).unwrap(); - let reservation = pool.try_reserve().unwrap().unwrap(); - let entered = Arc::new(Barrier::new(2)); - let release = Arc::new(Barrier::new(2)); - let task = tokio::spawn({ - let entered = Arc::clone(&entered); - let release = Arc::clone(&release); - async move { - reservation - .execute(Box::new(move || { - entered.wait(); - release.wait(); - Ok(ActivityExecution::Completed(vec![3])) - })) - .await - } - }); - tokio::task::spawn_blocking(move || entered.wait()) - .await - .unwrap(); - let shutdown = tokio::spawn({ - let pool = pool.clone(); - async move { pool.shutdown().await } - }); - tokio::task::spawn_blocking(move || release.wait()) - .await - .unwrap(); - shutdown.await.unwrap().unwrap(); - assert_eq!( - task.await.unwrap().unwrap(), - ActivityExecution::Completed(vec![3]) - ); - assert!(matches!(pool.try_reserve(), Err(Error::RuntimeClosed))); -} diff --git a/crates/crab-cell-runtime/src/primitives/blob.rs b/crates/crab-cell-runtime/src/primitives/blob.rs deleted file mode 100644 index 9c0da16f7..000000000 --- a/crates/crab-cell-runtime/src/primitives/blob.rs +++ /dev/null @@ -1,349 +0,0 @@ -//! Blob primitive: content-addressed parts with lease-based uploads. -use std::collections::BTreeSet; - -use bytes::Bytes; -use crab_storage::{ - GLOBAL_PREFIX, StorageError, Store, content_hash_from_path, global_content_path, - global_content_prefix, -}; -use futures_util::StreamExt; -use object_store::path::Path as ObjectPath; -use rusqlite::{Connection, OptionalExtension, Transaction}; - -use crate::{Error, Result}; - -mod api; -mod sql; -mod store; -#[cfg(test)] -mod tests; - -pub use api::{BlobCommand, BlobModule, BlobNamespace, BlobQueryCommand, register_blob}; -pub use store::{BlobArtifactStore, BlobGarbageCollectionReport}; - -use sql::*; -use store::part_digest; - -const BLOB_SCHEMA: &str = include_str!("../migrations/blob.sql"); -pub(crate) const BLOB_TABLE: &str = "blob_uploads"; -pub(crate) const MAX_BLOB_PART_BYTES: usize = 256 * 1024; -pub(crate) const MAX_BLOB_READ_BYTES: u32 = 512 * 1024; -pub(crate) const MAX_BLOB_PARTS: u32 = 4_096; -const MAX_BLOB_BYTES: u64 = MAX_BLOB_PART_BYTES as u64 * MAX_BLOB_PARTS as u64; -const MAX_KEY_BYTES: usize = 1_024; -const MAX_METADATA_BYTES: usize = 8 * 1_024; -const MAX_CONTENT_TYPE_BYTES: usize = 256; -const MIN_UPLOAD_LIFETIME_MS: i64 = 60_000; -const MAX_UPLOAD_LIFETIME_MS: i64 = 7 * 24 * 60 * 60 * 1_000; -const BLOB_PART_KIND: &str = "blob-parts"; -const MAX_BLOB_READ_PARTS: usize = 8; -const MAX_BLOB_GC_DELETIONS: u32 = 128; - -#[derive(Clone, Copy)] -struct BlobMutationTimes { - now_ms: i64, - issued_at_ms: i64, -} - -/// Conditional publication rule captured when an upload begins. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum BlobCondition { - /// Publishes regardless of the current object. - Any, - /// Publishes only when no object exists at the key. - Missing, - /// Publishes only while the object's ETag still matches. - Etag([u8; 32]), -} - -/// One bounded mutation against a blob namespace. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum BlobMutation { - /// Opens a bounded multipart upload. - Begin { - /// Blob key the upload targets. - key: Vec, - /// Client-chosen upload identity. - upload_id: [u8; 16], - /// Publication rule checked when the upload completes. - condition: BlobCondition, - /// Content type recorded when the object is published. - content_type: Option, - /// Opaque application metadata recorded with the object. - metadata: Vec, - /// Logical time after which an unfinished upload is abandoned. - expires_at_ms: i64, - }, - /// Stores one part of an open upload. - PutPart { - /// Blob key the upload targets. - key: Vec, - /// Upload the part belongs to. - upload_id: [u8; 16], - /// One-based part number. - part_number: u32, - /// Part bytes, bounded by `MAX_BLOB_PART_BYTES`. - payload: Vec, - }, - /// Stores a reference to a content-addressed object-store part. - /// - /// This variant is emitted by [`BlobNamespace`] after it uploads the - /// payload. Applications should use [`BlobMutation::PutPart`] instead. - #[doc(hidden)] - PutPartRef { - key: Vec, - upload_id: [u8; 16], - part_number: u32, - digest: [u8; 32], - size: u32, - }, - /// Completes an open upload and publishes the object. - Complete { - /// Blob key the upload targets. - key: Vec, - /// Upload to complete. - upload_id: [u8; 16], - /// Number of parts the finished object must contain. - part_count: u32, - }, - /// Discards an open upload without publishing. - Abort { - /// Blob key the upload targets. - key: Vec, - /// Upload to discard. - upload_id: [u8; 16], - }, - /// Removes a published object when its condition matches. - Delete { - /// Blob key to remove. - key: Vec, - /// Rule checked before the object is removed. - condition: BlobCondition, - }, -} - -/// Business result of a blob mutation. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum BlobMutationOutcome { - /// The upload is open and accepts parts. - Begun, - /// The part was stored. - PartStored { - /// Digest the part is stored under. - digest: [u8; 32], - }, - /// The object was published. - Committed { - /// ETag of the published bytes. - etag: [u8; 32], - /// Total object size in bytes. - size: u64, - }, - /// The open upload was discarded. - Aborted, - /// The object was removed. - Deleted, - /// No object or upload matched the request. - NotFound, - /// The condition did not match the current object. - Conflict, -} - -/// Immutable metadata for one published blob. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct BlobMetadata { - /// Blob key. - pub key: Vec, - /// ETag of the published bytes. - pub etag: [u8; 32], - /// Total object size in bytes. - pub size: u64, - /// Number of parts the object was assembled from. - pub part_count: u32, - /// Content type recorded at completion. - pub content_type: Option, - /// Opaque application metadata recorded at completion. - pub metadata: Vec, - /// Logical time the object was first published. - pub created_at_ms: i64, - /// Logical time the object was last replaced. - pub updated_at_ms: i64, -} - -/// One bounded blob read result. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct BlobRead { - /// Metadata of the object the range was read from. - pub metadata: BlobMetadata, - /// Byte offset the returned range starts at. - pub offset: u64, - /// Bytes of the requested range. - pub bytes: Vec, - pub(crate) parts: Vec, - pub(crate) end: u64, -} - -/// One object-store part intersecting a bounded Blob range. -#[derive(Clone, Debug, PartialEq, Eq)] -pub(crate) struct BlobPart { - digest: [u8; 32], - offset: u64, - size: u32, -} - -/// One lexicographically ordered page of blob metadata. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct BlobPage { - /// Object metadata in lexicographic key order. - pub objects: Vec, - /// Key to continue from when the page filled its limit. - pub next: Option>, -} - -/// One bounded read against a blob namespace. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum BlobQuery { - /// Reads one object's metadata. - Head { - /// Blob key to look up. - key: Vec, - }, - /// Reads a bounded byte range of one object. - Read { - /// Blob key to read. - key: Vec, - /// First byte of the range. - offset: u64, - /// Maximum bytes to return. - limit: u32, - }, - /// Lists object metadata in key order. - List { - /// Key prefix to list. - prefix: Vec, - /// Key to continue after, from a previous page. - after: Option>, - /// Maximum objects to return. - limit: u32, - }, -} - -/// Result shape for a blob query. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum BlobQueryResult { - /// Metadata for the key, absent when no object exists. - Head(Option), - /// The bounded range, absent when no object exists. - Read(Option), - /// One page of object metadata. - List(BlobPage), -} - -/// Installs the current Blob metadata and manifest schema. -pub fn install_blob_schema(transaction: &Transaction<'_>) -> Result<()> { - transaction.execute_batch(BLOB_SCHEMA)?; - Ok(()) -} - -/// Applies one multipart or conditional blob mutation atomically. -pub fn blob_mutate( - transaction: &Transaction<'_>, - now_ms: i64, - issued_at_ms: i64, - mutation: &BlobMutation, -) -> Result { - validate_now(now_ms)?; - validate_now(issued_at_ms)?; - match mutation { - BlobMutation::Begin { - key, - upload_id, - condition, - content_type, - metadata, - expires_at_ms, - } => begin_upload( - transaction, - BlobMutationTimes { - now_ms, - issued_at_ms, - }, - key, - *upload_id, - *condition, - content_type.as_deref(), - metadata, - *expires_at_ms, - ), - BlobMutation::PutPart { .. } => Err(Error::Command( - "blob part payload must be uploaded to object store", - )), - BlobMutation::PutPartRef { - key, - upload_id, - part_number, - digest, - size, - } => put_part_ref( - transaction, - now_ms, - key, - *upload_id, - *part_number, - *digest, - *size, - ), - BlobMutation::Complete { - key, - upload_id, - part_count, - } => complete_upload(transaction, now_ms, key, *upload_id, *part_count), - BlobMutation::Abort { key, upload_id } => abort_upload(transaction, key, *upload_id), - BlobMutation::Delete { key, condition } => delete_blob(transaction, key, *condition), - } -} - -/// Executes one bounded, integrity-checked blob query. -pub fn blob_query(connection: &Connection, query: &BlobQuery) -> Result { - match query { - BlobQuery::Head { key } => Ok(BlobQueryResult::Head(blob_metadata(connection, key)?)), - BlobQuery::Read { key, offset, limit } => Ok(BlobQueryResult::Read(read_blob( - connection, key, *offset, *limit, - )?)), - BlobQuery::List { - prefix, - after, - limit, - } => Ok(BlobQueryResult::List(list_blobs( - connection, - prefix, - after.as_deref(), - *limit, - )?)), - } -} - -/// Removes bounded abandoned uploads that are not referenced by published objects. -pub fn blob_cleanup_expired( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - validate_now(now_ms)?; - if limit > 128 { - return Err(Error::Command("blob cleanup limit exceeds 128")); - } - transaction - .execute( - "DELETE FROM blob_uploads WHERE upload_id IN (SELECT u.upload_id FROM blob_uploads u INDEXED BY blob_upload_expiry WHERE u.expires_at_ms <= ?1 AND NOT EXISTS (SELECT 1 FROM blob_objects o WHERE o.upload_id = u.upload_id) ORDER BY u.expires_at_ms, u.upload_id LIMIT ?2)", - (now_ms, limit as i64), - ) - .map_err(Into::into) -} - -fn validate_now(now_ms: i64) -> Result<()> { - if now_ms < 0 { - return Err(Error::Command("negative blob logical time")); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/primitives/blob/api.rs b/crates/crab-cell-runtime/src/primitives/blob/api.rs deleted file mode 100644 index 91ea63e37..000000000 --- a/crates/crab-cell-runtime/src/primitives/blob/api.rs +++ /dev/null @@ -1,274 +0,0 @@ -use std::marker::PhantomData; - -use crate::cell::catalog::CatalogRole; -use crate::client::{CellClient, Committed, InvocationError, Observed, Receipt}; -use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue, read_fixed}; -use crate::identity::{ApplicationId, CellTarget, NamespaceId, TenantId, shard_for_scope}; -use crate::primitives::maintenance::{MaintenanceModule, register_maintenance}; -use crate::registry::{Command, Query, RegistryBuilder}; -use crate::registry::{CommandContext, CommandResult, QueryContext}; - -use super::{ - BlobArtifactStore, BlobCondition, BlobMetadata, BlobMutation, BlobMutationOutcome, BlobPage, - BlobPart, BlobQuery, BlobQueryResult, BlobRead, MAX_BLOB_PART_BYTES, MAX_BLOB_READ_BYTES, - MAX_BLOB_READ_PARTS, blob_mutate, blob_query, -}; - -mod codec; - -/// Compile-time namespace and operation identifiers for one Blob module. -pub trait BlobModule: MaintenanceModule { - /// Namespace that owns this Blob module. - const NAMESPACE: NamespaceId; - /// Command id that mutates blobs and uploads. - const MUTATE_COMMAND_ID: u32; - /// Query id that reads objects and uploads. - const QUERY_ID: u32; -} - -/// Registers typed Blob command and query bindings. -pub fn register_blob(registry: &mut RegistryBuilder) -> crate::Result<()> { - registry.bind_blob_module(M::MODULE, M::NAMESPACE)?; - registry.bind_command::>()?; - registry.bind_query::>()?; - register_maintenance::(registry) -} - -/// Typed Blob mutation command. -pub struct BlobCommand(PhantomData M>); - -impl Command for BlobCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::MUTATE_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = BlobMutation; - type Output = BlobMutationOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let outcome = blob_mutate( - context.primitive_transaction(), - context.now_ms(), - context.issued_at_ms(), - &input, - )?; - Ok(match outcome { - BlobMutationOutcome::NotFound | BlobMutationOutcome::Conflict => { - CommandResult::Rejected(outcome) - } - _ => CommandResult::Success(outcome), - }) - } -} - -/// Typed Blob query. -pub struct BlobQueryCommand(PhantomData M>); - -impl Query for BlobQueryCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = BlobQuery; - type Output = BlobQueryResult; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - blob_query(context.primitive_connection(), &input) - } -} - -/// Authorized Blob capability with deterministic key sharding. -pub struct BlobNamespace { - client: CellClient, - artifact_store: BlobArtifactStore, - tenant: TenantId, - application: ApplicationId, - shards: u32, - module: PhantomData M>, -} - -impl Clone for BlobNamespace { - fn clone(&self) -> Self { - Self { - client: self.client.clone(), - artifact_store: self.artifact_store.clone(), - tenant: self.tenant, - application: self.application, - shards: self.shards, - module: PhantomData, - } - } -} - -impl BlobNamespace { - /// Creates a Blob capability after validating its compiled namespace role. - pub fn new( - client: CellClient, - tenant: TenantId, - application: ApplicationId, - ) -> crate::Result { - let shards = client.require_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Blob)?; - let artifact_store = client.blob_artifact_store().ok_or(crate::Error::Control( - "Blob artifact store is not configured", - ))?; - Ok(Self { - client, - artifact_store, - tenant, - application, - shards, - module: PhantomData, - }) - } - - /// Applies one multipart or conditional mutation on the key's shard. - pub async fn mutate( - &self, - identity: crate::cell::executor::MutationIdentity, - mutation: BlobMutation, - ) -> std::result::Result, InvocationError> - { - let target = self - .target(mutation_key(&mutation)) - .map_err(InvocationError::NotStarted)?; - let mutation = match mutation { - BlobMutation::PutPart { - key, - upload_id, - part_number, - payload, - } => { - if payload.len() > MAX_BLOB_PART_BYTES { - return Err(InvocationError::NotStarted(crate::Error::Command( - "blob part exceeds 256 KiB", - ))); - } - let digest = super::part_digest(&payload); - self.artifact_store - .put_part(digest, &payload) - .await - .map_err(InvocationError::NotStarted)?; - BlobMutation::PutPartRef { - key, - upload_id, - part_number, - digest, - size: payload.len() as u32, - } - } - mutation => mutation, - }; - self.client - .command::>(&target, identity, mutation) - .await - } - - /// Reads metadata or a bounded range from one key shard. - pub async fn query( - &self, - query: BlobQuery, - minimum: Option, - ) -> std::result::Result, InvocationError> { - let key = query_key(&query).ok_or_else(|| { - InvocationError::NotStarted(crate::Error::Identity( - "blob list requires explicit shard query", - )) - })?; - let target = self.target(key).map_err(InvocationError::NotStarted)?; - let mut observed = self - .client - .query::>(&target, minimum, query) - .await?; - if let BlobQueryResult::Read(Some(read)) = &mut observed.output { - hydrate_read(&self.artifact_store, read) - .await - .map_err(InvocationError::NotStarted)?; - } - Ok(observed) - } - - /// Lists one shard explicitly; global listing is a bounded fan-out concern. - pub async fn list_shard( - &self, - shard: u32, - prefix: Vec, - after: Option>, - limit: u32, - minimum: Option, - ) -> std::result::Result, InvocationError> { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - self.client - .query::>( - &target, - minimum, - BlobQuery::List { - prefix, - after, - limit, - }, - ) - .await - } - - fn target(&self, key: &[u8]) -> crate::Result { - let shard = shard_for_scope(M::NAMESPACE, key, self.shards)?; - self.shard_target(shard) - } - - fn shard_target(&self, shard: u32) -> crate::Result { - if shard >= self.shards { - return Err(crate::Error::Identity("blob shard outside namespace")); - } - CellTarget::new( - self.tenant, - self.application, - M::NAMESPACE, - &crate::partition_for_shard(shard), - ) - } -} - -fn mutation_key(mutation: &BlobMutation) -> &[u8] { - match mutation { - BlobMutation::Begin { key, .. } - | BlobMutation::PutPart { key, .. } - | BlobMutation::PutPartRef { key, .. } - | BlobMutation::Complete { key, .. } - | BlobMutation::Abort { key, .. } - | BlobMutation::Delete { key, .. } => key, - } -} - -fn query_key(query: &BlobQuery) -> Option<&[u8]> { - match query { - BlobQuery::Head { key } | BlobQuery::Read { key, .. } => Some(key), - BlobQuery::List { .. } => None, - } -} - -async fn hydrate_read(store: &BlobArtifactStore, read: &mut BlobRead) -> crate::Result<()> { - if read.parts.is_empty() || read.bytes.len() > MAX_BLOB_READ_BYTES as usize { - return Ok(()); - } - let end = read.end.min(read.metadata.size); - let mut bytes = Vec::with_capacity((end.saturating_sub(read.offset)) as usize); - for part in &read.parts { - let payload = store.read_part(part.digest, part.size).await?; - let part_end = part.offset.saturating_add(u64::from(part.size)); - let start = read.offset.saturating_sub(part.offset) as usize; - let take_end = end.saturating_sub(part.offset).min(u64::from(part.size)) as usize; - if part_end <= read.offset || part.offset >= end || start > take_end { - return Err(crate::Error::Command("blob range is incomplete")); - } - bytes.extend_from_slice(&payload[start..take_end]); - } - if bytes.len() != end.saturating_sub(read.offset) as usize { - return Err(crate::Error::Command("blob range is incomplete")); - } - read.bytes = bytes; - read.parts.clear(); - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/primitives/blob/api/codec.rs b/crates/crab-cell-runtime/src/primitives/blob/api/codec.rs deleted file mode 100644 index 62ff22f93..000000000 --- a/crates/crab-cell-runtime/src/primitives/blob/api/codec.rs +++ /dev/null @@ -1,389 +0,0 @@ -//! Blob wire codecs for mutations, conditions, queries, and payloads. - -use super::*; - -impl WireValue for BlobMutation { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Begin { - key, - upload_id, - condition, - content_type, - metadata, - expires_at_ms, - } => { - encoder.write_u8(0)?; - encoder.write_bytes(key)?; - encoder.write_bytes(upload_id)?; - condition.encode(encoder)?; - content_type.encode(encoder)?; - encoder.write_bytes(metadata)?; - encoder.write_i64(*expires_at_ms) - } - Self::PutPart { .. } => Err(CodecError::Invalid( - "blob part payload must be uploaded to object store", - )), - Self::PutPartRef { - key, - upload_id, - part_number, - digest, - size, - } => { - encoder.write_u8(1)?; - encoder.write_bytes(key)?; - encoder.write_bytes(upload_id)?; - encoder.write_u32(*part_number)?; - encoder.write_bytes(digest)?; - encoder.write_u32(*size) - } - Self::Complete { - key, - upload_id, - part_count, - } => { - encoder.write_u8(2)?; - encoder.write_bytes(key)?; - encoder.write_bytes(upload_id)?; - encoder.write_u32(*part_count) - } - Self::Abort { key, upload_id } => { - encoder.write_u8(3)?; - encoder.write_bytes(key)?; - encoder.write_bytes(upload_id) - } - Self::Delete { key, condition } => { - encoder.write_u8(4)?; - encoder.write_bytes(key)?; - condition.encode(encoder) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(Self::Begin { - key: decoder.read_bytes()?.to_vec(), - upload_id: read_fixed(decoder, "blob upload ID length")?, - condition: BlobCondition::decode(decoder)?, - content_type: Option::::decode(decoder)?, - metadata: decoder.read_bytes()?.to_vec(), - expires_at_ms: decoder.read_i64()?, - }), - 1 => { - let key = decoder.read_bytes()?.to_vec(); - let upload_id = read_fixed(decoder, "blob upload ID length")?; - let part_number = decoder.read_u32()?; - let digest = read_fixed(decoder, "blob part digest length")?; - let size = decoder.read_u32()?; - if size as usize > MAX_BLOB_PART_BYTES { - return Err(CodecError::Invalid("blob part exceeds 256 KiB")); - } - Ok(Self::PutPartRef { - key, - upload_id, - part_number, - digest, - size, - }) - } - 2 => Ok(Self::Complete { - key: decoder.read_bytes()?.to_vec(), - upload_id: read_fixed(decoder, "blob upload ID length")?, - part_count: decoder.read_u32()?, - }), - 3 => Ok(Self::Abort { - key: decoder.read_bytes()?.to_vec(), - upload_id: read_fixed(decoder, "blob upload ID length")?, - }), - 4 => Ok(Self::Delete { - key: decoder.read_bytes()?.to_vec(), - condition: BlobCondition::decode(decoder)?, - }), - _ => Err(CodecError::Invalid("invalid blob mutation tag")), - } - } -} - -impl WireValue for BlobCondition { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Any => encoder.write_u8(0), - Self::Missing => encoder.write_u8(1), - Self::Etag(etag) => { - encoder.write_u8(2)?; - encoder.write_bytes(etag) - } - } - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(Self::Any), - 1 => Ok(Self::Missing), - 2 => Ok(Self::Etag(read_fixed(decoder, "blob ETag length")?)), - _ => Err(CodecError::Invalid("invalid blob condition tag")), - } - } -} - -impl WireValue for BlobMutationOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Begun => encoder.write_u8(0), - Self::PartStored { digest } => { - encoder.write_u8(1)?; - encoder.write_bytes(digest) - } - Self::Committed { etag, size } => { - encoder.write_u8(2)?; - encoder.write_bytes(etag)?; - encoder.write_u64(*size) - } - Self::Aborted => encoder.write_u8(3), - Self::Deleted => encoder.write_u8(4), - Self::NotFound => encoder.write_u8(5), - Self::Conflict => encoder.write_u8(6), - } - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(Self::Begun), - 1 => Ok(Self::PartStored { - digest: read_fixed(decoder, "blob part digest length")?, - }), - 2 => Ok(Self::Committed { - etag: read_fixed(decoder, "blob ETag length")?, - size: decoder.read_u64()?, - }), - 3 => Ok(Self::Aborted), - 4 => Ok(Self::Deleted), - 5 => Ok(Self::NotFound), - 6 => Ok(Self::Conflict), - _ => Err(CodecError::Invalid("invalid blob outcome tag")), - } - } -} - -impl WireValue for BlobQuery { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Head { key } => { - encoder.write_u8(0)?; - encoder.write_bytes(key) - } - Self::Read { key, offset, limit } => { - if *limit > MAX_BLOB_READ_BYTES { - return Err(CodecError::Invalid("blob read exceeds 512 KiB")); - } - encoder.write_u8(1)?; - encoder.write_bytes(key)?; - encoder.write_u64(*offset)?; - encoder.write_u32(*limit) - } - Self::List { - prefix, - after, - limit, - } => { - encoder.write_u8(2)?; - encoder.write_bytes(prefix)?; - after.encode(encoder)?; - encoder.write_u32(*limit) - } - } - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(Self::Head { - key: decoder.read_bytes()?.to_vec(), - }), - 1 => { - let key = decoder.read_bytes()?.to_vec(); - let offset = decoder.read_u64()?; - let limit = decoder.read_u32()?; - if limit > MAX_BLOB_READ_BYTES { - return Err(CodecError::Invalid("blob read exceeds 512 KiB")); - } - Ok(Self::Read { key, offset, limit }) - } - 2 => Ok(Self::List { - prefix: decoder.read_bytes()?.to_vec(), - after: Option::>::decode(decoder)?, - limit: decoder.read_u32()?, - }), - _ => Err(CodecError::Invalid("invalid blob query tag")), - } - } -} - -impl WireValue for BlobQueryResult { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Head(value) => { - encoder.write_u8(0)?; - value.encode(encoder) - } - Self::Read(value) => { - encoder.write_u8(1)?; - value.encode(encoder) - } - Self::List(value) => { - encoder.write_u8(2)?; - value.encode(encoder) - } - } - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(Self::Head(Option::::decode(decoder)?)), - 1 => Ok(Self::Read(Option::::decode(decoder)?)), - 2 => Ok(Self::List(BlobPage::decode(decoder)?)), - _ => Err(CodecError::Invalid("invalid blob query result tag")), - } - } -} - -impl WireValue for BlobMetadata { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.key)?; - encoder.write_bytes(&self.etag)?; - encoder.write_u64(self.size)?; - encoder.write_u32(self.part_count)?; - self.content_type.encode(encoder)?; - encoder.write_bytes(&self.metadata)?; - encoder.write_i64(self.created_at_ms)?; - encoder.write_i64(self.updated_at_ms) - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - key: decoder.read_bytes()?.to_vec(), - etag: read_fixed(decoder, "blob ETag length")?, - size: decoder.read_u64()?, - part_count: decoder.read_u32()?, - content_type: Option::::decode(decoder)?, - metadata: decoder.read_bytes()?.to_vec(), - created_at_ms: decoder.read_i64()?, - updated_at_ms: decoder.read_i64()?, - }) - } -} - -impl WireValue for BlobRead { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if self.bytes.len() > MAX_BLOB_READ_BYTES as usize { - return Err(CodecError::Invalid("blob read exceeds 512 KiB")); - } - self.metadata.encode(encoder)?; - encoder.write_u64(self.offset)?; - encoder.write_u64(self.end)?; - if self.parts.len() > MAX_BLOB_READ_PARTS { - return Err(CodecError::Invalid("blob range references too many parts")); - } - encoder.write_count(self.parts.len())?; - for part in &self.parts { - encoder.write_bytes(&part.digest)?; - encoder.write_u64(part.offset)?; - encoder.write_u32(part.size)?; - } - encoder.write_bytes(&self.bytes) - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let metadata = BlobMetadata::decode(decoder)?; - let offset = decoder.read_u64()?; - let end = decoder.read_u64()?; - if end < offset { - return Err(CodecError::Invalid("blob range end precedes offset")); - } - let count = decoder.read_count()?; - if count > MAX_BLOB_READ_PARTS { - return Err(CodecError::Invalid("blob range references too many parts")); - } - let mut parts = Vec::with_capacity(count); - for _ in 0..count { - let digest = read_fixed(decoder, "blob part digest length")?; - let part_offset = decoder.read_u64()?; - let size = decoder.read_u32()?; - if size as usize > MAX_BLOB_PART_BYTES { - return Err(CodecError::Invalid("blob part exceeds 256 KiB")); - } - parts.push(BlobPart { - digest, - offset: part_offset, - size, - }); - } - let bytes = decoder.read_bytes()?.to_vec(); - if bytes.len() > MAX_BLOB_READ_BYTES as usize { - return Err(CodecError::Invalid("blob read exceeds 512 KiB")); - } - Ok(Self { - metadata, - offset, - bytes, - parts, - end, - }) - } -} - -impl WireValue for BlobPage { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if self.objects.len() > 128 { - return Err(CodecError::Invalid("blob page exceeds 128 objects")); - } - encoder.write_count(self.objects.len())?; - for object in &self.objects { - object.encode(encoder)?; - } - self.next.encode(encoder) - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let count = decoder.read_count()?; - if count > 128 { - return Err(CodecError::Invalid("blob page exceeds 128 objects")); - } - let mut objects = Vec::with_capacity(count); - for _ in 0..count { - objects.push(BlobMetadata::decode(decoder)?); - } - Ok(Self { - objects, - next: Option::>::decode(decoder)?, - }) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::codec::roundtrip; - - #[test] - fn blob_codecs_roundtrip_mutations_and_queries() { - roundtrip(BlobMutation::Begin { - key: b"logs/a".to_vec(), - upload_id: [1; 16], - condition: BlobCondition::Missing, - content_type: Some("text/plain".into()), - metadata: b"owner=a".to_vec(), - expires_at_ms: 100_000, - }); - roundtrip(BlobMutation::PutPartRef { - key: b"logs/a".to_vec(), - upload_id: [1; 16], - part_number: 1, - digest: [2; 32], - size: 4, - }); - roundtrip(BlobQuery::Read { - key: b"logs/a".to_vec(), - offset: 2, - limit: MAX_BLOB_READ_BYTES, - }); - roundtrip(BlobMutationOutcome::Committed { - etag: [2; 32], - size: 4, - }); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/blob/sql.rs b/crates/crab-cell-runtime/src/primitives/blob/sql.rs deleted file mode 100644 index 40c623c9a..000000000 --- a/crates/crab-cell-runtime/src/primitives/blob/sql.rs +++ /dev/null @@ -1,561 +0,0 @@ -//! Blob metadata and manifest SQL rows. - -use super::*; - -pub(super) fn begin_upload( - transaction: &Transaction<'_>, - times: BlobMutationTimes, - key: &[u8], - upload_id: [u8; 16], - condition: BlobCondition, - content_type: Option<&str>, - metadata: &[u8], - expires_at_ms: i64, -) -> Result { - validate_key(key)?; - validate_metadata(content_type, metadata)?; - let lifetime = expires_at_ms - .checked_sub(times.issued_at_ms) - .ok_or(Error::Command("blob upload expiry overflow"))?; - if !(MIN_UPLOAD_LIFETIME_MS..=MAX_UPLOAD_LIFETIME_MS).contains(&lifetime) { - return Err(Error::Command( - "blob upload lifetime must be between one minute and seven days", - )); - } - if expires_at_ms <= times.now_ms { - return Err(Error::Command("blob upload has already expired")); - } - let digest = begin_digest(key, condition, content_type, metadata, expires_at_ms); - let existing = transaction - .query_row( - "SELECT request_digest FROM blob_uploads WHERE upload_id = ?1", - [upload_id.as_slice()], - |row| row.get::<_, Vec>(0), - ) - .optional()?; - if let Some(existing) = existing { - return Ok(if existing.as_slice() == digest { - BlobMutationOutcome::Begun - } else { - BlobMutationOutcome::Conflict - }); - } - let (condition_code, expected_etag) = encode_condition(condition); - transaction.execute( - "INSERT INTO blob_uploads(upload_id, object_key, request_digest, condition, expected_etag, content_type, metadata, created_at_ms, expires_at_ms, completed, etag, size, part_count) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, 0, NULL, 0, 0)", - ( - upload_id.as_slice(), - key, - digest.as_slice(), - condition_code, - expected_etag.as_ref().map(<[u8; 32]>::as_slice), - content_type, - metadata, - times.now_ms, - expires_at_ms, - ), - )?; - Ok(BlobMutationOutcome::Begun) -} - -pub(super) fn put_part_ref( - transaction: &Transaction<'_>, - now_ms: i64, - key: &[u8], - upload_id: [u8; 16], - part_number: u32, - digest: [u8; 32], - size: u32, -) -> Result { - validate_key(key)?; - if !(1..=MAX_BLOB_PARTS).contains(&part_number) { - return Err(Error::Command("blob part number must be in 1..=4096")); - } - if usize::try_from(size) - .ok() - .is_none_or(|size| size > MAX_BLOB_PART_BYTES) - { - return Err(Error::Command("blob part exceeds 256 KiB")); - } - let upload = upload_state(transaction, upload_id)?; - let Some((stored_key, completed, expires_at_ms)) = upload else { - return Ok(BlobMutationOutcome::NotFound); - }; - if stored_key != key || completed || expires_at_ms <= now_ms { - return Ok(BlobMutationOutcome::Conflict); - } - let existing = transaction - .query_row( - "SELECT digest, size FROM blob_parts WHERE upload_id = ?1 AND part_number = ?2", - (upload_id.as_slice(), i64::from(part_number)), - |row| Ok((row.get::<_, Vec>(0)?, row.get::<_, i64>(1)?)), - ) - .optional()?; - if let Some(existing) = existing { - return Ok( - if existing.0.as_slice() == digest.as_slice() && existing.1 == i64::from(size) { - BlobMutationOutcome::PartStored { digest } - } else { - BlobMutationOutcome::Conflict - }, - ); - } - transaction.execute( - "INSERT INTO blob_parts(upload_id, part_number, digest, size, byte_offset) VALUES (?1, ?2, ?3, ?4, NULL)", - ( - upload_id.as_slice(), - i64::from(part_number), - digest.as_slice(), - i64::from(size), - ), - )?; - Ok(BlobMutationOutcome::PartStored { digest }) -} - -pub(super) fn complete_upload( - transaction: &Transaction<'_>, - now_ms: i64, - key: &[u8], - upload_id: [u8; 16], - part_count: u32, -) -> Result { - validate_key(key)?; - if !(1..=MAX_BLOB_PARTS).contains(&part_count) { - return Err(Error::Command( - "blob completion part count must be in 1..=4096", - )); - } - let upload = transaction - .query_row( - "SELECT object_key, condition, expected_etag, content_type, metadata, created_at_ms, expires_at_ms, completed, etag, size, part_count FROM blob_uploads WHERE upload_id = ?1", - [upload_id.as_slice()], - |row| { - Ok(( - row.get::<_, Vec>(0)?, row.get::<_, i64>(1)?, - row.get::<_, Option>>(2)?, row.get::<_, Option>(3)?, - row.get::<_, Vec>(4)?, row.get::<_, i64>(5)?, row.get::<_, i64>(6)?, - row.get::<_, i64>(7)?, row.get::<_, Option>>(8)?, - row.get::<_, i64>(9)?, row.get::<_, i64>(10)?, - )) - }, - ) - .optional()?; - let Some(( - stored_key, - condition, - expected, - content_type, - metadata, - created_at_ms, - expires_at_ms, - completed, - stored_etag, - stored_size, - stored_parts, - )) = upload - else { - return Ok(BlobMutationOutcome::NotFound); - }; - if stored_key != key { - return Ok(BlobMutationOutcome::Conflict); - } - if completed != 0 { - let etag = exact_etag(stored_etag)?; - return Ok(if stored_parts == i64::from(part_count) { - BlobMutationOutcome::Committed { - etag, - size: nonnegative_u64(stored_size, "invalid stored blob size")?, - } - } else { - BlobMutationOutcome::Conflict - }); - } - if expires_at_ms <= now_ms { - return Ok(BlobMutationOutcome::Conflict); - } - let condition = decode_condition(condition, expected)?; - if !condition_matches(transaction, key, condition)? { - return Ok(BlobMutationOutcome::Conflict); - } - let mut statement = transaction.prepare( - "SELECT part_number, digest, size FROM blob_parts WHERE upload_id = ?1 ORDER BY part_number", - )?; - let rows = statement.query_map([upload_id.as_slice()], |row| { - Ok(( - row.get::<_, i64>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, i64>(2)?, - )) - })?; - let mut parts = Vec::with_capacity(part_count as usize); - for row in rows { - parts.push(row?); - } - drop(statement); - if parts.len() != part_count as usize { - return Ok(BlobMutationOutcome::Conflict); - } - let mut manifest = blake3::Hasher::new(); - manifest.update(b"crab.blob.v1\0"); - let mut offset = 0_u64; - for (index, (number, digest, payload_len)) in parts.iter().enumerate() { - if *number != (index + 1) as i64 || digest.len() != 32 || *payload_len < 0 { - return Err(Error::Command("invalid stored blob part")); - } - manifest.update(&number.to_be_bytes()); - let payload_len = u32::try_from(*payload_len) - .map_err(|_| Error::Command("invalid stored blob part length"))?; - manifest.update(&payload_len.to_be_bytes()); - manifest.update(digest); - transaction.execute( - "UPDATE blob_parts SET byte_offset = ?1 WHERE upload_id = ?2 AND part_number = ?3 AND byte_offset IS NULL", - (offset as i64, upload_id.as_slice(), *number), - )?; - offset = offset - .checked_add(u64::from(payload_len)) - .ok_or(Error::Command("blob size overflow"))?; - } - if offset > MAX_BLOB_BYTES { - return Err(Error::Command("blob exceeds one GiB")); - } - let etag = *manifest.finalize().as_bytes(); - let prior_upload = transaction - .query_row( - "SELECT upload_id FROM blob_objects WHERE object_key = ?1", - [key], - |row| row.get::<_, Vec>(0), - ) - .optional()?; - transaction.execute( - "INSERT INTO blob_objects(object_key, upload_id, etag, size, part_count, content_type, metadata, created_at_ms, updated_at_ms) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9) ON CONFLICT(object_key) DO UPDATE SET upload_id = excluded.upload_id, etag = excluded.etag, size = excluded.size, part_count = excluded.part_count, content_type = excluded.content_type, metadata = excluded.metadata, updated_at_ms = excluded.updated_at_ms", - ( - key, upload_id.as_slice(), etag.as_slice(), offset as i64, - i64::from(part_count), content_type, metadata, created_at_ms, now_ms, - ), - )?; - transaction.execute( - "UPDATE blob_uploads SET completed = 1, etag = ?1, size = ?2, part_count = ?3 WHERE upload_id = ?4 AND completed = 0", - (etag.as_slice(), offset as i64, i64::from(part_count), upload_id.as_slice()), - )?; - if let Some(prior) = prior_upload.filter(|prior| prior.as_slice() != upload_id) { - transaction.execute("DELETE FROM blob_uploads WHERE upload_id = ?1", [prior])?; - } - Ok(BlobMutationOutcome::Committed { etag, size: offset }) -} - -pub(super) fn abort_upload( - transaction: &Transaction<'_>, - key: &[u8], - upload_id: [u8; 16], -) -> Result { - validate_key(key)?; - let changed = transaction.execute( - "DELETE FROM blob_uploads WHERE upload_id = ?1 AND object_key = ?2 AND completed = 0", - (upload_id.as_slice(), key), - )?; - Ok(if changed == 1 { - BlobMutationOutcome::Aborted - } else { - BlobMutationOutcome::NotFound - }) -} - -pub(super) fn delete_blob( - transaction: &Transaction<'_>, - key: &[u8], - condition: BlobCondition, -) -> Result { - validate_key(key)?; - if !condition_matches(transaction, key, condition)? { - return Ok(BlobMutationOutcome::Conflict); - } - let upload = transaction - .query_row( - "SELECT upload_id FROM blob_objects WHERE object_key = ?1", - [key], - |row| row.get::<_, Vec>(0), - ) - .optional()?; - let Some(upload) = upload else { - return Ok(BlobMutationOutcome::NotFound); - }; - transaction.execute("DELETE FROM blob_objects WHERE object_key = ?1", [key])?; - transaction.execute("DELETE FROM blob_uploads WHERE upload_id = ?1", [upload])?; - Ok(BlobMutationOutcome::Deleted) -} - -pub(super) fn read_blob( - connection: &Connection, - key: &[u8], - offset: u64, - limit: u32, -) -> Result> { - let Some(metadata) = blob_metadata(connection, key)? else { - return Ok(None); - }; - if limit > MAX_BLOB_READ_BYTES { - return Err(Error::Command("blob read exceeds 512 KiB")); - } - if offset >= metadata.size || limit == 0 { - return Ok(Some(BlobRead { - metadata, - offset, - bytes: Vec::new(), - parts: Vec::new(), - end: offset, - })); - } - let end = offset.saturating_add(u64::from(limit)).min(metadata.size); - let upload_id = connection.query_row( - "SELECT upload_id FROM blob_objects WHERE object_key = ?1", - [key], - |row| row.get::<_, Vec>(0), - )?; - let mut statement = connection.prepare( - "SELECT byte_offset, digest, size FROM blob_parts WHERE upload_id = ?1 AND byte_offset < ?2 AND byte_offset + size > ?3 ORDER BY part_number", - )?; - let rows = statement.query_map((upload_id, end as i64, offset as i64), |row| { - Ok(( - row.get::<_, i64>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, i64>(2)?, - )) - })?; - let mut parts = Vec::new(); - for row in rows { - let (part_offset, digest, size) = row?; - if part_offset < 0 || digest.len() != 32 || size < 0 { - return Err(Error::Command("blob part integrity check failed")); - } - let digest: [u8; 32] = digest - .try_into() - .map_err(|_| Error::Command("invalid stored blob part digest"))?; - let size = - u32::try_from(size).map_err(|_| Error::Command("invalid stored blob part size"))?; - if size as usize > MAX_BLOB_PART_BYTES { - return Err(Error::Command("invalid stored blob part size")); - } - let part_offset = part_offset as u64; - parts.push(BlobPart { - digest, - offset: part_offset, - size, - }); - } - if parts.len() > MAX_BLOB_READ_PARTS { - return Err(Error::Command("blob range references too many parts")); - } - if parts.is_empty() { - return Err(Error::Command("blob range is incomplete")); - } - Ok(Some(BlobRead { - metadata, - offset, - bytes: Vec::new(), - parts, - end, - })) -} - -pub(super) fn blob_metadata(connection: &Connection, key: &[u8]) -> Result> { - validate_key(key)?; - connection - .query_row( - "SELECT etag, size, part_count, content_type, metadata, created_at_ms, updated_at_ms FROM blob_objects WHERE object_key = ?1", - [key], - |row| decode_metadata(key.to_vec(), row), - ) - .optional() - .map_err(Into::into) -} - -pub(super) fn list_blobs( - connection: &Connection, - prefix: &[u8], - after: Option<&[u8]>, - limit: u32, -) -> Result { - if prefix.len() > MAX_KEY_BYTES || after.is_some_and(|value| value.len() > MAX_KEY_BYTES) { - return Err(Error::Command("blob list key exceeds 1024 bytes")); - } - if !(1..=128).contains(&limit) { - return Err(Error::Command("blob list limit must be in 1..=128")); - } - let after = after.unwrap_or_default(); - let candidates = if let Some(upper) = prefix_successor(prefix) { - let mut statement = connection.prepare( - "SELECT object_key, etag, size, part_count, content_type, metadata, created_at_ms, updated_at_ms FROM blob_objects WHERE object_key > ?1 AND object_key >= ?2 AND object_key < ?3 ORDER BY object_key LIMIT ?4", - )?; - statement - .query_map((after, prefix, upper, i64::from(limit) + 1), |row| { - let key = row.get::<_, Vec>(0)?; - decode_metadata(key, row) - })? - .collect::, _>>()? - } else { - let mut statement = connection.prepare( - "SELECT object_key, etag, size, part_count, content_type, metadata, created_at_ms, updated_at_ms FROM blob_objects WHERE object_key > ?1 AND object_key >= ?2 ORDER BY object_key LIMIT ?3", - )?; - statement - .query_map((after, prefix, i64::from(limit) + 1), |row| { - let key = row.get::<_, Vec>(0)?; - decode_metadata(key, row) - })? - .collect::, _>>()? - }; - let mut objects = Vec::with_capacity(limit as usize); - let mut bytes = 0_usize; - let mut truncated = false; - for object in candidates { - let object_bytes = object - .key - .len() - .checked_add(object.metadata.len()) - .and_then(|value| { - value.checked_add(object.content_type.as_ref().map_or(0, String::len)) - }) - .and_then(|value| value.checked_add(96)) - .ok_or(Error::Command("blob list byte count overflow"))?; - if objects.len() == limit as usize || bytes.saturating_add(object_bytes) > 512 * 1024 { - truncated = true; - break; - } - bytes += object_bytes; - objects.push(object); - } - let next = truncated - .then(|| objects.last().map(|object| object.key.clone())) - .flatten(); - Ok(BlobPage { objects, next }) -} - -fn prefix_successor(prefix: &[u8]) -> Option> { - let mut upper = prefix.to_vec(); - let index = upper.iter().rposition(|byte| *byte != u8::MAX)?; - upper[index] += 1; - upper.truncate(index + 1); - Some(upper) -} - -fn decode_metadata(key: Vec, row: &rusqlite::Row<'_>) -> rusqlite::Result { - let offset = usize::from(row.as_ref().column_count() == 8); - let etag = row.get::<_, Vec>(offset)?; - let etag: [u8; 32] = etag.try_into().map_err(|_| rusqlite::Error::InvalidQuery)?; - let size = row.get::<_, i64>(offset + 1)?; - let part_count = row.get::<_, i64>(offset + 2)?; - Ok(BlobMetadata { - key, - etag, - size: u64::try_from(size).map_err(|_| rusqlite::Error::InvalidQuery)?, - part_count: u32::try_from(part_count).map_err(|_| rusqlite::Error::InvalidQuery)?, - content_type: row.get(offset + 3)?, - metadata: row.get(offset + 4)?, - created_at_ms: row.get(offset + 5)?, - updated_at_ms: row.get(offset + 6)?, - }) -} - -fn upload_state( - transaction: &Transaction<'_>, - upload_id: [u8; 16], -) -> Result, bool, i64)>> { - transaction - .query_row( - "SELECT object_key, completed, expires_at_ms FROM blob_uploads WHERE upload_id = ?1", - [upload_id.as_slice()], - |row| Ok((row.get(0)?, row.get::<_, i64>(1)? != 0, row.get(2)?)), - ) - .optional() - .map_err(Into::into) -} - -fn condition_matches( - transaction: &Transaction<'_>, - key: &[u8], - condition: BlobCondition, -) -> Result { - let current = transaction - .query_row( - "SELECT etag FROM blob_objects WHERE object_key = ?1", - [key], - |row| row.get::<_, Vec>(0), - ) - .optional()?; - Ok(match condition { - BlobCondition::Any => true, - BlobCondition::Missing => current.is_none(), - BlobCondition::Etag(expected) => current.as_deref() == Some(expected.as_slice()), - }) -} - -fn encode_condition(condition: BlobCondition) -> (i64, Option<[u8; 32]>) { - match condition { - BlobCondition::Any => (0, None), - BlobCondition::Missing => (1, None), - BlobCondition::Etag(etag) => (2, Some(etag)), - } -} - -fn decode_condition(code: i64, etag: Option>) -> Result { - match (code, etag) { - (0, None) => Ok(BlobCondition::Any), - (1, None) => Ok(BlobCondition::Missing), - (2, Some(etag)) => { - Ok(BlobCondition::Etag(etag.try_into().map_err(|_| { - Error::Command("invalid stored blob condition") - })?)) - } - _ => Err(Error::Command("invalid stored blob condition")), - } -} - -fn begin_digest( - key: &[u8], - condition: BlobCondition, - content_type: Option<&str>, - metadata: &[u8], - expires_at_ms: i64, -) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.blob-upload.v1\0"); - hasher.update(key); - let (code, etag) = encode_condition(condition); - hasher.update(&code.to_be_bytes()); - if let Some(etag) = etag { - hasher.update(&etag); - } - if let Some(content_type) = content_type { - hasher.update(content_type.as_bytes()); - } - hasher.update(metadata); - hasher.update(&expires_at_ms.to_be_bytes()); - *hasher.finalize().as_bytes() -} - -fn validate_key(key: &[u8]) -> Result<()> { - if key.is_empty() || key.len() > MAX_KEY_BYTES { - return Err(Error::Command("blob key must be 1..=1024 bytes")); - } - Ok(()) -} - -fn validate_metadata(content_type: Option<&str>, metadata: &[u8]) -> Result<()> { - if metadata.len() > MAX_METADATA_BYTES - || content_type - .is_some_and(|value| value.is_empty() || value.len() > MAX_CONTENT_TYPE_BYTES) - { - return Err(Error::Command("blob metadata exceeds limits")); - } - Ok(()) -} - -fn exact_etag(value: Option>) -> Result<[u8; 32]> { - value - .ok_or(Error::Command("completed blob upload lacks ETag"))? - .try_into() - .map_err(|_| Error::Command("invalid stored blob ETag")) -} - -fn nonnegative_u64(value: i64, message: &'static str) -> Result { - u64::try_from(value).map_err(|_| Error::Command(message)) -} diff --git a/crates/crab-cell-runtime/src/primitives/blob/store.rs b/crates/crab-cell-runtime/src/primitives/blob/store.rs deleted file mode 100644 index 0be021a04..000000000 --- a/crates/crab-cell-runtime/src/primitives/blob/store.rs +++ /dev/null @@ -1,172 +0,0 @@ -//! Durable Blob part store backed by the Crab object store. - -use super::*; - -/// Object-store backing for Blob parts. -/// -/// SQLite stores only the bounded upload manifest and content digests. Part -/// bytes are immutable, content-addressed objects in the configured Crab -/// object store and are verified again on every range read. -#[derive(Clone)] -pub struct BlobArtifactStore { - store: Store, -} - -/// Result from one Blob part reachability sweep. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct BlobGarbageCollectionReport { - scanned: u64, - deleted: u32, - has_more: bool, -} - -impl BlobGarbageCollectionReport { - /// Returns the number of object-store entries inspected by the sweep. - #[must_use] - pub const fn scanned(self) -> u64 { - self.scanned - } - - /// Returns the number of unreferenced part objects deleted by the sweep. - #[must_use] - pub const fn deleted(self) -> u32 { - self.deleted - } - - /// Returns whether the deletion budget was reached and another sweep may be needed. - #[must_use] - pub const fn has_more(self) -> bool { - self.has_more - } -} - -impl BlobArtifactStore { - /// Wraps a configured Crab object store for Blob artifact data. - #[must_use] - pub fn new(store: Store) -> Self { - Self { store } - } - - pub(super) async fn put_part(&self, digest: [u8; 32], payload: &[u8]) -> Result<()> { - if part_digest(payload) != digest { - return Err(Error::Command("blob part digest does not match payload")); - } - self.store - .put(&self.part_path(&digest), Bytes::copy_from_slice(payload)) - .await?; - Ok(()) - } - - pub(super) async fn read_part(&self, digest: [u8; 32], size: u32) -> Result> { - if usize::try_from(size) - .ok() - .is_none_or(|size| size > MAX_BLOB_PART_BYTES) - { - return Err(Error::Command("invalid stored blob part size")); - } - let (bytes, _) = self - .store - .get_with_etag_bounded(&self.part_path(&digest), u64::from(size)) - .await?; - if bytes.len() != size as usize || part_digest(&bytes) != digest { - return Err(Error::Command("blob part integrity check failed")); - } - Ok(bytes.to_vec()) - } - - /// Reclaims old part objects absent from a complete cross-Cell reference set. - /// - /// Callers must quiesce Blob writes in this object-store scope and build - /// `live_digests` from every authoritative Cell database sharing it. A grace - /// boundary alone cannot protect an old part reused by a concurrent upload. - /// Objects newer than `cutoff_ms` are retained for uploads whose SQLite - /// manifests have not committed. Each call inspects the complete unordered - /// listing until 128 parts have been deleted; schedule another call while - /// [`BlobGarbageCollectionReport::has_more`] is true. - pub async fn sweep_unreferenced( - &self, - live_digests: &BTreeSet<[u8; 32]>, - cutoff_ms: i64, - ) -> Result { - if cutoff_ms < 0 { - return Err(Error::Command("negative blob garbage-collection cutoff")); - } - let prefix = self - .store - .storage_scope() - .map_or(GLOBAL_PREFIX, |scope| scope.global_prefix.as_str()); - let mut objects = self - .store - .list_stream(&global_content_prefix(prefix, BLOB_PART_KIND)); - let mut scanned = 0_u64; - let mut candidates = Vec::with_capacity(MAX_BLOB_GC_DELETIONS as usize); - while let Some(object) = objects.next().await { - let object = object?; - scanned = scanned.saturating_add(1); - let Some(hash) = content_hash_from_path(object.location.as_ref(), BLOB_PART_KIND) - else { - continue; - }; - let Some(digest) = decode_hex_digest(hash) else { - continue; - }; - if live_digests.contains(&digest) || object.last_modified.timestamp_millis() > cutoff_ms - { - continue; - } - candidates.push(object.location); - if candidates.len() == MAX_BLOB_GC_DELETIONS as usize { - break; - } - } - drop(objects); - let has_more = candidates.len() == MAX_BLOB_GC_DELETIONS as usize; - let mut deleted = 0_u32; - for path in candidates { - match self.store.delete(&path).await { - Ok(()) | Err(StorageError::NotFound { .. }) => deleted += 1, - Err(error) => return Err(error.into()), - } - } - Ok(BlobGarbageCollectionReport { - scanned, - deleted, - has_more, - }) - } - - fn part_path(&self, digest: &[u8; 32]) -> ObjectPath { - let hash = blake3::Hash::from_bytes(*digest).to_hex().to_string(); - let prefix = self - .store - .storage_scope() - .map_or(GLOBAL_PREFIX, |scope| scope.global_prefix.as_str()); - global_content_path(prefix, BLOB_PART_KIND, &hash) - } -} - -fn decode_hex_digest(value: &str) -> Option<[u8; 32]> { - if value.len() != 64 { - return None; - } - let mut digest = [0_u8; 32]; - for (index, pair) in value.as_bytes().as_chunks::<2>().0.iter().enumerate() { - digest[index] = (hex_nibble(pair[0])? << 4) | hex_nibble(pair[1])?; - } - Some(digest) -} - -fn hex_nibble(value: u8) -> Option { - match value { - b'0'..=b'9' => Some(value - b'0'), - b'a'..=b'f' => Some(value - b'a' + 10), - _ => None, - } -} - -pub(super) fn part_digest(payload: &[u8]) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.blob-part.v1\0"); - hasher.update(payload); - *hasher.finalize().as_bytes() -} diff --git a/crates/crab-cell-runtime/src/primitives/blob/tests.rs b/crates/crab-cell-runtime/src/primitives/blob/tests.rs deleted file mode 100644 index 9de76d3b0..000000000 --- a/crates/crab-cell-runtime/src/primitives/blob/tests.rs +++ /dev/null @@ -1,557 +0,0 @@ -use super::*; -use crab_ltx::rusqlite::Connection; -use std::collections::BTreeSet; - -#[tokio::test] -async fn multipart_publish_is_atomic_conditional_and_range_readable() { - use crab_storage::Store; - use object_store::memory::InMemory; - let artifacts = BlobArtifactStore::new(Store::new(std::sync::Arc::new(InMemory::new()))); - let mut connection = Connection::open_in_memory().unwrap(); - connection - .execute_batch("PRAGMA foreign_keys = ON") - .unwrap(); - let transaction = connection.transaction().unwrap(); - install_blob_schema(&transaction).unwrap(); - let key = b"artifacts/build.log".to_vec(); - let upload_id = [1; 16]; - assert_eq!( - blob_mutate( - &transaction, - 1, - 1, - &BlobMutation::Begin { - key: key.clone(), - upload_id, - condition: BlobCondition::Missing, - content_type: Some("text/plain".into()), - metadata: b"build=42".to_vec(), - expires_at_ms: 60_001, - }, - ) - .unwrap(), - BlobMutationOutcome::Begun - ); - for (part_number, payload) in [(1, b"hello ".as_slice()), (2, b"world".as_slice())] { - let digest = part_digest(payload); - artifacts.put_part(digest, payload).await.unwrap(); - assert!(matches!( - blob_mutate( - &transaction, - 2, - 2, - &BlobMutation::PutPartRef { - key: key.clone(), - upload_id, - part_number, - digest, - size: payload.len() as u32, - }, - ) - .unwrap(), - BlobMutationOutcome::PartStored { .. } - )); - } - let BlobMutationOutcome::Committed { etag, size } = blob_mutate( - &transaction, - 3, - 3, - &BlobMutation::Complete { - key: key.clone(), - upload_id, - part_count: 2, - }, - ) - .unwrap() else { - panic!("blob did not commit"); - }; - assert_eq!(size, 11); - let BlobQueryResult::Read(Some(read)) = blob_query( - &transaction, - &BlobQuery::Read { - key: key.clone(), - offset: 3, - limit: 5, - }, - ) - .unwrap() else { - panic!("blob range was not returned"); - }; - let mut bytes = Vec::new(); - for part in &read.parts { - let payload = artifacts.read_part(part.digest, part.size).await.unwrap(); - let start = 3_u64.saturating_sub(part.offset) as usize; - let end = 8_u64.saturating_sub(part.offset).min(u64::from(part.size)) as usize; - bytes.extend_from_slice(&payload[start..end]); - } - assert_eq!(bytes, b"lo wo"); - assert_eq!(read.metadata.etag, etag); - let BlobQueryResult::List(page) = blob_query( - &transaction, - &BlobQuery::List { - prefix: b"artifacts/".to_vec(), - after: None, - limit: 10, - }, - ) - .unwrap() else { - panic!("blob list was not returned"); - }; - assert_eq!(page.objects.len(), 1); - assert_eq!(page.objects[0].key, key); - - assert_eq!( - blob_mutate( - &transaction, - 4, - 4, - &BlobMutation::Delete { - key, - condition: BlobCondition::Etag([9; 32]), - }, - ) - .unwrap(), - BlobMutationOutcome::Conflict - ); -} - -#[tokio::test] -async fn object_store_sweep_keeps_live_parts_and_reclaims_old_orphans() { - use crab_storage::{GLOBAL_PREFIX, Store, global_content_prefix}; - use object_store::memory::InMemory; - - let store = Store::new(std::sync::Arc::new(InMemory::new())); - let artifacts = BlobArtifactStore::new(store.clone()); - let live_payload = b"live"; - let orphan_payload = b"orphan"; - let live_digest = part_digest(live_payload); - let orphan_digest = part_digest(orphan_payload); - artifacts.put_part(live_digest, live_payload).await.unwrap(); - artifacts - .put_part(orphan_digest, orphan_payload) - .await - .unwrap(); - - let report = artifacts - .sweep_unreferenced(&BTreeSet::from([live_digest]), i64::MAX) - .await - .unwrap(); - assert_eq!(report.scanned(), 2); - assert_eq!(report.deleted(), 1); - assert!(!report.has_more()); - - let objects = store - .list_prefix(&global_content_prefix(GLOBAL_PREFIX, BLOB_PART_KIND)) - .await - .unwrap(); - assert_eq!(objects.len(), 1); - assert!( - objects[0] - .location - .to_string() - .ends_with(&blake3::Hash::from_bytes(live_digest).to_hex().to_string()) - ); -} - -#[tokio::test] -async fn object_store_sweep_reaches_orphans_beyond_live_entries() { - use crab_storage::Store; - use object_store::memory::InMemory; - - let store = Store::new(std::sync::Arc::new(InMemory::new())); - let artifacts = BlobArtifactStore::new(store); - let mut live_digests = BTreeSet::new(); - for index in 0_u32..129 { - let payload = index.to_be_bytes(); - let digest = part_digest(&payload); - artifacts.put_part(digest, &payload).await.unwrap(); - live_digests.insert(digest); - } - let orphan_payload = b"orphan past the scan budget"; - artifacts - .put_part(part_digest(orphan_payload), orphan_payload) - .await - .unwrap(); - - let report = artifacts - .sweep_unreferenced(&live_digests, i64::MAX) - .await - .unwrap(); - assert_eq!(report.scanned(), 130); - assert_eq!(report.deleted(), 1); - assert!(!report.has_more()); -} - -#[tokio::test] -async fn object_store_sweep_bounds_deletions_and_finishes_on_retry() { - use crab_storage::Store; - use object_store::memory::InMemory; - - let store = Store::new(std::sync::Arc::new(InMemory::new())); - let artifacts = BlobArtifactStore::new(store); - for index in 0_u32..129 { - let payload = index.to_be_bytes(); - artifacts - .put_part(part_digest(&payload), &payload) - .await - .unwrap(); - } - - let first = artifacts - .sweep_unreferenced(&BTreeSet::new(), i64::MAX) - .await - .unwrap(); - assert_eq!(first.deleted(), MAX_BLOB_GC_DELETIONS); - assert!(first.has_more()); - - let second = artifacts - .sweep_unreferenced(&BTreeSet::new(), i64::MAX) - .await - .unwrap(); - assert_eq!(second.deleted(), 1); - assert!(!second.has_more()); -} - -#[test] -fn checked_in_blob_schema_matches_runtime_schema() { - assert_eq!( - BLOB_SCHEMA, - include_str!("../../../docs/contracts/blob.sql") - ); -} - -#[test] -fn blob_schema_keeps_part_bytes_out_of_sqlite() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - install_blob_schema(&transaction).unwrap(); - let columns = transaction - .prepare("PRAGMA table_info(blob_parts)") - .unwrap() - .query_map([], |row| row.get::<_, String>(1)) - .unwrap() - .collect::, _>>() - .unwrap(); - assert_eq!( - columns, - ["upload_id", "part_number", "digest", "size", "byte_offset"] - ); -} - -#[test] -fn upload_lifetime_uses_request_issue_time_but_rejects_expired_acceptance() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - install_blob_schema(&transaction).unwrap(); - let mutation = BlobMutation::Begin { - key: b"logs/issue-time".to_vec(), - upload_id: [2; 16], - condition: BlobCondition::Missing, - content_type: None, - metadata: Vec::new(), - expires_at_ms: 60_001, - }; - - assert_eq!( - blob_mutate(&transaction, 1_500, 1, &mutation).unwrap(), - BlobMutationOutcome::Begun - ); - assert!(matches!( - blob_mutate( - &transaction, - 60_001, - 1, - &BlobMutation::Begin { - key: b"logs/issue-time".to_vec(), - upload_id: [3; 16], - condition: BlobCondition::Missing, - content_type: None, - metadata: Vec::new(), - expires_at_ms: 60_001, - }, - ), - Err(Error::Command("blob upload has already expired")) - )); -} - -fn begin(key: &[u8], metadata: &[u8], expires_at_ms: i64, upload_id: [u8; 16]) -> BlobMutation { - BlobMutation::Begin { - key: key.to_vec(), - upload_id, - condition: BlobCondition::Missing, - content_type: None, - metadata: metadata.to_vec(), - expires_at_ms, - } -} - -fn blob_connection() -> Connection { - let connection = Connection::open_in_memory().unwrap(); - let transaction = connection.unchecked_transaction().unwrap(); - install_blob_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - connection -} - -#[test] -fn key_bounds_accept_the_documented_range() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - assert_eq!( - blob_mutate(&transaction, 1, 1, &begin(b"k", &[], 60_001, [10; 16])).unwrap(), - BlobMutationOutcome::Begun - ); - assert_eq!( - blob_mutate( - &transaction, - 1, - 1, - &begin(&[b'k'; 1_024], &[], 60_001, [11; 16]) - ) - .unwrap(), - BlobMutationOutcome::Begun - ); -} - -#[test] -fn key_bounds_reject_empty_and_oversized_keys() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - assert!(blob_mutate(&transaction, 1, 1, &begin(b"", &[], 60_001, [12; 16])).is_err()); - assert!( - blob_mutate( - &transaction, - 1, - 1, - &begin(&[b'k'; 1_025], &[], 60_001, [13; 16]) - ) - .is_err() - ); -} - -#[test] -fn metadata_bounds_accept_exactly_eight_kib() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - assert_eq!( - blob_mutate( - &transaction, - 1, - 1, - &begin(b"k", &vec![b'm'; 8 * 1_024], 60_001, [14; 16]), - ) - .unwrap(), - BlobMutationOutcome::Begun - ); -} - -#[test] -fn metadata_bounds_reject_more_than_eight_kib() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - assert!( - blob_mutate( - &transaction, - 1, - 1, - &begin(b"k", &vec![b'm'; 8 * 1_024 + 1], 60_001, [15; 16]), - ) - .is_err() - ); -} - -#[test] -fn part_size_bounds_reject_more_than_256_kib() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - blob_mutate(&transaction, 1, 1, &begin(b"k", &[], 60_001, [16; 16])).unwrap(); - assert!( - blob_mutate( - &transaction, - 1, - 1, - &BlobMutation::PutPartRef { - key: b"k".to_vec(), - upload_id: [16; 16], - part_number: 1, - digest: [17; 32], - size: MAX_BLOB_PART_BYTES as u32 + 1, - }, - ) - .is_err() - ); -} - -#[test] -fn upload_lifetime_accepts_the_one_minute_and_seven_day_bounds() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - assert_eq!( - blob_mutate(&transaction, 1, 1, &begin(b"k", &[], 60_001, [18; 16])).unwrap(), - BlobMutationOutcome::Begun - ); - assert_eq!( - blob_mutate( - &transaction, - 1, - 1, - &begin(b"k", &[], 7 * 24 * 60 * 60_000 + 1, [19; 16]), - ) - .unwrap(), - BlobMutationOutcome::Begun - ); -} - -#[test] -fn upload_lifetime_rejects_below_one_minute_and_past_seven_days() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - assert!(blob_mutate(&transaction, 1, 1, &begin(b"k", &[], 60_000, [20; 16])).is_err()); - assert!( - blob_mutate( - &transaction, - 1, - 1, - &begin(b"k", &[], 7 * 24 * 60 * 60_000 + 2, [21; 16]), - ) - .is_err() - ); -} - -#[test] -fn list_rejects_limits_outside_one_to_128() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - assert!( - blob_query( - &transaction, - &BlobQuery::List { - prefix: b"logs/".to_vec(), - after: None, - limit: 0, - }, - ) - .is_err() - ); - assert!( - blob_query( - &transaction, - &BlobQuery::List { - prefix: b"logs/".to_vec(), - after: None, - limit: 129, - }, - ) - .is_err() - ); -} - -#[test] -fn list_accepts_the_full_page_limit() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - let result = blob_query( - &transaction, - &BlobQuery::List { - prefix: b"logs/".to_vec(), - after: None, - limit: 128, - }, - ) - .unwrap(); - let BlobQueryResult::List(page) = result else { - panic!("list query must return a page"); - }; - assert!(page.objects.is_empty()); -} - -#[test] -fn part_number_bounds_reject_outside_one_to_4096() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - blob_mutate(&transaction, 1, 1, &begin(b"k", &[], 60_001, [22; 16])).unwrap(); - for part_number in [0_u32, MAX_BLOB_PARTS + 1] { - assert!( - blob_mutate( - &transaction, - 1, - 1, - &BlobMutation::PutPartRef { - key: b"k".to_vec(), - upload_id: [22; 16], - part_number, - digest: [23; 32], - size: 1, - }, - ) - .is_err() - ); - } -} - -#[test] -fn completion_part_count_bounds_reject_outside_one_to_4096() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - blob_mutate(&transaction, 1, 1, &begin(b"k", &[], 60_001, [24; 16])).unwrap(); - for part_count in [0_u32, MAX_BLOB_PARTS + 1] { - assert!( - blob_mutate( - &transaction, - 1, - 1, - &BlobMutation::Complete { - key: b"k".to_vec(), - upload_id: [24; 16], - part_count, - }, - ) - .is_err() - ); - } -} - -#[test] -fn range_read_rejects_more_than_512_kib() { - let mut connection = blob_connection(); - let transaction = connection.transaction().unwrap(); - blob_mutate(&transaction, 1, 1, &begin(b"k", &[], 60_001, [25; 16])).unwrap(); - blob_mutate( - &transaction, - 1, - 1, - &BlobMutation::PutPartRef { - key: b"k".to_vec(), - upload_id: [25; 16], - part_number: 1, - digest: [26; 32], - size: 4, - }, - ) - .unwrap(); - assert!(matches!( - blob_mutate( - &transaction, - 1, - 1, - &BlobMutation::Complete { - key: b"k".to_vec(), - upload_id: [25; 16], - part_count: 1, - }, - ) - .unwrap(), - BlobMutationOutcome::Committed { .. } - )); - assert!( - blob_query( - &transaction, - &BlobQuery::Read { - key: b"k".to_vec(), - offset: 0, - limit: MAX_BLOB_READ_BYTES + 1, - }, - ) - .is_err() - ); -} diff --git a/crates/crab-cell-runtime/src/primitives/capacity.rs b/crates/crab-cell-runtime/src/primitives/capacity.rs deleted file mode 100644 index 5a472d6b5..000000000 --- a/crates/crab-cell-runtime/src/primitives/capacity.rs +++ /dev/null @@ -1,108 +0,0 @@ -//! Durable SQLite page reservations shared by application commands and runtime writes. - -use rusqlite::{Connection, OptionalExtension, Transaction}; - -use crate::{Error, Result}; - -/// Schema installed by Cells that reserve database capacity for deferred work. -pub const SCHEMA: &str = include_str!("../migrations/capacity.sql"); - -pub(crate) fn reserve(transaction: &Transaction<'_>, key: &[u8], bytes: u64) -> Result<()> { - if key.is_empty() || key.len() > 128 || bytes == 0 { - return Err(Error::Command("invalid database capacity reservation")); - } - let page_size: u64 = transaction.query_row("PRAGMA page_size", [], |row| row.get(0))?; - let pages = i64::try_from(bytes.div_ceil(page_size)) - .map_err(|_| Error::Capacity("database reservation is too large"))?; - let maximum: i64 = transaction.query_row("PRAGMA max_page_count", [], |row| row.get(0))?; - if pages > maximum { - return Err(Error::Capacity( - "database reservation exceeds the Cell limit", - )); - } - let existing: Option = transaction - .query_row( - "SELECT pages FROM capacity_reservations WHERE reservation_key = ?1", - [key], - |row| row.get(0), - ) - .optional()?; - if let Some(existing) = existing { - return if existing == pages { - Ok(()) - } else { - Err(Error::Command("database reservation identity changed")) - }; - } - transaction.execute( - "INSERT INTO capacity_reservations(reservation_key, pages) VALUES (?1, ?2)", - (key, pages), - )?; - if transaction.execute( - "UPDATE capacity_total SET pages = pages + ?1 WHERE singleton = 1", - [pages], - )? != 1 - { - return Err(Error::Command("database reservation total is missing")); - } - validate(transaction) -} - -pub(crate) fn release(transaction: &Transaction<'_>, key: &[u8]) -> Result { - if key.is_empty() || key.len() > 128 { - return Err(Error::Command("invalid database capacity reservation key")); - } - let pages: Option = transaction - .query_row( - "DELETE FROM capacity_reservations WHERE reservation_key = ?1 RETURNING pages", - [key], - |row| row.get(0), - ) - .optional()?; - let Some(pages) = pages else { return Ok(false) }; - if transaction.execute( - "UPDATE capacity_total SET pages = pages - ?1 WHERE singleton = 1 AND pages >= ?1", - [pages], - )? != 1 - { - return Err(Error::Command("database reservation total is invalid")); - } - Ok(true) -} - -// Run after the runtime receipt and metadata writes, not just after the handler. -// Otherwise those writes, effect delivery, or migration could spend capacity -// already promised to another command. Refusal rolls back the whole transaction. -pub(crate) fn validate(connection: &Connection) -> Result<()> { - let installed: i64 = connection.query_row( - "SELECT COUNT(*) FROM sqlite_schema WHERE type = 'table' AND name IN ('capacity_reservations', 'capacity_total')", - [], |row| row.get(0), - )?; - if installed == 0 { - return Ok(()); - } - if installed != 2 { - return Err(Error::Command("database reservation schema is incomplete")); - } - let reserved: i64 = connection.query_row( - "SELECT pages FROM capacity_total WHERE singleton = 1", - [], - |row| row.get(0), - )?; - if reserved == 0 { - return Ok(()); - } - let pages: i64 = connection.query_row("PRAGMA page_count", [], |row| row.get(0))?; - let free: i64 = connection.query_row("PRAGMA freelist_count", [], |row| row.get(0))?; - let maximum: i64 = connection.query_row("PRAGMA max_page_count", [], |row| row.get(0))?; - let available = maximum - .checked_sub(pages) - .and_then(|value| value.checked_add(free)) - .ok_or(Error::Command("invalid database page accounting"))?; - if reserved > available { - return Err(Error::Capacity( - "database pages are reserved for deferred work", - )); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/primitives/cron.rs b/crates/crab-cell-runtime/src/primitives/cron.rs deleted file mode 100644 index 35592b688..000000000 --- a/crates/crab-cell-runtime/src/primitives/cron.rs +++ /dev/null @@ -1,857 +0,0 @@ -//! Cron primitive: fixed-interval schedules with an explicit first due time. -use rusqlite::{Connection, OptionalExtension, Transaction}; - -use crate::codec::{BoundedEncoder, WireValue}; -use crate::identity::{CellTarget, NamespaceId}; -use crate::primitives::effects::EffectBatch; -use crate::primitives::effects::EffectCommandIntent; -use crate::{Error, Result}; - -mod api; - -pub use api::{CronCommand, CronModule, CronNamespace, CronQueryCommand, register_cron}; - -const CRON_SCHEMA: &str = include_str!("../migrations/cron.sql"); -pub(crate) const CRON_TABLE: &str = "cron_schedules"; -const MIN_INTERVAL_MS: u64 = 1_000; -const MAX_INTERVAL_MS: u64 = 365 * 24 * 60 * 60 * 1_000; -const MAX_PAYLOAD_BYTES: usize = 256 * 1_024; -const MAX_FUTURE_MS: i64 = 5 * 365 * 24 * 60 * 60 * 1_000; -const EFFECT_LIFETIME_MS: i64 = 7 * 24 * 60 * 60 * 1_000; - -/// Compile-time destination contract for Cron invocations. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct CronTarget { - module: &'static str, - namespace: NamespaceId, - command_id: u32, - codec_version: u32, - input_limit: u32, -} - -impl CronTarget { - /// Declares one delivery target: its module, namespace, command id, codec - /// version, and input limit. - #[must_use] - pub const fn new( - module: &'static str, - namespace: NamespaceId, - command_id: u32, - codec_version: u32, - input_limit: u32, - ) -> Self { - Self { - module, - namespace, - command_id, - codec_version, - input_limit, - } - } - - pub(crate) const fn module(self) -> &'static str { - self.module - } - pub(crate) const fn namespace(self) -> NamespaceId { - self.namespace - } - pub(crate) const fn command_id(self) -> u32 { - self.command_id - } - pub(crate) const fn codec_version(self) -> u32 { - self.codec_version - } - pub(crate) const fn input_limit(self) -> u32 { - self.input_limit - } -} - -/// Payload delivered exactly once to a destination inbox for one Cron occurrence. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct CronInvocation { - /// Schedule that produced the occurrence. - pub schedule_id: [u8; 16], - /// Schedule generation the occurrence belongs to. - pub generation: u64, - /// Monotonic occurrence number within the generation. - pub occurrence: u64, - /// Logical time the occurrence was due. - pub scheduled_at_ms: i64, - /// Payload to deliver to the target. - pub payload: Vec, -} - -/// Durable Cron schedule mutation. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum CronMutation { - /// Creates or replaces one schedule. - Upsert { - /// Schedule identity. - schedule_id: [u8; 16], - /// Index of the registered target the schedule fires at. - target_index: u32, - /// Partition key of the destination Cell. - target_partition: Vec, - /// Payload delivered with each occurrence. - payload: Vec, - /// Fixed interval between occurrences. - interval_ms: u64, - /// Logical time of the first due occurrence. - next_due_ms: i64, - }, - /// Stops one schedule from firing without deleting it. - Pause { - /// Schedule to pause. - schedule_id: [u8; 16], - }, - /// Re-enables one paused schedule. - Resume { - /// Schedule to resume. - schedule_id: [u8; 16], - /// Logical time of the next due occurrence. - next_due_ms: i64, - }, - /// Removes one schedule and its pending occurrences. - Delete { - /// Schedule to delete. - schedule_id: [u8; 16], - }, -} - -/// Result of one Cron schedule mutation. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum CronMutationOutcome { - /// The mutation was applied at this generation. - Applied { - /// Schedule generation after the mutation. - generation: u64, - }, - /// The schedule was removed. - Deleted, - /// No schedule matched the identity. - NotFound, -} - -/// Materialized Cron schedule state. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct CronSchedule { - /// Schedule identity. - pub schedule_id: [u8; 16], - /// Index of the registered target the schedule fires at. - pub target_index: u32, - /// Partition key of the destination Cell. - pub target_partition: Vec, - /// Payload delivered with each occurrence. - pub payload: Vec, - /// Fixed interval between occurrences. - pub interval_ms: u64, - /// Logical time of the next due occurrence. - pub next_due_ms: i64, - /// Occurrences fired in this generation. - pub occurrence: u64, - /// Whether the schedule currently fires. - pub enabled: bool, - /// Generation raised by the last mutation. - pub generation: u64, -} - -/// Bounded Cron schedule query. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum CronQuery { - /// Reads one schedule. - Get { - /// Schedule to read. - schedule_id: [u8; 16], - }, - /// Lists schedules in identity order. - List { - /// Schedule identity to continue after, from a previous page. - after: Option<[u8; 16]>, - /// Maximum schedules to return. - limit: u32, - }, -} - -/// Result of a Cron schedule query. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum CronQueryResult { - /// The schedule, absent when no schedule matches. - Get(Option), - /// One page of schedules in identity order. - List { - /// Schedules in identity order. - schedules: Vec, - /// Identity to continue from when the page filled its limit. - next: Option<[u8; 16]>, - }, -} - -/// Installs the exact version-one Cron schema. -pub fn install_cron_schema(transaction: &Transaction<'_>) -> Result<()> { - transaction.execute_batch(CRON_SCHEMA)?; - Ok(()) -} - -/// Applies one durable Cron schedule mutation. -pub fn cron_mutate( - transaction: &Transaction<'_>, - now_ms: i64, - issued_at_ms: i64, - targets: &[CronTarget], - mutation: &CronMutation, -) -> Result { - validate_now(now_ms)?; - validate_now(issued_at_ms)?; - match mutation { - CronMutation::Upsert { - schedule_id, - target_index, - target_partition, - payload, - interval_ms, - next_due_ms, - } => { - let target = target_at(targets, *target_index)?; - validate_schedule( - issued_at_ms, - target, - target_partition, - payload, - *interval_ms, - *next_due_ms, - )?; - transaction.execute( - "INSERT INTO cron_schedules(schedule_id, target_index, target_partition, payload, interval_ms, next_due_ms, occurrence, enabled, generation, updated_at_ms) VALUES (?1, ?2, ?3, ?4, ?5, ?6, 0, 1, 1, ?7) ON CONFLICT(schedule_id) DO UPDATE SET target_index = excluded.target_index, target_partition = excluded.target_partition, payload = excluded.payload, interval_ms = excluded.interval_ms, next_due_ms = excluded.next_due_ms, occurrence = 0, enabled = 1, generation = cron_schedules.generation + 1, updated_at_ms = excluded.updated_at_ms", - (schedule_id.as_slice(), i64::from(*target_index), target_partition, payload, i64::try_from(*interval_ms).map_err(|_| Error::Command("cron interval overflow"))?, *next_due_ms, now_ms), - )?; - Ok(CronMutationOutcome::Applied { - generation: schedule_generation(transaction, *schedule_id)?, - }) - } - CronMutation::Pause { schedule_id } => { - set_enabled(transaction, now_ms, *schedule_id, false, None) - } - CronMutation::Resume { - schedule_id, - next_due_ms, - } => { - validate_due(issued_at_ms, *next_due_ms)?; - set_enabled(transaction, now_ms, *schedule_id, true, Some(*next_due_ms)) - } - CronMutation::Delete { schedule_id } => { - let changed = transaction.execute( - "DELETE FROM cron_schedules WHERE schedule_id = ?1", - [schedule_id.as_slice()], - )?; - Ok(if changed == 1 { - CronMutationOutcome::Deleted - } else { - CronMutationOutcome::NotFound - }) - } - } -} - -/// Reads one or one bounded page of Cron schedules. -pub fn cron_query(connection: &Connection, query: &CronQuery) -> Result { - match query { - CronQuery::Get { schedule_id } => Ok(CronQueryResult::Get(load_schedule( - connection, - *schedule_id, - )?)), - CronQuery::List { after, limit } => { - if !(1..=128).contains(limit) { - return Err(Error::Command("cron list limit must be in 1..=128")); - } - let after = after.unwrap_or([0; 16]); - let mut statement = connection.prepare( - "SELECT schedule_id, target_index, target_partition, payload, interval_ms, next_due_ms, occurrence, enabled, generation FROM cron_schedules WHERE schedule_id > ?1 ORDER BY schedule_id LIMIT ?2", - )?; - let rows = - statement.query_map((after.as_slice(), i64::from(*limit) + 1), decode_schedule)?; - let mut schedules = Vec::with_capacity(*limit as usize); - let mut bytes = 0_usize; - let mut truncated = false; - for row in rows { - let schedule = row?; - let schedule_bytes = schedule - .target_partition - .len() - .checked_add(schedule.payload.len()) - .and_then(|value| value.checked_add(96)) - .ok_or(Error::Command("cron list byte count overflow"))?; - if schedules.len() == *limit as usize - || bytes.saturating_add(schedule_bytes) > 512 * 1024 - { - truncated = true; - break; - } - bytes += schedule_bytes; - schedules.push(schedule); - } - let next = truncated - .then(|| schedules.last().map(|value| value.schedule_id)) - .flatten(); - Ok(CronQueryResult::List { schedules, next }) - } - } -} - -pub(crate) fn cron_fire_due_bounded( - transaction: &Transaction<'_>, - effects: &mut EffectBatch, - source: &CellTarget, - now_ms: i64, - targets: &[CronTarget], - limit: usize, -) -> Result { - let mut processed = 0; - while processed < limit { - let row = transaction - .query_row( - "SELECT schedule_id, target_index, target_partition, payload, interval_ms, next_due_ms, occurrence, generation FROM cron_schedules INDEXED BY cron_due WHERE enabled = 1 AND next_due_ms <= ?1 ORDER BY next_due_ms, schedule_id LIMIT 1", - [now_ms], - |row| Ok((row.get::<_, Vec>(0)?, row.get::<_, i64>(1)?, row.get::<_, Vec>(2)?, row.get::<_, Vec>(3)?, row.get::<_, i64>(4)?, row.get::<_, i64>(5)?, row.get::<_, i64>(6)?, row.get::<_, i64>(7)?)), - ) - .optional()?; - let Some(( - schedule_id, - target_index, - partition, - payload, - interval_ms, - scheduled_at_ms, - occurrence, - generation, - )) = row - else { - break; - }; - let schedule_id: [u8; 16] = schedule_id - .try_into() - .map_err(|_| Error::Command("invalid stored cron schedule ID"))?; - let target_index = u32::try_from(target_index) - .map_err(|_| Error::Command("invalid stored cron target index"))?; - let target = target_at(targets, target_index)?; - let occurrence = u64::try_from(occurrence) - .map_err(|_| Error::Command("invalid stored cron occurrence"))?; - let generation = u64::try_from(generation) - .map_err(|_| Error::Command("invalid stored cron generation"))?; - let next_occurrence = occurrence - .checked_add(1) - .ok_or(Error::Command("cron occurrence overflow"))?; - let invocation = CronInvocation { - schedule_id, - generation, - occurrence: next_occurrence, - scheduled_at_ms, - payload, - }; - let mut encoder = BoundedEncoder::new(target.input_limit())?; - invocation.encode(&mut encoder)?; - let destination = CellTarget::new( - source.tenant(), - source.application(), - target.namespace(), - &partition, - )?; - effects.insert_command( - transaction, - &EffectCommandIntent { - target: destination, - command_id: target.command_id(), - codec_version: target.codec_version(), - input: encoder.finish(), - expires_at_ms: now_ms - .checked_add(EFFECT_LIFETIME_MS) - .ok_or(Error::Command("cron effect expiry overflow"))?, - }, - )?; - let next_due_ms = scheduled_at_ms - .checked_add(interval_ms) - .ok_or(Error::Command("cron due time overflow"))?; - if transaction.execute( - "UPDATE cron_schedules SET next_due_ms = ?1, occurrence = ?2 WHERE schedule_id = ?3 AND enabled = 1 AND next_due_ms = ?4 AND occurrence = ?5 AND generation = ?6", - (next_due_ms, i64::try_from(next_occurrence).map_err(|_| Error::Command("cron occurrence overflow"))?, schedule_id.as_slice(), scheduled_at_ms, i64::try_from(occurrence).map_err(|_| Error::Command("cron occurrence overflow"))?, i64::try_from(generation).map_err(|_| Error::Command("cron generation overflow"))?), - )? != 1 { return Err(Error::Command("cron schedule changed during serialized fire")); } - processed += 1; - } - Ok(processed) -} - -fn set_enabled( - transaction: &Transaction<'_>, - now_ms: i64, - schedule_id: [u8; 16], - enabled: bool, - next_due_ms: Option, -) -> Result { - let enabled_value = if enabled { 1_i64 } else { 0_i64 }; - let changed = transaction.execute( - "UPDATE cron_schedules SET enabled = ?1, next_due_ms = coalesce(?2, next_due_ms), generation = generation + CASE WHEN enabled = ?1 AND (?2 IS NULL OR next_due_ms = ?2) THEN 0 ELSE 1 END, updated_at_ms = ?3 WHERE schedule_id = ?4", - (enabled_value, next_due_ms, now_ms, schedule_id.as_slice()), - )?; - if changed == 0 { - return Ok(CronMutationOutcome::NotFound); - } - Ok(CronMutationOutcome::Applied { - generation: schedule_generation(transaction, schedule_id)?, - }) -} - -fn schedule_generation(transaction: &Transaction<'_>, schedule_id: [u8; 16]) -> Result { - let value = transaction.query_row( - "SELECT generation FROM cron_schedules WHERE schedule_id = ?1", - [schedule_id.as_slice()], - |row| row.get::<_, i64>(0), - )?; - u64::try_from(value).map_err(|_| Error::Command("invalid stored cron generation")) -} - -fn load_schedule(connection: &Connection, schedule_id: [u8; 16]) -> Result> { - connection.query_row( - "SELECT schedule_id, target_index, target_partition, payload, interval_ms, next_due_ms, occurrence, enabled, generation FROM cron_schedules WHERE schedule_id = ?1", - [schedule_id.as_slice()], decode_schedule, - ).optional().map_err(Into::into) -} - -fn decode_schedule(row: &rusqlite::Row<'_>) -> rusqlite::Result { - let schedule_id = row - .get::<_, Vec>(0)? - .try_into() - .map_err(|_| rusqlite::Error::InvalidQuery)?; - Ok(CronSchedule { - schedule_id, - target_index: u32::try_from(row.get::<_, i64>(1)?) - .map_err(|_| rusqlite::Error::InvalidQuery)?, - target_partition: row.get(2)?, - payload: row.get(3)?, - interval_ms: u64::try_from(row.get::<_, i64>(4)?) - .map_err(|_| rusqlite::Error::InvalidQuery)?, - next_due_ms: row.get(5)?, - occurrence: u64::try_from(row.get::<_, i64>(6)?) - .map_err(|_| rusqlite::Error::InvalidQuery)?, - enabled: row.get::<_, i64>(7)? != 0, - generation: u64::try_from(row.get::<_, i64>(8)?) - .map_err(|_| rusqlite::Error::InvalidQuery)?, - }) -} - -fn validate_schedule( - issued_at_ms: i64, - target: CronTarget, - partition: &[u8], - payload: &[u8], - interval_ms: u64, - next_due_ms: i64, -) -> Result<()> { - if partition.len() > 1_024 || payload.len() > MAX_PAYLOAD_BYTES { - return Err(Error::Command("cron target or payload exceeds limits")); - } - if !(MIN_INTERVAL_MS..=MAX_INTERVAL_MS).contains(&interval_ms) { - return Err(Error::Command( - "cron interval must be between one second and one year", - )); - } - validate_due(issued_at_ms, next_due_ms)?; - let wrapper_bytes = payload - .len() - .checked_add(64) - .ok_or(Error::Command("cron invocation size overflow"))?; - if wrapper_bytes > target.input_limit() as usize { - return Err(Error::Command("cron invocation exceeds target input limit")); - } - Ok(()) -} - -fn validate_due(issued_at_ms: i64, due_ms: i64) -> Result<()> { - if due_ms < issued_at_ms || due_ms > issued_at_ms.saturating_add(MAX_FUTURE_MS) { - return Err(Error::Command("cron due time is outside five-year window")); - } - Ok(()) -} - -fn target_at(targets: &[CronTarget], index: u32) -> Result { - targets - .get(index as usize) - .copied() - .ok_or(Error::Command("cron target index is unavailable")) -} - -fn validate_now(now_ms: i64) -> Result<()> { - if now_ms < 0 { - return Err(Error::Command("negative cron logical time")); - } - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::identity::IncarnationId; - use crate::identity::{ApplicationId, TenantId}; - use crab_ltx::rusqlite::Connection; - - const SOURCE_NAMESPACE: NamespaceId = NamespaceId::from_bytes([1; 16]); - const TARGET_NAMESPACE: NamespaceId = NamespaceId::from_bytes([2; 16]); - const TARGETS: &[CronTarget] = &[CronTarget::new( - "cron-destination", - TARGET_NAMESPACE, - 9, - 1, - 1024, - )]; - - #[test] - fn due_occurrence_and_advance_are_one_transaction() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - let source = CellTarget::new( - TenantId::from_bytes([3; 16]), - ApplicationId::from_bytes([4; 16]), - SOURCE_NAMESPACE, - b"cron-shard", - ) - .unwrap(); - crate::cell::schema::install_runtime_schema_in( - &transaction, - source.cell_id(), - IncarnationId::from_bytes([5; 16]), - 1, - ) - .unwrap(); - install_cron_schema(&transaction).unwrap(); - cron_mutate( - &transaction, - 200, - 10, - TARGETS, - &CronMutation::Upsert { - schedule_id: [6; 16], - target_index: 0, - target_partition: b"destination".to_vec(), - payload: b"compact".to_vec(), - interval_ms: 1_000, - next_due_ms: 100, - }, - ) - .unwrap(); - let mut effects = EffectBatch::new(&transaction, &source, 1, 200).unwrap(); - assert_eq!( - cron_fire_due_bounded(&transaction, &mut effects, &source, 200, TARGETS, 8).unwrap(), - 1 - ); - assert_eq!( - transaction - .query_row("SELECT count(*) FROM sys_effects", [], |row| { - row.get::<_, i64>(0) - }) - .unwrap(), - 1 - ); - let schedule = load_schedule(&transaction, [6; 16]).unwrap().unwrap(); - assert_eq!(schedule.occurrence, 1); - assert_eq!(schedule.next_due_ms, 1_100); - let CronQueryResult::List { schedules, next } = cron_query( - &transaction, - &CronQuery::List { - after: None, - limit: 10, - }, - ) - .unwrap() else { - panic!("cron list was not returned"); - }; - assert_eq!(schedules.len(), 1); - assert_eq!(next, None); - assert_eq!( - cron_fire_due_bounded(&transaction, &mut effects, &source, 200, TARGETS, 8).unwrap(), - 0 - ); - } - - #[test] - fn checked_in_cron_schema_matches_runtime_schema() { - assert_eq!(CRON_SCHEMA, include_str!("../../docs/contracts/cron.sql")); - } - - #[test] - fn catch_up_fires_each_missed_occurrence_with_its_own_effect() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - let source = CellTarget::new( - TenantId::from_bytes([7; 16]), - ApplicationId::from_bytes([8; 16]), - SOURCE_NAMESPACE, - b"cron-catch-up", - ) - .unwrap(); - crate::cell::schema::install_runtime_schema_in( - &transaction, - source.cell_id(), - IncarnationId::from_bytes([9; 16]), - 1, - ) - .unwrap(); - install_cron_schema(&transaction).unwrap(); - - let upsert = |schedule_id: [u8; 16], next_due_ms: i64| CronMutation::Upsert { - schedule_id, - target_index: 0, - target_partition: b"destination".to_vec(), - payload: b"compact".to_vec(), - interval_ms: 1_000, - next_due_ms, - }; - let overdue = [11; 16]; - let paused = [12; 16]; - let future = [13; 16]; - cron_mutate(&transaction, 200, 10, TARGETS, &upsert(overdue, 1_000)).unwrap(); - cron_mutate(&transaction, 200, 10, TARGETS, &upsert(paused, 1_000)).unwrap(); - cron_mutate(&transaction, 200, 10, TARGETS, &upsert(future, 9_000)).unwrap(); - cron_mutate( - &transaction, - 200, - 10, - TARGETS, - &CronMutation::Pause { - schedule_id: paused, - }, - ) - .unwrap(); - - let mut effects = EffectBatch::new(&transaction, &source, 1, 4_000).unwrap(); - assert_eq!( - cron_fire_due_bounded(&transaction, &mut effects, &source, 4_000, TARGETS, 8).unwrap(), - 4, - "every occurrence that came due fires once" - ); - let schedule = load_schedule(&transaction, overdue).unwrap().unwrap(); - assert_eq!( - (schedule.next_due_ms, schedule.occurrence), - (5_000, 4), - "each fire advances the schedule by exactly one interval" - ); - let paused = load_schedule(&transaction, paused).unwrap().unwrap(); - assert_eq!((paused.next_due_ms, paused.occurrence), (1_000, 0)); - let future = load_schedule(&transaction, future).unwrap().unwrap(); - assert_eq!((future.next_due_ms, future.occurrence), (9_000, 0)); - - let (effect_rows, distinct): (i64, i64) = transaction - .query_row( - "SELECT count(*), count(DISTINCT effect_id) FROM sys_effects", - [], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .unwrap(); - assert_eq!( - effect_rows, 4, - "each occurrence publishes its own durable effect" - ); - assert_eq!( - distinct, 4, - "each occurrence keeps a distinct effect identity" - ); - - assert_eq!( - cron_fire_due_bounded(&transaction, &mut effects, &source, 4_000, TARGETS, 8).unwrap(), - 0, - "a settled schedule does not fire the same occurrences twice" - ); - } - - #[test] - fn fire_budget_stops_at_the_tick_limit_with_occurrences_left_due() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - let source = CellTarget::new( - TenantId::from_bytes([3; 16]), - ApplicationId::from_bytes([4; 16]), - SOURCE_NAMESPACE, - b"cron-shard", - ) - .unwrap(); - crate::cell::schema::install_runtime_schema_in( - &transaction, - source.cell_id(), - IncarnationId::from_bytes([5; 16]), - 1, - ) - .unwrap(); - install_cron_schema(&transaction).unwrap(); - cron_mutate( - &transaction, - 200, - 10, - TARGETS, - &CronMutation::Upsert { - schedule_id: [21; 16], - target_index: 0, - target_partition: b"destination".to_vec(), - payload: b"compact".to_vec(), - interval_ms: 1_000, - next_due_ms: 1_000, - }, - ) - .unwrap(); - let mut effects = EffectBatch::new(&transaction, &source, 1, 4_000).unwrap(); - assert_eq!( - cron_fire_due_bounded(&transaction, &mut effects, &source, 4_000, TARGETS, 1).unwrap(), - 1, - "one tick fires at most its budget" - ); - let schedule = load_schedule(&transaction, [21; 16]).unwrap().unwrap(); - assert_eq!( - (schedule.next_due_ms, schedule.occurrence), - (2_000, 1), - "the remaining occurrences stay due for the next tick" - ); - } - - #[test] - fn unavailable_target_index_fails_the_fire_instead_of_skipping() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - let source = CellTarget::new( - TenantId::from_bytes([3; 16]), - ApplicationId::from_bytes([4; 16]), - SOURCE_NAMESPACE, - b"cron-shard", - ) - .unwrap(); - crate::cell::schema::install_runtime_schema_in( - &transaction, - source.cell_id(), - IncarnationId::from_bytes([5; 16]), - 1, - ) - .unwrap(); - install_cron_schema(&transaction).unwrap(); - transaction - .execute( - "INSERT INTO cron_schedules(schedule_id, target_index, target_partition, payload, interval_ms, next_due_ms, occurrence, enabled, generation, updated_at_ms) VALUES (?1, 1, X'00', X'00', 1000, 1000, 0, 1, 1, 10)", - [[22_u8; 16].as_slice()], - ) - .unwrap(); - let mut effects = EffectBatch::new(&transaction, &source, 1, 2_000).unwrap(); - assert!( - cron_fire_due_bounded(&transaction, &mut effects, &source, 2_000, TARGETS, 1).is_err() - ); - } - - #[test] - fn repeated_pause_keeps_the_generation_and_resume_bumps_it() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - crate::cell::schema::install_runtime_schema_in( - &transaction, - CellTarget::new( - TenantId::from_bytes([3; 16]), - ApplicationId::from_bytes([4; 16]), - SOURCE_NAMESPACE, - b"cron-shard", - ) - .unwrap() - .cell_id(), - IncarnationId::from_bytes([5; 16]), - 1, - ) - .unwrap(); - install_cron_schema(&transaction).unwrap(); - let schedule_id = [23; 16]; - cron_mutate( - &transaction, - 200, - 10, - TARGETS, - &CronMutation::Upsert { - schedule_id, - target_index: 0, - target_partition: b"destination".to_vec(), - payload: b"compact".to_vec(), - interval_ms: 1_000, - next_due_ms: 1_000, - }, - ) - .unwrap(); - let paused = cron_mutate( - &transaction, - 200, - 10, - TARGETS, - &CronMutation::Pause { schedule_id }, - ) - .unwrap(); - let repaused = cron_mutate( - &transaction, - 200, - 10, - TARGETS, - &CronMutation::Pause { schedule_id }, - ) - .unwrap(); - let resumed = cron_mutate( - &transaction, - 200, - 10, - TARGETS, - &CronMutation::Resume { - schedule_id, - next_due_ms: 3_000, - }, - ) - .unwrap(); - assert_eq!( - (paused, repaused, resumed), - ( - CronMutationOutcome::Applied { generation: 2 }, - CronMutationOutcome::Applied { generation: 2 }, - CronMutationOutcome::Applied { generation: 3 }, - ) - ); - } - - #[test] - fn upsert_accepts_a_due_exactly_at_the_five_year_boundary() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - install_cron_schema(&transaction).unwrap(); - let outcome = cron_mutate( - &transaction, - 200, - 10, - TARGETS, - &CronMutation::Upsert { - schedule_id: [24; 16], - target_index: 0, - target_partition: b"destination".to_vec(), - payload: b"compact".to_vec(), - interval_ms: 1_000, - next_due_ms: 10 + MAX_FUTURE_MS, - }, - ); - assert!(matches!(outcome, Ok(CronMutationOutcome::Applied { .. }))); - } - - #[test] - fn upsert_rejects_a_due_past_the_five_year_boundary() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - install_cron_schema(&transaction).unwrap(); - let outcome = cron_mutate( - &transaction, - 200, - 10, - TARGETS, - &CronMutation::Upsert { - schedule_id: [25; 16], - target_index: 0, - target_partition: b"destination".to_vec(), - payload: b"compact".to_vec(), - interval_ms: 1_000, - next_due_ms: 11 + MAX_FUTURE_MS, - }, - ); - assert!(outcome.is_err()); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/cron/api.rs b/crates/crab-cell-runtime/src/primitives/cron/api.rs deleted file mode 100644 index 9417f3676..000000000 --- a/crates/crab-cell-runtime/src/primitives/cron/api.rs +++ /dev/null @@ -1,428 +0,0 @@ -use std::marker::PhantomData; - -use crate::cell::catalog::CatalogRole; -use crate::client::{CellClient, Committed, InvocationError, Observed, Receipt}; -use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue, read_fixed}; -use crate::identity::{ - ApplicationId, CellTarget, NamespaceId, TenantId, partition_for_shard, shard_for_scope, -}; -use crate::primitives::maintenance::{MaintenanceModule, register_maintenance}; -use crate::registry::{Command, Query, RegistryBuilder}; -use crate::registry::{CommandContext, CommandResult, QueryContext}; - -use super::{ - CronInvocation, CronMutation, CronMutationOutcome, CronQuery, CronQueryResult, CronSchedule, - MAX_PAYLOAD_BYTES, cron_mutate, cron_query, -}; - -/// Compile-time namespace, targets, and operation IDs for one Cron module. -pub trait CronModule: MaintenanceModule { - /// Namespace that owns this Cron module. - const NAMESPACE: NamespaceId; - /// Command id that mutates cron schedules and occurrence state. - const MUTATE_COMMAND_ID: u32; - /// Query id that reads schedules and occurrences. - const QUERY_ID: u32; -} - -/// Registers Cron bindings and its scheduler Tick command. -pub fn register_cron(registry: &mut RegistryBuilder) -> crate::Result<()> { - registry.bind_cron_module(M::MODULE, M::NAMESPACE, M::CRON_TARGETS)?; - registry.bind_command::>()?; - registry.bind_query::>()?; - register_maintenance::(registry) -} - -/// Typed Cron mutation command. -pub struct CronCommand(PhantomData M>); - -impl Command for CronCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::MUTATE_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = CronMutation; - type Output = CronMutationOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let outcome = cron_mutate( - context.primitive_transaction(), - context.now_ms(), - context.issued_at_ms(), - M::CRON_TARGETS, - &input, - )?; - Ok(match outcome { - CronMutationOutcome::NotFound => CommandResult::Rejected(outcome), - _ => CommandResult::Success(outcome), - }) - } -} - -/// Typed Cron query. -pub struct CronQueryCommand(PhantomData M>); - -impl Query for CronQueryCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = CronQuery; - type Output = CronQueryResult; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - cron_query(context.primitive_connection(), &input) - } -} - -/// Authorized Cron capability with deterministic schedule sharding. -pub struct CronNamespace { - client: CellClient, - tenant: TenantId, - application: ApplicationId, - shards: u32, - module: PhantomData M>, -} - -impl Clone for CronNamespace { - fn clone(&self) -> Self { - Self { - client: self.client.clone(), - tenant: self.tenant, - application: self.application, - shards: self.shards, - module: PhantomData, - } - } -} - -impl CronNamespace { - /// Creates a Cron capability after validating its compiled namespace role. - pub fn new( - client: CellClient, - tenant: TenantId, - application: ApplicationId, - ) -> crate::Result { - let shards = client.require_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Cron)?; - Ok(Self { - client, - tenant, - application, - shards, - module: PhantomData, - }) - } - - /// Applies one durable schedule mutation on its deterministic shard. - pub async fn mutate( - &self, - identity: crate::cell::executor::MutationIdentity, - mutation: CronMutation, - ) -> std::result::Result, InvocationError> - { - let target = self - .target(mutation_id(&mutation)) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>(&target, identity, mutation) - .await - } - - /// Reads one schedule at an optional minimum publication receipt. - pub async fn get( - &self, - schedule_id: [u8; 16], - minimum: Option, - ) -> std::result::Result, InvocationError> { - let target = self - .target(schedule_id) - .map_err(InvocationError::NotStarted)?; - self.client - .query::>(&target, minimum, CronQuery::Get { schedule_id }) - .await - } - - /// Lists one explicit Cron shard without unbounded fleet fan-out. - pub async fn list_shard( - &self, - shard: u32, - after: Option<[u8; 16]>, - limit: u32, - minimum: Option, - ) -> std::result::Result, InvocationError> { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - self.client - .query::>(&target, minimum, CronQuery::List { after, limit }) - .await - } - - fn target(&self, schedule_id: [u8; 16]) -> crate::Result { - let shard = shard_for_scope(M::NAMESPACE, &schedule_id, self.shards)?; - self.shard_target(shard) - } - - fn shard_target(&self, shard: u32) -> crate::Result { - if shard >= self.shards { - return Err(crate::Error::Identity("cron shard outside namespace")); - } - CellTarget::new( - self.tenant, - self.application, - M::NAMESPACE, - &partition_for_shard(shard), - ) - } -} - -fn mutation_id(mutation: &CronMutation) -> [u8; 16] { - match mutation { - CronMutation::Upsert { schedule_id, .. } - | CronMutation::Pause { schedule_id } - | CronMutation::Resume { schedule_id, .. } - | CronMutation::Delete { schedule_id } => *schedule_id, - } -} - -impl WireValue for CronInvocation { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if self.payload.len() > MAX_PAYLOAD_BYTES { - return Err(CodecError::Invalid("cron payload exceeds 256 KiB")); - } - encoder.write_bytes(&self.schedule_id)?; - encoder.write_u64(self.generation)?; - encoder.write_u64(self.occurrence)?; - encoder.write_i64(self.scheduled_at_ms)?; - encoder.write_bytes(&self.payload) - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let value = Self { - schedule_id: read_fixed(decoder, "cron schedule ID length")?, - generation: decoder.read_u64()?, - occurrence: decoder.read_u64()?, - scheduled_at_ms: decoder.read_i64()?, - payload: decoder.read_bytes()?.to_vec(), - }; - if value.payload.len() > MAX_PAYLOAD_BYTES { - return Err(CodecError::Invalid("cron payload exceeds 256 KiB")); - } - Ok(value) - } -} - -impl WireValue for CronMutation { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Upsert { - schedule_id, - target_index, - target_partition, - payload, - interval_ms, - next_due_ms, - } => { - if payload.len() > MAX_PAYLOAD_BYTES { - return Err(CodecError::Invalid("cron payload exceeds 256 KiB")); - } - encoder.write_u8(0)?; - encoder.write_bytes(schedule_id)?; - encoder.write_u32(*target_index)?; - encoder.write_bytes(target_partition)?; - encoder.write_bytes(payload)?; - encoder.write_u64(*interval_ms)?; - encoder.write_i64(*next_due_ms) - } - Self::Pause { schedule_id } => { - encoder.write_u8(1)?; - encoder.write_bytes(schedule_id) - } - Self::Resume { - schedule_id, - next_due_ms, - } => { - encoder.write_u8(2)?; - encoder.write_bytes(schedule_id)?; - encoder.write_i64(*next_due_ms) - } - Self::Delete { schedule_id } => { - encoder.write_u8(3)?; - encoder.write_bytes(schedule_id) - } - } - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => { - let schedule_id = read_fixed(decoder, "cron schedule ID length")?; - let target_index = decoder.read_u32()?; - let target_partition = decoder.read_bytes()?.to_vec(); - let payload = decoder.read_bytes()?.to_vec(); - if payload.len() > MAX_PAYLOAD_BYTES { - return Err(CodecError::Invalid("cron payload exceeds 256 KiB")); - } - Ok(Self::Upsert { - schedule_id, - target_index, - target_partition, - payload, - interval_ms: decoder.read_u64()?, - next_due_ms: decoder.read_i64()?, - }) - } - 1 => Ok(Self::Pause { - schedule_id: read_fixed(decoder, "cron schedule ID length")?, - }), - 2 => Ok(Self::Resume { - schedule_id: read_fixed(decoder, "cron schedule ID length")?, - next_due_ms: decoder.read_i64()?, - }), - 3 => Ok(Self::Delete { - schedule_id: read_fixed(decoder, "cron schedule ID length")?, - }), - _ => Err(CodecError::Invalid("invalid cron mutation tag")), - } - } -} - -impl WireValue for CronMutationOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Applied { generation } => { - encoder.write_u8(0)?; - encoder.write_u64(*generation) - } - Self::Deleted => encoder.write_u8(1), - Self::NotFound => encoder.write_u8(2), - } - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(Self::Applied { - generation: decoder.read_u64()?, - }), - 1 => Ok(Self::Deleted), - 2 => Ok(Self::NotFound), - _ => Err(CodecError::Invalid("invalid cron outcome tag")), - } - } -} - -impl WireValue for CronQuery { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Get { schedule_id } => { - encoder.write_u8(0)?; - encoder.write_bytes(schedule_id) - } - Self::List { after, limit } => { - encoder.write_u8(1)?; - encode_optional_id(*after, encoder)?; - encoder.write_u32(*limit) - } - } - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(Self::Get { - schedule_id: read_fixed(decoder, "cron schedule ID length")?, - }), - 1 => Ok(Self::List { - after: decode_optional_id(decoder)?, - limit: decoder.read_u32()?, - }), - _ => Err(CodecError::Invalid("invalid cron query tag")), - } - } -} - -impl WireValue for CronSchedule { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.schedule_id)?; - encoder.write_u32(self.target_index)?; - encoder.write_bytes(&self.target_partition)?; - encoder.write_bytes(&self.payload)?; - encoder.write_u64(self.interval_ms)?; - encoder.write_i64(self.next_due_ms)?; - encoder.write_u64(self.occurrence)?; - self.enabled.encode(encoder)?; - encoder.write_u64(self.generation) - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - schedule_id: read_fixed(decoder, "cron schedule ID length")?, - target_index: decoder.read_u32()?, - target_partition: decoder.read_bytes()?.to_vec(), - payload: decoder.read_bytes()?.to_vec(), - interval_ms: decoder.read_u64()?, - next_due_ms: decoder.read_i64()?, - occurrence: decoder.read_u64()?, - enabled: bool::decode(decoder)?, - generation: decoder.read_u64()?, - }) - } -} - -impl WireValue for CronQueryResult { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Get(value) => { - encoder.write_u8(0)?; - value.encode(encoder) - } - Self::List { schedules, next } => { - if schedules.len() > 128 { - return Err(CodecError::Invalid("cron page exceeds 128 schedules")); - } - encoder.write_u8(1)?; - encoder.write_count(schedules.len())?; - for schedule in schedules { - schedule.encode(encoder)?; - } - encode_optional_id(*next, encoder) - } - } - } - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(Self::Get(Option::::decode(decoder)?)), - 1 => { - let count = decoder.read_count()?; - if count > 128 { - return Err(CodecError::Invalid("cron page exceeds 128 schedules")); - } - let mut schedules = Vec::with_capacity(count); - for _ in 0..count { - schedules.push(CronSchedule::decode(decoder)?); - } - Ok(Self::List { - schedules, - next: decode_optional_id(decoder)?, - }) - } - _ => Err(CodecError::Invalid("invalid cron query result tag")), - } - } -} - -fn encode_optional_id( - value: Option<[u8; 16]>, - encoder: &mut BoundedEncoder, -) -> Result<(), CodecError> { - match value { - None => encoder.write_u8(0), - Some(value) => { - encoder.write_u8(1)?; - encoder.write_bytes(&value) - } - } -} - -fn decode_optional_id(decoder: &mut BoundedDecoder<'_>) -> Result, CodecError> { - match decoder.read_u8()? { - 0 => Ok(None), - 1 => Ok(Some(read_fixed(decoder, "cron schedule ID length")?)), - _ => Err(CodecError::Invalid("invalid optional cron schedule ID")), - } -} diff --git a/crates/crab-cell-runtime/src/primitives/effects.rs b/crates/crab-cell-runtime/src/primitives/effects.rs deleted file mode 100644 index b7d4ecd4c..000000000 --- a/crates/crab-cell-runtime/src/primitives/effects.rs +++ /dev/null @@ -1,938 +0,0 @@ -//! Effect primitive: delivery intents, leases, and acknowledgements. -use prost::Message; -use rand::RngCore; -use rusqlite::{Connection, OptionalExtension, Transaction}; - -use crate::cell::executor::Resolution; -use crate::cell::executor::{HandlerOutcome, StoredOutcome}; -use crate::identity::IncarnationId; -use crate::identity::{CellId, CellTarget, Digest}; -use crate::peer::wire; -use crate::{Error, Result}; - -mod api; -mod lease; -mod supervisor; - -pub use lease::*; - -#[cfg(test)] -mod tests; - -pub use api::{ - EffectAckRequest, EffectClaimCommand, EffectClaimRequest, EffectLeaseCommand, - EffectLeaseRequest, EffectModule, EffectSource, EffectStatusQuery, EffectStatusRequest, - EffectValidateClaimQuery, EffectValidateRequest, register_effect_delivery, -}; -pub use supervisor::{EffectRunOutcome, EffectSupervisor, EffectSupervisorError}; - -const MAX_EFFECTS_PER_COMMAND: usize = 128; -const MAX_EFFECT_BYTES: usize = 1 << 20; -const EFFECT_CLAIM_FIXED_BYTES: usize = 160; -const EFFECT_CLAIM_LIST_BYTES: usize = 4; -const EFFECT_ACK_FIXED_BYTES: usize = 73; -// Full typed claim outputs and acknowledgement inputs share the registry's 1 MiB ceiling. -pub(crate) const MAX_EFFECT_OPERATION_BYTES: usize = - MAX_EFFECT_BYTES - EFFECT_CLAIM_LIST_BYTES - EFFECT_CLAIM_FIXED_BYTES; -pub(crate) const MAX_EFFECT_RESULT_BYTES: usize = MAX_EFFECT_BYTES - EFFECT_ACK_FIXED_BYTES; -const MAX_CLAIM_ITEMS: usize = 32; -const MAX_RECLAIM_ITEMS: usize = 128; -const MAX_ATTEMPTS: u32 = 20; -const MIN_LEASE_MS: u32 = 5_000; -const MAX_LEASE_MS: u32 = 300_000; -const DELIVERY_MARGIN_MS: i64 = 1_000; -const EFFECT_LIFETIME_MS: i64 = 7 * 24 * 60 * 60 * 1_000; -const INBOX_RETENTION_MS: i64 = 7 * 24 * 60 * 60 * 1_000; - -/// Durable source-side effect state. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum EffectState { - /// The effect awaits its next delivery attempt. - Ready, - /// Delivery is leased to one destination. - Leased, - /// The destination applied the effect. - Delivered, - /// The destination rejected the effect terminally. - Failed, -} - -/// Durable source-side effect outcome and lease state. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct EffectStatus { - /// Current durable state. - pub state: EffectState, - /// Delivery attempts made so far. - pub attempt: u32, - /// Whether a live lease token exists; the token itself is never revealed. - pub token_present: bool, - /// Lease expiry, while the effect is leased. - pub lease_until_ms: Option, - /// Logical time the effect stops being deliverable. - pub expires_at_ms: i64, - /// Result recorded once the destination answered. - pub result: Option>, -} - -/// Business outcome of a source-side effect lease mutation. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum EffectLeaseOutcome { - /// The destination applied the effect. - Delivered, - /// The effect returns for another attempt. - Retrying { - /// Logical time the next attempt becomes claimable. - due_at_ms: i64, - }, - /// The destination rejected the effect terminally. - Failed, - /// The lease was extended. - Extended { - /// New lease expiry. - lease_until_ms: i64, - }, - /// The token no longer matches the source lease. - LeaseLost, -} - -impl EffectState { - fn encode(self) -> i64 { - match self { - Self::Ready => 0, - Self::Leased => 1, - Self::Delivered => 2, - Self::Failed => 3, - } - } - - fn decode(value: i64) -> Result { - match value { - 0 => Ok(Self::Ready), - 1 => Ok(Self::Leased), - 2 => Ok(Self::Delivered), - 3 => Ok(Self::Failed), - _ => Err(Error::Command("invalid stored effect state")), - } - } -} - -/// Reads one exact durable source effect without exposing its lease token. -pub fn effect_status(connection: &Connection, effect_id: [u8; 32]) -> Result> { - let row = connection - .query_row( - "SELECT state, attempt, token IS NOT NULL, lease_until_ms, expires_at_ms, result FROM sys_effects WHERE effect_id = ?1", - [effect_id.as_slice()], - |row| { - Ok(( - row.get::<_, i64>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, bool>(2)?, - row.get::<_, Option>(3)?, - row.get::<_, i64>(4)?, - row.get::<_, Option>>(5)?, - )) - }, - ) - .optional()?; - let Some((state, attempt, token_present, lease_until_ms, expires_at_ms, result)) = row else { - return Ok(None); - }; - if result - .as_ref() - .is_some_and(|result| result.len() > MAX_EFFECT_RESULT_BYTES) - { - return Err(Error::Command("stored effect result exceeds wire limit")); - } - Ok(Some(EffectStatus { - state: EffectState::decode(state)?, - attempt: u32::try_from(attempt) - .map_err(|_| Error::Command("invalid stored effect attempt"))?, - token_present, - lease_until_ms, - expires_at_ms, - result, - })) -} - -#[derive(Clone, Debug, PartialEq, Eq)] -struct EffectIntent { - pub destination: CellId, - pub operation: Vec, - pub expires_at_ms: i64, -} - -/// One stable typed command whose destination incarnation is resolved at delivery time. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct EffectCommandIntent { - /// Destination Cell, resolved against its current incarnation at delivery. - pub target: CellTarget, - /// Registered command the destination executes. - pub command_id: u32, - /// Input codec version the destination must support. - pub codec_version: u32, - /// Command input bytes. - pub input: Vec, - /// Logical time the effect stops being deliverable. - pub expires_at_ms: i64, -} - -/// Command-scoped allocator for durable effect identities. -/// -/// One batch must be shared by every primitive transition performed by the -/// same Cell command so each emitted effect receives a unique ordinal. -pub(crate) struct EffectBatch { - source: CellTarget, - incarnation: IncarnationId, - sequence: u64, - now_ms: i64, - next_ordinal: u32, - operation_bytes: usize, -} - -impl EffectBatch { - /// Verifies the supplied source against the authoritative command transaction. - pub(crate) fn new( - transaction: &Transaction<'_>, - source: &CellTarget, - sequence: u64, - now_ms: i64, - ) -> Result { - if sequence == 0 { - return Err(Error::Command("effect batch sequence is zero")); - } - validate_now(now_ms)?; - let (cell, incarnation, commit_sequence) = transaction.query_row( - "SELECT cell_id, incarnation, commit_sequence FROM sys_meta WHERE singleton = 1", - [], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, u64>(2)?, - )) - }, - )?; - if commit_sequence.checked_add(1) != Some(sequence) { - return Err(Error::Command( - "effect batch sequence is not the next Cell commit", - )); - } - let cell = CellId::from_bytes( - cell.try_into() - .map_err(|_| Error::Command("invalid stored effect source Cell"))?, - ); - if source.cell_id() != cell { - return Err(Error::Command( - "effect source target does not match its Cell", - )); - } - Ok(Self { - source: source.clone(), - incarnation: IncarnationId::from_bytes( - incarnation - .try_into() - .map_err(|_| Error::Command("invalid stored effect source incarnation"))?, - ), - sequence, - now_ms, - next_ordinal: 0, - operation_bytes: 0, - }) - } - - /// Inserts one canonical Cell command without pinning a destination owner incarnation. - pub(crate) fn insert_command( - &mut self, - transaction: &Transaction<'_>, - intent: &EffectCommandIntent, - ) -> Result<[u8; 32]> { - if intent.target.tenant() != self.source.tenant() - || intent.target.application() != self.source.application() - { - return Err(Error::Identity( - "effect target is outside the source application scope", - )); - } - validate_effect_command_intent(self.now_ms, intent)?; - if self.next_ordinal as usize >= MAX_EFFECTS_PER_COMMAND { - return Err(Error::Command("command effects exceed limits")); - } - let ordinal = self.next_ordinal; - let id = effect_id( - self.source.cell_id(), - self.incarnation, - self.sequence, - ordinal, - ); - let operation = wire::EffectRequest { - target: Some(wire::Target { - tenant_id: intent.target.tenant().as_bytes().to_vec(), - application_id: intent.target.application().as_bytes().to_vec(), - namespace_id: intent.target.namespace().as_bytes().to_vec(), - partition: intent.target.partition().to_vec(), - }), - // The durable operation is owner-independent. The delivery client - // fills this field from a fresh Describe immediately before send. - destination_incarnation: Vec::new(), - identity: Some(wire::EffectIdentity { - effect_id: id.to_vec(), - source_cell: self.source.cell_id().as_bytes().to_vec(), - source_incarnation: self.incarnation.as_bytes().to_vec(), - source_sequence: self.sequence, - ordinal, - expires_at_ms: intent.expires_at_ms, - }), - operation: Some(wire::effect_request::Operation::CellCommand( - wire::CellCommand { - command_id: intent.command_id, - codec_version: intent.codec_version, - input: intent.input.clone(), - }, - )), - } - .encode_to_vec(); - let operation_len = operation.len(); - let prospective = self - .operation_bytes - .checked_add(operation_len) - .ok_or(Error::Command("effect byte count overflow"))?; - if prospective > MAX_EFFECT_BYTES { - return Err(Error::Command("command effects exceed limits")); - } - let next_ordinal = ordinal - .checked_add(1) - .ok_or(Error::Command("effect ordinal overflow"))?; - let id = effect_insert( - transaction, - self.source.cell_id(), - self.incarnation, - self.sequence, - ordinal, - self.now_ms, - &EffectIntent { - destination: intent.target.cell_id(), - operation, - expires_at_ms: intent.expires_at_ms, - }, - )?; - self.next_ordinal = next_ordinal; - self.operation_bytes = prospective; - Ok(id) - } - - pub(crate) const fn source_incarnation(&self) -> IncarnationId { - self.incarnation - } - - pub(crate) const fn source_target(&self) -> &CellTarget { - &self.source - } -} - -/// Source of unpredictable effect lease tokens. -pub trait EffectTokenSource { - /// Returns an unpredictable, non-zero lease token. - fn next_token(&mut self) -> Result<[u8; 16]>; -} - -/// Cryptographically seeded process-local effect token source. -pub struct SystemEffectTokens; - -impl EffectTokenSource for SystemEffectTokens { - fn next_token(&mut self) -> Result<[u8; 16]> { - loop { - let mut token = [0; 16]; - rand::rng().fill_bytes(&mut token); - if token != [0; 16] { - return Ok(token); - } - } - } -} - -/// One published source-side delivery lease. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct EffectClaim { - /// Durable effect identity. - pub effect_id: [u8; 32], - /// Cell the effect was allocated for. - pub destination: CellId, - /// Opaque operation bytes carried with the effect. - pub operation: Vec, - /// Digest of the operation bytes. - pub operation_digest: Digest, - /// Delivery attempt this claim represents. - pub attempt: u32, - /// Lease token required to settle the claim. - pub token: [u8; 16], - /// Logical time the claim expires. - pub lease_until_ms: i64, - /// Logical time the effect stops being deliverable. - pub expires_at_ms: i64, - /// Source commit sequence that created the effect. - pub created_sequence: u64, -} - -/// Minimal identity needed to mutate one exact source lease. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct EffectLease { - /// Effect being settled. - pub effect_id: [u8; 32], - /// Attempt the token belongs to. - pub attempt: u32, - /// Lease token that must match the current claim. - pub token: [u8; 16], - /// Expiry recorded with the lease. - pub expires_at_ms: i64, -} - -impl From<&EffectClaim> for EffectLease { - fn from(claim: &EffectClaim) -> Self { - Self { - effect_id: claim.effect_id, - attempt: claim.attempt, - token: claim.token, - expires_at_ms: claim.expires_at_ms, - } - } -} - -/// Exact target-side identity and expiry carried by a private delivery. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct InboxDelivery { - /// Effect identity the destination records. - pub effect_id: [u8; 32], - /// Digest of the operation bytes. - pub operation_digest: Digest, - /// Logical time after which the delivery must be refused. - pub expires_at_ms: i64, -} - -/// Durable target-side result of an effect delivery. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum InboxApplyOutcome { - /// The destination applied the effect. - Success { - /// Handler result bytes returned to the source. - result: Vec, - /// Destination commit sequence the result is bound to. - commit_sequence: u64, - /// Whether this delivery repeated an already-applied effect. - duplicate: bool, - }, - /// The destination refused the effect terminally. - Rejected { - /// Handler result bytes returned to the source. - result: Vec, - /// Destination commit sequence the result is bound to. - commit_sequence: u64, - /// Whether this delivery repeated an already-applied effect. - duplicate: bool, - }, - /// The same effect ID arrived with different operation bytes. - Conflict, - /// The delivery arrived after its expiry. - Expired, -} - -/// Inserts one immutable effect intention inside its source command. -fn effect_insert( - transaction: &Transaction<'_>, - cell: CellId, - incarnation: IncarnationId, - sequence: u64, - ordinal: u32, - now_ms: i64, - intent: &EffectIntent, -) -> Result<[u8; 32]> { - validate_now(now_ms)?; - validate_effect_intent(now_ms, intent)?; - if sequence == 0 || sequence > i64::MAX as u64 || ordinal as usize >= MAX_EFFECTS_PER_COMMAND { - return Err(Error::Command("invalid effect intention")); - } - let id = effect_id(cell, incarnation, sequence, ordinal); - let existing = transaction - .query_row( - "SELECT destination, operation, expires_at_ms, created_sequence FROM sys_effects WHERE effect_id = ?1", - [id.as_slice()], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, i64>(3)?, - )) - }, - ) - .optional()?; - if let Some((destination, operation, expires_at_ms, created_sequence)) = existing { - if destination.as_slice() == intent.destination.as_bytes() - && operation == intent.operation - && expires_at_ms == intent.expires_at_ms - && created_sequence == sequence as i64 - { - return Ok(id); - } - return Err(Error::Command( - "effect ordinal was reused for different bytes", - )); - } - transaction.execute( - "INSERT INTO sys_effects(effect_id, destination, operation, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, created_sequence, result) VALUES (?1, ?2, ?3, 0, 0, ?4, ?5, NULL, NULL, ?6, NULL)", - ( - id.as_slice(), - intent.destination.as_bytes().as_slice(), - intent.operation.as_slice(), - now_ms, - intent.expires_at_ms, - sequence as i64, - ), - )?; - Ok(id) -} - -fn validate_effect_intent(now_ms: i64, intent: &EffectIntent) -> Result<()> { - validate_now(now_ms)?; - if intent.operation.is_empty() - || intent.operation.len() > MAX_EFFECT_OPERATION_BYTES - || intent.expires_at_ms <= now_ms - || intent.expires_at_ms > now_ms.saturating_add(EFFECT_LIFETIME_MS) - { - return Err(Error::Command("invalid effect intention")); - } - Ok(()) -} - -pub(crate) fn validate_effect_command_intent( - now_ms: i64, - intent: &EffectCommandIntent, -) -> Result<()> { - if intent.command_id == 0 || intent.codec_version == 0 { - return Err(Error::Command("invalid effect Cell command identifier")); - } - let operation = wire::EffectRequest { - target: Some(wire::Target { - tenant_id: intent.target.tenant().as_bytes().to_vec(), - application_id: intent.target.application().as_bytes().to_vec(), - namespace_id: intent.target.namespace().as_bytes().to_vec(), - partition: intent.target.partition().to_vec(), - }), - destination_incarnation: Vec::new(), - identity: Some(wire::EffectIdentity { - effect_id: vec![u8::MAX; 32], - source_cell: vec![u8::MAX; 32], - source_incarnation: vec![u8::MAX; 16], - source_sequence: u64::MAX, - ordinal: u32::MAX, - expires_at_ms: intent.expires_at_ms, - }), - operation: Some(wire::effect_request::Operation::CellCommand( - wire::CellCommand { - command_id: intent.command_id, - codec_version: intent.codec_version, - input: intent.input.clone(), - }, - )), - } - .encode_to_vec(); - validate_effect_intent( - now_ms, - &EffectIntent { - destination: intent.target.cell_id(), - operation, - expires_at_ms: intent.expires_at_ms, - }, - ) -} - -/// Applies or replays one private effect in the destination inbox. -pub fn inbox_apply( - transaction: &Transaction<'_>, - now_ms: i64, - delivery: InboxDelivery, - max_result_bytes: usize, - handler: impl FnOnce(&Transaction<'_>) -> Result, -) -> Result { - validate_now(now_ms)?; - if delivery.expires_at_ms <= now_ms { - return Ok(InboxApplyOutcome::Expired); - } - if max_result_bytes > MAX_EFFECT_BYTES { - return Err(Error::Command("effect result limit exceeds 1 MiB")); - } - let existing = transaction - .query_row( - "SELECT operation_digest, outcome, result, commit_sequence, expires_at_ms FROM sys_inbox WHERE effect_id = ?1", - [delivery.effect_id.as_slice()], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, Vec>(2)?, - row.get::<_, i64>(3)?, - row.get::<_, i64>(4)?, - )) - }, - ) - .optional()?; - if let Some((digest, outcome, result, sequence, expires_at_ms)) = existing { - if digest.as_slice() != delivery.operation_digest.as_bytes() - || expires_at_ms != delivery.expires_at_ms - { - return Ok(InboxApplyOutcome::Conflict); - } - return inbox_outcome(outcome, result, sequence, max_result_bytes, true); - } - - let sequence: i64 = transaction.query_row( - "SELECT commit_sequence + 1 FROM sys_meta WHERE singleton = 1", - [], - |row| row.get(0), - )?; - if sequence <= 0 { - return Err(Error::Command("effect destination sequence overflow")); - } - transaction.execute_batch("SAVEPOINT effect_application")?; - let decision = handler(transaction)?; - let (outcome, result) = match decision { - HandlerOutcome::Success(result) => { - transaction.execute_batch("RELEASE effect_application")?; - (1, result) - } - HandlerOutcome::Rejected(result) => { - transaction - .execute_batch("ROLLBACK TO effect_application; RELEASE effect_application")?; - (2, result) - } - }; - if result.len() > max_result_bytes { - return Err(Error::Command("effect handler result exceeds limit")); - } - let retain_until_ms = delivery - .expires_at_ms - .checked_add(INBOX_RETENTION_MS) - .ok_or(Error::Command("effect inbox retention overflow"))?; - transaction.execute( - "INSERT INTO sys_inbox(effect_id, operation_digest, outcome, result, commit_sequence, expires_at_ms, retain_until_ms) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7)", - ( - delivery.effect_id.as_slice(), - delivery.operation_digest.as_bytes().as_slice(), - outcome, - result.as_slice(), - sequence, - delivery.expires_at_ms, - retain_until_ms, - ), - )?; - inbox_outcome(outcome, result, sequence, max_result_bytes, false) -} - -/// Resolves one destination inbox identity without executing its handler. -pub fn inbox_resolve( - connection: &Connection, - now_ms: i64, - delivery: InboxDelivery, - max_result_bytes: usize, -) -> Result { - validate_now(now_ms)?; - if delivery.expires_at_ms <= now_ms { - return Ok(Resolution::Expired); - } - if max_result_bytes > MAX_EFFECT_BYTES { - return Err(Error::Command("effect result limit exceeds 1 MiB")); - } - let existing = connection - .query_row( - "SELECT operation_digest, outcome, result, commit_sequence, expires_at_ms FROM sys_inbox WHERE effect_id = ?1", - [delivery.effect_id.as_slice()], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, Vec>(2)?, - row.get::<_, i64>(3)?, - row.get::<_, i64>(4)?, - )) - }, - ) - .optional()?; - let Some((digest, outcome, result, sequence, expires_at_ms)) = existing else { - return Ok(Resolution::Absent); - }; - if digest.as_slice() != delivery.operation_digest.as_bytes() - || expires_at_ms != delivery.expires_at_ms - { - return Err(Error::RequestConflict); - } - Ok(Resolution::Committed(stored_inbox_outcome( - outcome, - result, - sequence, - max_result_bytes, - )?)) -} - -/// Removes at most 128 terminal source effects after their delivery horizon. -pub fn effect_cleanup_terminal(transaction: &Transaction<'_>, now_ms: i64) -> Result { - effect_cleanup_terminal_bounded(transaction, now_ms, MAX_RECLAIM_ITEMS) -} - -pub(crate) fn effect_cleanup_terminal_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - validate_now(now_ms)?; - validate_maintenance_limit(limit)?; - if limit == 0 { - return Ok(0); - } - let ids = select_ids( - transaction, - "SELECT effect_id FROM sys_effects WHERE state IN (2, 3) AND expires_at_ms <= ?1 ORDER BY expires_at_ms, effect_id LIMIT ?2", - now_ms, - limit, - )?; - for id in &ids { - transaction.execute( - "DELETE FROM sys_effects WHERE effect_id = ?1 AND state IN (2, 3) AND expires_at_ms <= ?2", - (id.as_slice(), now_ms), - )?; - } - Ok(ids.len()) -} - -/// Removes at most 128 inbox receipts after sender expiry plus seven days. -pub fn inbox_cleanup_expired(transaction: &Transaction<'_>, now_ms: i64) -> Result { - inbox_cleanup_expired_bounded(transaction, now_ms, MAX_RECLAIM_ITEMS) -} - -pub(crate) fn inbox_cleanup_expired_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - validate_now(now_ms)?; - validate_maintenance_limit(limit)?; - if limit == 0 { - return Ok(0); - } - let ids = select_ids( - transaction, - "SELECT effect_id FROM sys_inbox WHERE retain_until_ms <= ?1 ORDER BY retain_until_ms, effect_id LIMIT ?2", - now_ms, - limit, - )?; - for id in &ids { - transaction.execute( - "DELETE FROM sys_inbox WHERE effect_id = ?1 AND retain_until_ms <= ?2", - (id.as_slice(), now_ms), - )?; - } - Ok(ids.len()) -} - -/// Computes the destination operation digest carried unchanged across retries. -#[must_use] -pub fn effect_operation_digest( - destination: CellId, - effect_id: [u8; 32], - operation: &[u8], -) -> Digest { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.effect-op.v1\0"); - hasher.update(destination.as_bytes()); - hasher.update(&effect_id); - hasher.update(&(operation.len() as u32).to_be_bytes()); - hasher.update(operation); - Digest::from_bytes(*hasher.finalize().as_bytes()) -} - -/// Derives the immutable identity for one source transaction effect ordinal. -#[must_use] -pub fn effect_id( - cell: CellId, - incarnation: IncarnationId, - sequence: u64, - ordinal: u32, -) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.effect.v1\0"); - hasher.update(cell.as_bytes()); - hasher.update(incarnation.as_bytes()); - hasher.update(&sequence.to_be_bytes()); - hasher.update(&ordinal.to_be_bytes()); - *hasher.finalize().as_bytes() -} - -fn apply_lease( - transaction: &Transaction<'_>, - now_ms: i64, - lease: &EffectLease, - state: EffectState, - due_at_ms: Option, - result: Option<&[u8]>, -) -> Result { - let due_at_ms = due_at_ms.unwrap_or(0); - let changed = transaction.execute( - "UPDATE sys_effects SET state = ?1, due_at_ms = CASE WHEN ?1 = 0 THEN ?2 ELSE due_at_ms END, token = NULL, lease_until_ms = NULL, result = ?3 WHERE effect_id = ?4 AND state = 1 AND attempt = ?5 AND token = ?6 AND lease_until_ms > ?7 AND expires_at_ms = ?8", - ( - state.encode(), - due_at_ms, - result, - lease.effect_id.as_slice(), - i64::from(lease.attempt), - lease.token.as_slice(), - now_ms, - lease.expires_at_ms, - ), - )?; - Ok(changed == 1) -} - -pub(crate) fn effect_reclaim_expired_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - validate_now(now_ms)?; - validate_maintenance_limit(limit)?; - if limit == 0 { - return Ok(0); - } - let expired = { - let mut statement = transaction.prepare( - "SELECT effect_id, attempt, expires_at_ms FROM sys_effects INDEXED BY sys_effects_leases WHERE state = 1 AND lease_until_ms <= ?1 ORDER BY lease_until_ms, effect_id LIMIT ?2", - )?; - statement - .query_map((now_ms, limit as i64), |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, i64>(2)?, - )) - })? - .collect::, _>>()? - }; - let count = expired.len(); - for (id, attempt, expires_at_ms) in expired { - if attempt < 0 { - return Err(Error::Command("invalid stored effect attempt")); - } - let attempt = - u32::try_from(attempt).map_err(|_| Error::Command("invalid stored effect attempt"))?; - let due_at_ms = now_ms.saturating_add(retry_delay_ms(attempt)); - let state = if attempt >= MAX_ATTEMPTS || due_at_ms >= expires_at_ms { - EffectState::Failed - } else { - EffectState::Ready - }; - transaction.execute( - "UPDATE sys_effects SET state = ?1, due_at_ms = CASE WHEN ?1 = 0 THEN ?2 ELSE due_at_ms END, token = NULL, lease_until_ms = NULL WHERE effect_id = ?3 AND state = 1 AND lease_until_ms <= ?4", - (state.encode(), due_at_ms, id, now_ms), - )?; - } - Ok(count) -} - -pub(crate) fn effect_expire_ready_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - validate_now(now_ms)?; - validate_maintenance_limit(limit)?; - if limit == 0 { - return Ok(0); - } - Ok(transaction.execute( - "UPDATE sys_effects SET state = 3 WHERE effect_id IN (SELECT effect_id FROM sys_effects INDEXED BY sys_effects_due WHERE state = 0 AND (expires_at_ms <= ?1 OR attempt >= ?2) ORDER BY due_at_ms, effect_id LIMIT ?3)", - (now_ms, i64::from(MAX_ATTEMPTS), limit as i64), - )?) -} - -fn inbox_outcome( - outcome: i64, - result: Vec, - sequence: i64, - max_result_bytes: usize, - duplicate: bool, -) -> Result { - match stored_inbox_outcome(outcome, result, sequence, max_result_bytes)? { - StoredOutcome::Success { - result, - commit_sequence, - } => Ok(InboxApplyOutcome::Success { - result, - commit_sequence, - duplicate, - }), - StoredOutcome::Rejected { - result, - commit_sequence, - } => Ok(InboxApplyOutcome::Rejected { - result, - commit_sequence, - duplicate, - }), - } -} - -fn stored_inbox_outcome( - outcome: i64, - result: Vec, - sequence: i64, - max_result_bytes: usize, -) -> Result { - if result.len() > max_result_bytes || sequence <= 0 { - return Err(Error::Command("invalid stored inbox outcome")); - } - let commit_sequence = - u64::try_from(sequence).map_err(|_| Error::Command("invalid stored inbox sequence"))?; - match outcome { - 1 => Ok(StoredOutcome::Success { - result, - commit_sequence, - }), - 2 => Ok(StoredOutcome::Rejected { - result, - commit_sequence, - }), - _ => Err(Error::Command("invalid stored inbox outcome")), - } -} - -fn select_ids( - transaction: &Transaction<'_>, - sql: &str, - now_ms: i64, - limit: usize, -) -> Result> { - let mut statement = transaction.prepare(sql)?; - statement - .query_map((now_ms, limit as i64), |row| row.get::<_, Vec>(0))? - .map(|row| exact::<32>(row.map_err(Error::from)?, "invalid stored effect ID")) - .collect() -} - -fn validate_maintenance_limit(limit: usize) -> Result<()> { - if limit > MAX_RECLAIM_ITEMS { - return Err(Error::Command("effect maintenance limit exceeds 128")); - } - Ok(()) -} - -fn retry_delay_ms(attempt: u32) -> i64 { - (100_i64 << attempt.min(10)).min(60_000) -} - -fn exact(value: Vec, message: &'static str) -> Result<[u8; N]> { - value.try_into().map_err(|_| Error::Command(message)) -} - -fn validate_now(now_ms: i64) -> Result<()> { - if now_ms < 0 { - return Err(Error::Command("effect time must be non-negative")); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/primitives/effects/api.rs b/crates/crab-cell-runtime/src/primitives/effects/api.rs deleted file mode 100644 index 15727b773..000000000 --- a/crates/crab-cell-runtime/src/primitives/effects/api.rs +++ /dev/null @@ -1,611 +0,0 @@ -use std::marker::PhantomData; - -use crate::client::{CellClient, Committed, InvocationError, Observed, Receipt}; -use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue, read_fixed}; -use crate::identity::CellTarget; -use crate::registry::{Command, Query, RegistryBuilder}; -use crate::registry::{CommandContext, CommandResult, QueryContext}; - -use super::{ - EffectClaim, EffectLease, EffectLeaseOutcome, EffectState, EffectStatus, - MAX_EFFECT_OPERATION_BYTES, MAX_EFFECT_RESULT_BYTES, SystemEffectTokens, effect_ack_lease, - effect_claim, effect_retry_lease, effect_status, effect_validate_claim, -}; - -const ACK_TAG: u8 = 0; -const RETRY_TAG: u8 = 1; -const DELIVERED_TAG: u8 = 0; -const RETRYING_TAG: u8 = 1; -const FAILED_TAG: u8 = 2; -const EXTENDED_TAG: u8 = 3; -const LEASE_LOST_TAG: u8 = 4; - -/// Compile-time operation identifiers for source effect supervision. -pub trait EffectModule: Send + Sync + 'static { - /// Module the effect surfaces register under. - const MODULE: &'static str; - /// Codec version of the claim and lease commands. - const CODEC_VERSION: u32 = 1; - /// Command id that claims source effects. - const CLAIM_COMMAND_ID: u32; - /// Command id that applies lease transitions. - const LEASE_COMMAND_ID: u32; - /// Query id that revalidates published claims. - const VALIDATE_QUERY_ID: u32; - /// Query id that reads one effect's status. - const STATUS_QUERY_ID: u32; -} - -/// Registers source effect claim, lease transition and validation bindings. -pub fn register_effect_delivery( - registry: &mut RegistryBuilder, -) -> crate::Result<()> { - registry.bind_effect_runner::()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_query::>()?; - registry.bind_query::>() -} - -/// Bounded claim parameters for one source Cell. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct EffectClaimRequest { - /// Maximum effects to claim. - pub limit: u32, - /// Lease duration granted to each claim. - pub lease_ms: u32, -} - -/// Claims a bounded source effect batch through ordinary publication. -pub struct EffectClaimCommand(PhantomData M>); - -impl Command for EffectClaimCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::CLAIM_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = EffectClaimRequest; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let limit = usize::try_from(input.limit) - .map_err(|_| crate::Error::Command("effect claim limit overflow"))?; - let mut tokens = SystemEffectTokens; - Ok(CommandResult::Success(effect_claim( - context.primitive_transaction(), - context.now_ms(), - limit, - input.lease_ms, - &mut tokens, - )?)) - } -} - -/// Published claims to revalidate before network emission. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct EffectValidateRequest { - /// Claims to revalidate against the source ledger. - pub claimed: Vec, -} - -/// Revalidates exact source leases at or after the claim receipt. -pub struct EffectValidateClaimQuery(PhantomData M>); - -impl Query for EffectValidateClaimQuery { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::VALIDATE_QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = EffectValidateRequest; - type Output = bool; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - effect_validate_claim( - context.primitive_connection(), - context.now_ms(), - &input.claimed, - ) - } -} - -/// Selects one exact source effect by its stable ID. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct EffectStatusRequest { - /// Effect to read. - pub effect_id: [u8; 32], -} - -/// Reads the durable source outcome without exposing the lease token. -pub struct EffectStatusQuery(PhantomData M>); - -impl Query for EffectStatusQuery { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::STATUS_QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = EffectStatusRequest; - type Output = Option; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - effect_status(context.primitive_connection(), input.effect_id) - } -} - -/// Acknowledges one target result for an exact source lease. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct EffectAckRequest { - /// Exact source lease to acknowledge. - pub lease: EffectLease, - /// Target result bytes to record. - pub result: Vec, -} - -/// Selects one idempotent source lease transition. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum EffectLeaseRequest { - /// Records a target result for the lease. - Ack(EffectAckRequest), - /// Returns the effect for another delivery attempt. - Retry(EffectLease), -} - -/// Applies target acknowledgement or retry to an exact source lease. -pub struct EffectLeaseCommand(PhantomData M>); - -impl Command for EffectLeaseCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::LEASE_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = EffectLeaseRequest; - type Output = EffectLeaseOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let outcome = match input { - EffectLeaseRequest::Ack(request) => effect_ack_lease( - context.primitive_transaction(), - context.now_ms(), - request.lease, - &request.result, - )?, - EffectLeaseRequest::Retry(lease) => { - effect_retry_lease(context.primitive_transaction(), context.now_ms(), lease)? - } - }; - Ok(match outcome { - EffectLeaseOutcome::LeaseLost => CommandResult::Rejected(outcome), - _ => CommandResult::Success(outcome), - }) - } -} - -/// Capability for one explicit source Cell's durable effect ledger. -#[derive(Clone)] -pub struct EffectSource { - client: CellClient, - target: CellTarget, - module: PhantomData M>, -} - -impl EffectSource { - /// Binds one source Cell client to this module's effect surfaces. - #[must_use] - pub fn new(client: CellClient, target: CellTarget) -> Self { - Self { - client, - target, - module: PhantomData, - } - } - - /// Returns the source Cell this handle targets. - #[must_use] - pub const fn target(&self) -> &CellTarget { - &self.target - } - - /// Claims a bounded batch of source effects through ordinary publication. - pub async fn claim( - &self, - identity: crate::cell::executor::MutationIdentity, - request: EffectClaimRequest, - ) -> std::result::Result>, InvocationError>> { - self.client - .command::>(&self.target, identity, request) - .await - } - - /// Revalidates exact leases at or after the given receipt. - pub async fn validate( - &self, - claimed: Vec, - minimum: Receipt, - ) -> std::result::Result, InvocationError> { - // Lease checks gate external work; a stale snapshot cannot prove that - // the owner has not revoked or replaced the claim. - self.client - .with_read_policy(crate::client::ReadPolicy::CurrentOwner) - .query::>( - &self.target, - Some(minimum), - EffectValidateRequest { claimed }, - ) - .await - } - - /// Reads one source effect at or after the requested receipt. - pub async fn status( - &self, - effect_id: [u8; 32], - minimum: Option, - ) -> std::result::Result>, InvocationError>> - { - self.client - .query::>(&self.target, minimum, EffectStatusRequest { effect_id }) - .await - } - - /// Acknowledges one target result for a published claim. - pub async fn ack( - &self, - identity: crate::cell::executor::MutationIdentity, - claim: EffectClaim, - result: Vec, - ) -> std::result::Result, InvocationError> - { - self.client - .command::>( - &self.target, - identity, - EffectLeaseRequest::Ack(EffectAckRequest { - lease: EffectLease::from(&claim), - result, - }), - ) - .await - } - - /// Returns a published claim for another delivery attempt. - pub async fn retry( - &self, - identity: crate::cell::executor::MutationIdentity, - claim: EffectClaim, - ) -> std::result::Result, InvocationError> - { - self.client - .command::>( - &self.target, - identity, - EffectLeaseRequest::Retry(EffectLease::from(&claim)), - ) - .await - } -} - -impl WireValue for EffectClaimRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_u32(self.limit)?; - encoder.write_u32(self.lease_ms) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - limit: decoder.read_u32()?, - lease_ms: decoder.read_u32()?, - }) - } -} - -impl WireValue for EffectClaim { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - validate_claim(self)?; - encoder.write_bytes(&self.effect_id)?; - encoder.write_bytes(self.destination.as_bytes())?; - encoder.write_bytes(&self.operation)?; - encoder.write_bytes(self.operation_digest.as_bytes())?; - encoder.write_u32(self.attempt)?; - encoder.write_bytes(&self.token)?; - encoder.write_i64(self.lease_until_ms)?; - encoder.write_i64(self.expires_at_ms)?; - encoder.write_u64(self.created_sequence) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let claim = Self { - effect_id: read_fixed(decoder, "effect ID length")?, - destination: crate::CellId::try_from(decoder.read_bytes()?) - .map_err(|_| CodecError::Invalid("effect destination length"))?, - operation: decoder.read_bytes()?.to_vec(), - operation_digest: crate::Digest::try_from(decoder.read_bytes()?) - .map_err(|_| CodecError::Invalid("effect digest length"))?, - attempt: decoder.read_u32()?, - token: read_fixed(decoder, "effect token length")?, - lease_until_ms: decoder.read_i64()?, - expires_at_ms: decoder.read_i64()?, - created_sequence: decoder.read_u64()?, - }; - validate_claim(&claim)?; - Ok(claim) - } -} - -impl WireValue for EffectLease { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - validate_lease(self)?; - encoder.write_bytes(&self.effect_id)?; - encoder.write_u32(self.attempt)?; - encoder.write_bytes(&self.token)?; - encoder.write_i64(self.expires_at_ms) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let lease = Self { - effect_id: read_fixed(decoder, "effect ID length")?, - attempt: decoder.read_u32()?, - token: read_fixed(decoder, "effect token length")?, - expires_at_ms: decoder.read_i64()?, - }; - validate_lease(&lease)?; - Ok(lease) - } -} - -impl WireValue for Vec { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if self.len() > 32 { - return Err(CodecError::Invalid("too many effect claims")); - } - encoder.write_count(self.len())?; - for claim in self { - claim.encode(encoder)?; - } - Ok(()) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let count = decoder.read_count()?; - if count > 32 { - return Err(CodecError::Invalid("too many effect claims")); - } - let mut claims = Vec::with_capacity(count); - for _ in 0..count { - claims.push(EffectClaim::decode(decoder)?); - } - Ok(claims) - } -} - -impl WireValue for EffectValidateRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - self.claimed.encode(encoder) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - claimed: Vec::::decode(decoder)?, - }) - } -} - -impl WireValue for EffectStatusRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.effect_id) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - effect_id: read_fixed(decoder, "effect ID length")?, - }) - } -} - -impl WireValue for EffectStatus { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if self - .result - .as_ref() - .is_some_and(|result| result.len() > MAX_EFFECT_RESULT_BYTES) - { - return Err(CodecError::Invalid( - "effect status result exceeds wire limit", - )); - } - let state = match self.state { - EffectState::Ready => 0, - EffectState::Leased => 1, - EffectState::Delivered => 2, - EffectState::Failed => 3, - }; - encoder.write_u8(state)?; - encoder.write_u32(self.attempt)?; - encoder.write_bool(self.token_present)?; - self.lease_until_ms.encode(encoder)?; - encoder.write_i64(self.expires_at_ms)?; - self.result.encode(encoder) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let state = match decoder.read_u8()? { - 0 => EffectState::Ready, - 1 => EffectState::Leased, - 2 => EffectState::Delivered, - 3 => EffectState::Failed, - _ => return Err(CodecError::Invalid("invalid effect status state")), - }; - let status = Self { - state, - attempt: decoder.read_u32()?, - token_present: decoder.read_bool()?, - lease_until_ms: Option::::decode(decoder)?, - expires_at_ms: decoder.read_i64()?, - result: Option::>::decode(decoder)?, - }; - if status - .result - .as_ref() - .is_some_and(|result| result.len() > MAX_EFFECT_RESULT_BYTES) - { - return Err(CodecError::Invalid( - "effect status result exceeds wire limit", - )); - } - Ok(status) - } -} - -impl WireValue for EffectAckRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if self.result.len() > MAX_EFFECT_RESULT_BYTES { - return Err(CodecError::Invalid("effect result exceeds wire limit")); - } - self.lease.encode(encoder)?; - encoder.write_bytes(&self.result) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let lease = EffectLease::decode(decoder)?; - let result = decoder.read_bytes()?.to_vec(); - if result.len() > MAX_EFFECT_RESULT_BYTES { - return Err(CodecError::Invalid("effect result exceeds wire limit")); - } - Ok(Self { lease, result }) - } -} - -impl WireValue for EffectLeaseRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Ack(request) => { - encoder.write_u8(ACK_TAG)?; - request.encode(encoder) - } - Self::Retry(lease) => { - encoder.write_u8(RETRY_TAG)?; - lease.encode(encoder) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - ACK_TAG => Ok(Self::Ack(EffectAckRequest::decode(decoder)?)), - RETRY_TAG => Ok(Self::Retry(EffectLease::decode(decoder)?)), - _ => Err(CodecError::Invalid("invalid effect lease request tag")), - } - } -} - -impl WireValue for EffectLeaseOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Delivered => encoder.write_u8(DELIVERED_TAG), - Self::Retrying { due_at_ms } => { - encoder.write_u8(RETRYING_TAG)?; - encoder.write_i64(*due_at_ms) - } - Self::Failed => encoder.write_u8(FAILED_TAG), - Self::Extended { lease_until_ms } => { - encoder.write_u8(EXTENDED_TAG)?; - encoder.write_i64(*lease_until_ms) - } - Self::LeaseLost => encoder.write_u8(LEASE_LOST_TAG), - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - DELIVERED_TAG => Ok(Self::Delivered), - RETRYING_TAG => Ok(Self::Retrying { - due_at_ms: decoder.read_i64()?, - }), - FAILED_TAG => Ok(Self::Failed), - EXTENDED_TAG => Ok(Self::Extended { - lease_until_ms: decoder.read_i64()?, - }), - LEASE_LOST_TAG => Ok(Self::LeaseLost), - _ => Err(CodecError::Invalid("invalid effect lease outcome tag")), - } - } -} - -fn validate_claim(claim: &EffectClaim) -> Result<(), CodecError> { - if claim.operation.is_empty() - || claim.operation.len() > MAX_EFFECT_OPERATION_BYTES - || !(1..=20).contains(&claim.attempt) - || claim.token.iter().all(|byte| *byte == 0) - || claim.lease_until_ms <= 0 - || claim.expires_at_ms < claim.lease_until_ms - || claim.created_sequence == 0 - { - return Err(CodecError::Invalid("invalid effect claim")); - } - Ok(()) -} - -fn validate_lease(lease: &EffectLease) -> Result<(), CodecError> { - if !(1..=20).contains(&lease.attempt) - || lease.token.iter().all(|byte| *byte == 0) - || lease.expires_at_ms <= 0 - { - return Err(CodecError::Invalid("invalid effect lease")); - } - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - - fn claim(operation_bytes: usize) -> EffectClaim { - EffectClaim { - effect_id: [1; 32], - destination: crate::CellId::from_bytes([2; 32]), - operation: vec![3; operation_bytes], - operation_digest: crate::Digest::from_bytes([4; 32]), - attempt: 1, - token: [5; 16], - lease_until_ms: 10, - expires_at_ms: 11, - created_sequence: 1, - } - } - - #[test] - fn maximum_claim_and_ack_fit_the_registry_wire_limit() { - let claim = claim(MAX_EFFECT_OPERATION_BYTES); - let mut claims = BoundedEncoder::new(1 << 20).unwrap(); - vec![claim.clone()].encode(&mut claims).unwrap(); - assert_eq!(claims.finish().len(), 1 << 20); - - let mut acknowledgement = BoundedEncoder::new(1 << 20).unwrap(); - EffectLeaseRequest::Ack(EffectAckRequest { - lease: EffectLease::from(&claim), - result: vec![6; MAX_EFFECT_RESULT_BYTES], - }) - .encode(&mut acknowledgement) - .unwrap(); - assert_eq!(acknowledgement.finish().len(), 1 << 20); - } - - #[test] - fn claim_and_ack_reject_values_one_byte_past_the_wire_budget() { - let mut oversized_claim = BoundedEncoder::new(1 << 20).unwrap(); - assert!( - vec![claim(MAX_EFFECT_OPERATION_BYTES + 1)] - .encode(&mut oversized_claim) - .is_err() - ); - - let mut oversized_ack = BoundedEncoder::new(1 << 20).unwrap(); - assert!( - EffectLeaseRequest::Ack(EffectAckRequest { - lease: EffectLease::from(&claim(1)), - result: vec![7; MAX_EFFECT_RESULT_BYTES + 1], - }) - .encode(&mut oversized_ack) - .is_err() - ); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/effects/lease.rs b/crates/crab-cell-runtime/src/primitives/effects/lease.rs deleted file mode 100644 index 341cbbd90..000000000 --- a/crates/crab-cell-runtime/src/primitives/effects/lease.rs +++ /dev/null @@ -1,277 +0,0 @@ -//! Claiming, validating, acking, retrying, and extending effect leases. -//! -//! Every transition here is a CAS on one attempt: a claim reserves a bounded -//! lease, an ack settles it, and a retry or extension only moves forward, so a -//! duplicate delivery can never double-apply an effect. - -use super::*; - -/// Reclaims expired leases and claims bounded due effects for private delivery. -pub fn effect_claim( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, - lease_ms: u32, - tokens: &mut impl EffectTokenSource, -) -> Result> { - validate_now(now_ms)?; - if !(1..=MAX_CLAIM_ITEMS).contains(&limit) { - return Err(Error::Command("effect claim limit must be in 1..=32")); - } - if !(MIN_LEASE_MS..=MAX_LEASE_MS).contains(&lease_ms) { - return Err(Error::Command("effect lease must be in 5..=300 seconds")); - } - effect_reclaim_expired_bounded(transaction, now_ms, MAX_RECLAIM_ITEMS)?; - let candidates = { - let mut statement = transaction.prepare( - "SELECT effect_id, destination, operation, attempt, expires_at_ms, created_sequence FROM sys_effects INDEXED BY sys_effects_due WHERE state = 0 AND due_at_ms <= ?1 AND expires_at_ms > ?1 AND attempt < ?2 ORDER BY due_at_ms, effect_id LIMIT ?3", - )?; - statement - .query_map( - (now_ms, i64::from(MAX_ATTEMPTS), (limit + 1) as i64), - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, Vec>(2)?, - row.get::<_, i64>(3)?, - row.get::<_, i64>(4)?, - row.get::<_, i64>(5)?, - )) - }, - )? - .collect::, _>>()? - }; - let requested_deadline = now_ms - .checked_add(i64::from(lease_ms)) - .ok_or(Error::Command("effect lease deadline overflow"))?; - let mut bytes = EFFECT_CLAIM_LIST_BYTES; - let mut claimed = Vec::with_capacity(limit); - for (effect_id, destination, operation, attempt, expires_at_ms, created_sequence) in candidates - { - if claimed.len() == limit { - break; - } - let effect_id = exact::<32>(effect_id, "invalid stored effect ID")?; - let destination = CellId::from_bytes(exact::<32>( - destination, - "invalid stored effect destination", - )?); - if operation.is_empty() - || operation.len() > MAX_EFFECT_OPERATION_BYTES - || attempt < 0 - || created_sequence <= 0 - { - return Err(Error::Command("invalid stored effect")); - } - let prospective = bytes - .checked_add(EFFECT_CLAIM_FIXED_BYTES) - .and_then(|bytes| bytes.checked_add(operation.len())) - .ok_or(Error::Command("effect claim byte count overflow"))?; - if prospective > MAX_EFFECT_BYTES { - break; - } - let attempt = - u32::try_from(attempt).map_err(|_| Error::Command("invalid stored effect attempt"))?; - let created_sequence = u64::try_from(created_sequence) - .map_err(|_| Error::Command("invalid stored effect sequence"))?; - let token = tokens.next_token()?; - let lease_until_ms = requested_deadline.min(expires_at_ms); - if transaction.execute( - "UPDATE sys_effects SET state = 1, attempt = attempt + 1, token = ?1, lease_until_ms = ?2 WHERE effect_id = ?3 AND state = 0 AND due_at_ms <= ?4 AND expires_at_ms > ?4 AND attempt = ?5", - ( - token.as_slice(), - lease_until_ms, - effect_id.as_slice(), - now_ms, - i64::from(attempt), - ), - )? != 1 - { - return Err(Error::Command("effect claim lost selected row")); - } - bytes = prospective; - claimed.push(EffectClaim { - effect_id, - destination, - operation_digest: effect_operation_digest(destination, effect_id, &operation), - operation, - attempt: attempt + 1, - token, - lease_until_ms, - expires_at_ms, - created_sequence, - }); - } - Ok(claimed) -} - -/// Verifies that every effect lease is still published and safe to emit. -pub fn effect_validate_claim( - connection: &Connection, - now_ms: i64, - claimed: &[EffectClaim], -) -> Result { - validate_now(now_ms)?; - if claimed.len() > MAX_CLAIM_ITEMS { - return Err(Error::Command("effect claim exceeds 32 items")); - } - let minimum = now_ms - .checked_add(DELIVERY_MARGIN_MS) - .ok_or(Error::Command("effect delivery margin overflow"))?; - for effect in claimed { - if effect.operation.is_empty() - || effect.operation.len() > MAX_EFFECT_OPERATION_BYTES - || effect.lease_until_ms < minimum - || effect.operation_digest - != effect_operation_digest(effect.destination, effect.effect_id, &effect.operation) - { - return Ok(false); - } - let live = connection - .query_row( - "SELECT 1 FROM sys_effects WHERE effect_id = ?1 AND destination = ?2 AND operation = ?3 AND state = 1 AND attempt = ?4 AND token = ?5 AND lease_until_ms = ?6 AND lease_until_ms >= ?7 AND expires_at_ms = ?8 AND created_sequence = ?9", - ( - effect.effect_id.as_slice(), - effect.destination.as_bytes().as_slice(), - effect.operation.as_slice(), - i64::from(effect.attempt), - effect.token.as_slice(), - effect.lease_until_ms, - minimum, - effect.expires_at_ms, - effect.created_sequence as i64, - ), - |_| Ok(()), - ) - .optional()? - .is_some(); - if !live { - return Ok(false); - } - } - Ok(true) -} - -/// Marks one exact published effect lease delivered after target publication. -pub fn effect_ack_delivered( - transaction: &Transaction<'_>, - now_ms: i64, - claim: &EffectClaim, - result: &[u8], -) -> Result { - effect_ack_lease(transaction, now_ms, EffectLease::from(claim), result) -} - -pub(crate) fn effect_ack_lease( - transaction: &Transaction<'_>, - now_ms: i64, - lease: EffectLease, - result: &[u8], -) -> Result { - validate_now(now_ms)?; - if result.len() > MAX_EFFECT_RESULT_BYTES { - return Err(Error::Command( - "effect result exceeds acknowledgement wire limit", - )); - } - if apply_lease( - transaction, - now_ms, - &lease, - EffectState::Delivered, - None, - Some(result), - )? { - Ok(EffectLeaseOutcome::Delivered) - } else { - Ok(EffectLeaseOutcome::LeaseLost) - } -} - -/// Releases one failed delivery for bounded retry or terminal expiry. -pub fn effect_retry( - transaction: &Transaction<'_>, - now_ms: i64, - claim: &EffectClaim, -) -> Result { - effect_retry_lease(transaction, now_ms, EffectLease::from(claim)) -} - -pub(crate) fn effect_retry_lease( - transaction: &Transaction<'_>, - now_ms: i64, - lease: EffectLease, -) -> Result { - validate_now(now_ms)?; - let delay = retry_delay_ms(lease.attempt); - let due_at_ms = now_ms.saturating_add(delay); - let state = if lease.attempt >= MAX_ATTEMPTS || due_at_ms >= lease.expires_at_ms { - EffectState::Failed - } else { - EffectState::Ready - }; - if !apply_lease( - transaction, - now_ms, - &lease, - state, - (state == EffectState::Ready).then_some(due_at_ms), - None, - )? { - return Ok(EffectLeaseOutcome::LeaseLost); - } - match state { - EffectState::Ready => Ok(EffectLeaseOutcome::Retrying { due_at_ms }), - EffectState::Failed => Ok(EffectLeaseOutcome::Failed), - _ => Err(Error::Command("invalid effect retry state")), - } -} - -/// Extends one current effect lease without shortening it. -pub fn effect_extend( - transaction: &Transaction<'_>, - now_ms: i64, - claim: &EffectClaim, - extension_ms: u32, -) -> Result { - validate_now(now_ms)?; - if !(MIN_LEASE_MS..=MAX_LEASE_MS).contains(&extension_ms) { - return Err(Error::Command( - "effect extension must be in 5..=300 seconds", - )); - } - let current = transaction - .query_row( - "SELECT lease_until_ms, expires_at_ms FROM sys_effects WHERE effect_id = ?1 AND state = 1 AND attempt = ?2 AND token = ?3 AND lease_until_ms > ?4", - ( - claim.effect_id.as_slice(), - i64::from(claim.attempt), - claim.token.as_slice(), - now_ms, - ), - |row| Ok((row.get::<_, i64>(0)?, row.get::<_, i64>(1)?)), - ) - .optional()?; - let Some((current, expires_at_ms)) = current else { - return Ok(EffectLeaseOutcome::LeaseLost); - }; - let requested = now_ms - .checked_add(i64::from(extension_ms)) - .ok_or(Error::Command("effect extension deadline overflow"))?; - let lease_until_ms = current.max(requested.min(expires_at_ms)); - if transaction.execute( - "UPDATE sys_effects SET lease_until_ms = ?1 WHERE effect_id = ?2 AND state = 1 AND attempt = ?3 AND token = ?4 AND lease_until_ms = ?5", - ( - lease_until_ms, - claim.effect_id.as_slice(), - i64::from(claim.attempt), - claim.token.as_slice(), - current, - ), - )? != 1 - { - return Err(Error::Command("effect lease changed during extension")); - } - Ok(EffectLeaseOutcome::Extended { lease_until_ms }) -} diff --git a/crates/crab-cell-runtime/src/primitives/effects/supervisor.rs b/crates/crab-cell-runtime/src/primitives/effects/supervisor.rs deleted file mode 100644 index 7a556e20b..000000000 --- a/crates/crab-cell-runtime/src/primitives/effects/supervisor.rs +++ /dev/null @@ -1,297 +0,0 @@ -use std::{ - fmt, - time::{SystemTime, UNIX_EPOCH}, -}; - -use rand::RngCore; - -use crate::Error; -use crate::cell::executor::StoredOutcome; -use crate::cell::executor::{MutationIdentity, Resolution}; -use crate::client::{InvocationError, PendingMutation, Receipt}; -use crate::identity::RequestId; -use crate::peer::EffectPeerClient; - -use super::{EffectClaim, EffectClaimRequest, EffectLeaseOutcome, EffectModule, EffectSource}; - -const SUPERVISOR_IDENTITY_LIFETIME_MS: i64 = 60_000; - -/// Outcome of one bounded source claim, destination delivery and source transition. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum EffectRunOutcome { - /// No claimable effect remained. - Idle { - /// Receipt the observation is bound to. - receipt: Receipt, - }, - /// The destination applied the effect and the source recorded it. - Delivered { - /// Destination outcome the source recorded. - destination: StoredOutcome, - /// Receipt the source transition published. - receipt: Receipt, - }, - /// The destination asked for another attempt. - Retrying { - /// Logical time the next attempt is due. - due_at_ms: i64, - /// Receipt the source transition published. - receipt: Receipt, - }, - /// The destination rejected the effect terminally. - Failed { - /// Receipt the source transition published. - receipt: Receipt, - }, - /// The source lease was lost before the transition. - LeaseLost { - /// Receipt the failed resolution is bound to. - receipt: Receipt, - }, -} - -/// Failure that preserves unresolved source mutation evidence. -pub enum EffectSupervisorError { - /// A source transition is committed but unresolved; resolve it before - /// running another cycle. - Pending(Box), - /// The published source result could not be decoded. - InvalidPublishedResult { - /// Receipt the published result was observed at. - receipt: Receipt, - /// Decoding failure that produced this error. - source: Box, - }, - /// The supervisor failed before it could resolve the source transition. - Runtime(Error), -} - -impl fmt::Debug for EffectSupervisorError { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Pending(pending) => formatter.debug_tuple("Pending").field(pending).finish(), - Self::InvalidPublishedResult { receipt, source } => formatter - .debug_struct("InvalidPublishedResult") - .field("receipt", receipt) - .field("source", source) - .finish(), - Self::Runtime(error) => formatter.debug_tuple("Runtime").field(error).finish(), - } - } -} - -impl fmt::Display for EffectSupervisorError { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Pending(_) => formatter.write_str("effect supervisor mutation needs resolution"), - Self::InvalidPublishedResult { .. } => { - formatter.write_str("effect supervisor received an invalid published result") - } - Self::Runtime(error) => write!(formatter, "effect supervisor failed: {error}"), - } - } -} - -impl std::error::Error for EffectSupervisorError { - fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { - match self { - Self::InvalidPublishedResult { source, .. } => Some(source.as_ref()), - Self::Runtime(error) => Some(error), - Self::Pending(_) => None, - } - } -} - -/// Delivers published effects without holding a SQLite transaction across await. -pub struct EffectSupervisor { - source: EffectSource, - peer: EffectPeerClient, - lease_ms: u32, -} - -impl EffectSupervisor { - /// Creates a supervisor using a 5..=300 second source lease. - pub fn new( - source: EffectSource, - peer: EffectPeerClient, - lease_ms: u32, - ) -> crate::Result { - if !(5_000..=300_000).contains(&lease_ms) { - return Err(Error::Command( - "effect supervisor lease must be in 5..=300 seconds", - )); - } - Ok(Self { - source, - peer, - lease_ms, - }) - } - - /// Claims and delivers at most one effect from the explicit source Cell. - pub async fn run_once(&self) -> std::result::Result { - let claimed = self - .source - .claim( - internal_identity()?, - EffectClaimRequest { - limit: 1, - lease_ms: self.lease_ms, - }, - ) - .await - .map_err(unexpected_invocation)?; - let Some(claim) = claimed.output.into_iter().next() else { - return Ok(EffectRunOutcome::Idle { - receipt: claimed.receipt, - }); - }; - let validation = self - .source - .validate(vec![claim.clone()], claimed.receipt) - .await - .map_err(unexpected_invocation)?; - if !validation.output { - return Ok(EffectRunOutcome::LeaseLost { - receipt: validation.receipt, - }); - } - - let now_ms = system_time_ms()?; - let destination = match self.peer.deliver(&claim, now_ms).await { - Ok(outcome) => Some(outcome), - Err(Error::EffectOutcomeUnknown { .. }) => { - match self.peer.resolve(&claim, now_ms).await { - Ok(Resolution::Committed(outcome)) => Some(outcome), - Ok(Resolution::Absent | Resolution::Unknown | Resolution::Expired) => None, - Err(error) => return Err(EffectSupervisorError::Runtime(error)), - } - } - Err(error) if retryable_delivery_error(&error) => None, - Err(error) => return Err(EffectSupervisorError::Runtime(error)), - }; - match destination { - Some(destination) => self.ack(claim, destination).await, - None => self.retry(claim).await, - } - } - - async fn ack( - &self, - claim: EffectClaim, - destination: StoredOutcome, - ) -> std::result::Result { - match self - .source - .ack(internal_identity()?, claim, destination.result().to_vec()) - .await - { - Ok(committed) if committed.output == EffectLeaseOutcome::Delivered => { - Ok(EffectRunOutcome::Delivered { - destination, - receipt: committed.receipt, - }) - } - Err(InvocationError::Rejected(committed)) - if committed.output == EffectLeaseOutcome::LeaseLost => - { - Ok(EffectRunOutcome::LeaseLost { - receipt: committed.receipt, - }) - } - Ok(_) | Err(InvocationError::Rejected(_)) => Err(EffectSupervisorError::Runtime( - Error::Command("effect acknowledgement returned an invalid outcome"), - )), - Err(error) => Err(unexpected_invocation(error)), - } - } - - async fn retry( - &self, - claim: EffectClaim, - ) -> std::result::Result { - match self.source.retry(internal_identity()?, claim).await { - Ok(committed) => match committed.output { - EffectLeaseOutcome::Retrying { due_at_ms } => Ok(EffectRunOutcome::Retrying { - due_at_ms, - receipt: committed.receipt, - }), - EffectLeaseOutcome::Failed => Ok(EffectRunOutcome::Failed { - receipt: committed.receipt, - }), - _ => Err(EffectSupervisorError::Runtime(Error::Command( - "effect retry returned an invalid outcome", - ))), - }, - Err(InvocationError::Rejected(committed)) - if committed.output == EffectLeaseOutcome::LeaseLost => - { - Ok(EffectRunOutcome::LeaseLost { - receipt: committed.receipt, - }) - } - Err(InvocationError::Rejected(_)) => Err(EffectSupervisorError::Runtime( - Error::Command("effect retry returned an invalid rejection"), - )), - Err(error) => Err(unexpected_invocation(error)), - } - } -} - -fn retryable_delivery_error(error: &Error) -> bool { - matches!( - error, - Error::PeerTransport { .. } - | Error::PeerTransportUnknown { .. } - | Error::EffectExpired - | Error::CellNotActive - | Error::CellDraining - | Error::Fenced - | Error::Capacity(_) - | Error::Deadline - | Error::RuntimeClosed - ) -} - -fn internal_identity() -> std::result::Result { - let issued_at_ms = system_time_ms()?; - let expires_at_ms = issued_at_ms - .checked_add(SUPERVISOR_IDENTITY_LIFETIME_MS) - .ok_or(EffectSupervisorError::Runtime(Error::Command( - "effect mutation expiry overflow", - )))?; - let mut request_id = [0; 16]; - rand::rng().fill_bytes(&mut request_id); - Ok(MutationIdentity { - request_id: RequestId::from_bytes(request_id), - issued_at_ms, - expires_at_ms, - }) -} - -fn system_time_ms() -> std::result::Result { - i64::try_from( - SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_err(|_| { - EffectSupervisorError::Runtime(Error::Command("system clock is before Unix epoch")) - })? - .as_millis(), - ) - .map_err(|_| { - EffectSupervisorError::Runtime(Error::Command("system clock exceeds i64 milliseconds")) - }) -} - -fn unexpected_invocation(error: InvocationError) -> EffectSupervisorError { - match error { - InvocationError::Pending(pending) => EffectSupervisorError::Pending(pending), - InvocationError::InvalidPublishedResult { receipt, source } => { - EffectSupervisorError::InvalidPublishedResult { receipt, source } - } - InvocationError::NotStarted(error) => EffectSupervisorError::Runtime(error), - InvocationError::Rejected(_) => EffectSupervisorError::Runtime(Error::Command( - "effect supervisor command was unexpectedly rejected", - )), - } -} diff --git a/crates/crab-cell-runtime/src/primitives/effects/tests.rs b/crates/crab-cell-runtime/src/primitives/effects/tests.rs deleted file mode 100644 index d0ff53516..000000000 --- a/crates/crab-cell-runtime/src/primitives/effects/tests.rs +++ /dev/null @@ -1,597 +0,0 @@ -use super::EffectBatch; -use crate::cell::executor::HandlerOutcome; -use crate::cell::schema::install_runtime_schema; -use crate::identity::IncarnationId; -use crate::identity::{ApplicationId, CellId, CellTarget, Digest, NamespaceId, TenantId}; -use crate::peer::wire as peer_wire; -use crate::primitives::effects::{ - EffectCommandIntent, EffectLeaseOutcome, EffectTokenSource, InboxApplyOutcome, InboxDelivery, - effect_ack_delivered, effect_claim, effect_cleanup_terminal, effect_extend, effect_retry, - effect_validate_claim, inbox_apply, inbox_cleanup_expired, -}; - -use super::{EFFECT_LIFETIME_MS, MAX_ATTEMPTS, MAX_CLAIM_ITEMS, MAX_LEASE_MS, MIN_LEASE_MS}; -use prost::Message; - -struct Tokens(u8); - -impl EffectTokenSource for Tokens { - fn next_token(&mut self) -> crate::Result<[u8; 16]> { - self.0 = self - .0 - .checked_add(1) - .ok_or(crate::Error::Command("test effect token overflow"))?; - Ok([self.0; 16]) - } -} - -fn connection(cell: u8, incarnation: u8) -> crab_ltx::rusqlite::Connection { - let mut connection = crab_ltx::rusqlite::Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - CellId::from_bytes([cell; 32]), - IncarnationId::from_bytes([incarnation; 16]), - 1, - ) - .unwrap(); - connection - .execute("CREATE TABLE applied(value BLOB NOT NULL)", []) - .unwrap(); - connection -} - -fn source_target() -> CellTarget { - CellTarget::new( - TenantId::from_bytes([9; 16]), - ApplicationId::from_bytes([10; 16]), - NamespaceId::from_bytes([11; 16]), - b"source-partition", - ) - .unwrap() -} - -fn source_connection() -> crab_ltx::rusqlite::Connection { - let mut connection = crab_ltx::rusqlite::Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - source_target().cell_id(), - IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - connection -} - -fn command_intent(target: CellTarget, input: &[u8], expires_at_ms: i64) -> EffectCommandIntent { - EffectCommandIntent { - target, - command_id: 1, - codec_version: 1, - input: input.to_vec(), - expires_at_ms, - } -} - -#[test] -fn command_effect_is_canonical_and_does_not_pin_destination_incarnation() { - let mut source = source_connection(); - let source_target = source_target(); - let target = CellTarget::new( - source_target.tenant(), - source_target.application(), - NamespaceId::from_bytes([5; 16]), - b"target-partition", - ) - .unwrap(); - let transaction = source.transaction().unwrap(); - let mut batch = EffectBatch::new(&transaction, &source_target, 1, 10).unwrap(); - let effect_id = batch - .insert_command( - &transaction, - &EffectCommandIntent { - target: target.clone(), - command_id: 7, - codec_version: 1, - input: b"typed-input".to_vec(), - expires_at_ms: 10_000, - }, - ) - .unwrap(); - let (destination, operation): (Vec, Vec) = transaction - .query_row( - "SELECT destination, operation FROM sys_effects WHERE effect_id = ?1", - [effect_id.as_slice()], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .unwrap(); - let request = peer_wire::EffectRequest::decode(operation.as_slice()).unwrap(); - assert_eq!(destination.as_slice(), target.cell_id().as_bytes()); - assert!(request.destination_incarnation.is_empty()); - assert_eq!(request.encode_to_vec(), operation); - let identity = request.identity.unwrap(); - assert_eq!(identity.effect_id, effect_id); - assert_eq!(identity.source_cell, source_target.cell_id().as_bytes()); - assert_eq!(identity.source_incarnation, [2; 16]); - assert_eq!(identity.source_sequence, 1); - assert_eq!(identity.ordinal, 0); - transaction.commit().unwrap(); -} - -#[test] -fn command_effect_rejects_a_source_target_for_another_cell() { - let mut source = source_connection(); - let other = CellTarget::new( - source_target().tenant(), - source_target().application(), - source_target().namespace(), - b"another-source", - ) - .unwrap(); - let transaction = source.transaction().unwrap(); - assert!(EffectBatch::new(&transaction, &other, 1, 10).is_err()); -} - -#[test] -fn command_effect_rejects_a_foreign_application_before_writes() { - let mut source = source_connection(); - let source_target = source_target(); - let target = CellTarget::new( - TenantId::from_bytes([3; 16]), - ApplicationId::from_bytes([4; 16]), - NamespaceId::from_bytes([5; 16]), - b"foreign-application", - ) - .unwrap(); - let transaction = source.transaction().unwrap(); - let result = EffectBatch::new(&transaction, &source_target, 1, 10) - .unwrap() - .insert_command(&transaction, &command_intent(target, b"foreign", 10_000)); - assert!(matches!( - result, - Err(crate::Error::Identity( - "effect target is outside the source application scope" - )) - )); - let count: i64 = transaction - .query_row("SELECT COUNT(*) FROM sys_effects", [], |row| row.get(0)) - .unwrap(); - assert_eq!(count, 0); - transaction.rollback().unwrap(); -} - -#[test] -fn target_commit_and_lost_response_retry_execute_destination_once() { - const EXPIRES_AT_MS: i64 = 10_000; - const INBOX_RETENTION_MS: i64 = 7 * 24 * 60 * 60 * 1_000; - let source_target = source_target(); - let destination = CellTarget::new( - source_target.tenant(), - source_target.application(), - NamespaceId::from_bytes([5; 16]), - b"destination", - ) - .unwrap(); - let mut source = source_connection(); - let mut target = connection(3, 4); - - let source_transaction = source.transaction().unwrap(); - let operation = command_intent(destination.clone(), b"apply-value", EXPIRES_AT_MS); - let effect_id = EffectBatch::new(&source_transaction, &source_target, 1, 0) - .unwrap() - .insert_command(&source_transaction, &operation) - .unwrap(); - assert_eq!( - EffectBatch::new(&source_transaction, &source_target, 1, 0) - .unwrap() - .insert_command(&source_transaction, &operation) - .unwrap(), - effect_id - ); - assert!( - EffectBatch::new(&source_transaction, &source_target, 1, 0) - .unwrap() - .insert_command( - &source_transaction, - &command_intent(destination, b"changed", EXPIRES_AT_MS), - ) - .is_err() - ); - source_transaction.commit().unwrap(); - - let source_transaction = source.transaction().unwrap(); - let mut tokens = Tokens(0); - let first = effect_claim(&source_transaction, 10, 1, 5_000, &mut tokens) - .unwrap() - .remove(0); - source_transaction.commit().unwrap(); - assert!(effect_validate_claim(&source, 11, std::slice::from_ref(&first)).unwrap()); - - let delivery = InboxDelivery { - effect_id: first.effect_id, - operation_digest: first.operation_digest, - expires_at_ms: first.expires_at_ms, - }; - let target_transaction = target.transaction().unwrap(); - assert_eq!( - inbox_apply(&target_transaction, 12, delivery, 128, |transaction| { - transaction.execute( - "INSERT INTO applied(value) VALUES (?1)", - [b"value".as_slice()], - )?; - Ok(HandlerOutcome::Success(b"ok".to_vec())) - }) - .unwrap(), - InboxApplyOutcome::Success { - result: b"ok".to_vec(), - commit_sequence: 1, - duplicate: false, - } - ); - target_transaction.commit().unwrap(); - - let source_transaction = source.transaction().unwrap(); - assert!( - effect_claim(&source_transaction, 5_010, 1, 5_000, &mut tokens) - .unwrap() - .is_empty() - ); - source_transaction.commit().unwrap(); - let source_transaction = source.transaction().unwrap(); - let second = effect_claim(&source_transaction, 5_210, 1, 5_000, &mut tokens) - .unwrap() - .remove(0); - assert_eq!(second.effect_id, first.effect_id); - assert_eq!(second.operation, first.operation); - assert_eq!(second.operation_digest, first.operation_digest); - assert_eq!(second.attempt, 2); - source_transaction.commit().unwrap(); - - let target_transaction = target.transaction().unwrap(); - assert_eq!( - inbox_apply(&target_transaction, 5_211, delivery, 128, |_| { - panic!("duplicate inbox delivery must not invoke its handler") - }) - .unwrap(), - InboxApplyOutcome::Success { - result: b"ok".to_vec(), - commit_sequence: 1, - duplicate: true, - } - ); - assert_eq!( - inbox_apply( - &target_transaction, - 5_211, - InboxDelivery { - operation_digest: Digest::from_bytes([99; 32]), - ..delivery - }, - 128, - |_| panic!("conflicting inbox delivery must not invoke its handler"), - ) - .unwrap(), - InboxApplyOutcome::Conflict - ); - target_transaction.commit().unwrap(); - assert_eq!( - target - .query_row("SELECT count(*) FROM applied", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 1 - ); - - let source_transaction = source.transaction().unwrap(); - assert_eq!( - effect_ack_delivered(&source_transaction, 5_212, &second, b"ok").unwrap(), - EffectLeaseOutcome::Delivered - ); - assert_eq!( - effect_ack_delivered(&source_transaction, 5_212, &second, b"ok").unwrap(), - EffectLeaseOutcome::LeaseLost - ); - assert_eq!( - effect_cleanup_terminal(&source_transaction, EXPIRES_AT_MS).unwrap(), - 1 - ); - source_transaction.commit().unwrap(); - - let target_transaction = target.transaction().unwrap(); - assert_eq!( - inbox_cleanup_expired(&target_transaction, EXPIRES_AT_MS).unwrap(), - 0 - ); - assert_eq!( - inbox_cleanup_expired(&target_transaction, EXPIRES_AT_MS + INBOX_RETENTION_MS,).unwrap(), - 1 - ); - target_transaction.commit().unwrap(); -} - -#[test] -fn rejected_effect_rolls_back_target_writes_but_publishes_inbox_result() { - let mut target = connection(3, 4); - let delivery = InboxDelivery { - effect_id: [5; 32], - operation_digest: Digest::from_bytes([6; 32]), - expires_at_ms: 10_000, - }; - let transaction = target.transaction().unwrap(); - assert_eq!( - inbox_apply(&transaction, 10, delivery, 128, |transaction| { - transaction.execute("INSERT INTO applied(value) VALUES (X'01')", [])?; - Ok(HandlerOutcome::Rejected(b"denied".to_vec())) - }) - .unwrap(), - InboxApplyOutcome::Rejected { - result: b"denied".to_vec(), - commit_sequence: 1, - duplicate: false, - } - ); - assert_eq!( - transaction - .query_row("SELECT count(*) FROM applied", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 0 - ); - assert_eq!( - inbox_apply(&transaction, 11, delivery, 128, |_| { - panic!("stored rejection must not invoke its handler") - }) - .unwrap(), - InboxApplyOutcome::Rejected { - result: b"denied".to_vec(), - commit_sequence: 1, - duplicate: true, - } - ); - transaction.commit().unwrap(); -} - -#[test] -fn one_command_cannot_exceed_effect_count_or_byte_limits() { - let source_target = source_target(); - let destination = CellTarget::new( - source_target.tenant(), - source_target.application(), - NamespaceId::from_bytes([5; 16]), - b"destination", - ) - .unwrap(); - let mut source = source_connection(); - let transaction = source.transaction().unwrap(); - let mut effects = EffectBatch::new(&transaction, &source_target, 1, 0).unwrap(); - assert!( - effects - .insert_command( - &transaction, - &command_intent(destination.clone(), &vec![0; 1 << 20], 10_000), - ) - .is_err() - ); - for ordinal in 0..128 { - effects - .insert_command( - &transaction, - &command_intent(destination.clone(), &[ordinal as u8], 10_000), - ) - .unwrap(); - } - assert!( - effects - .insert_command( - &transaction, - &command_intent(destination, b"overflow", 10_000), - ) - .is_err() - ); - transaction.commit().unwrap(); - assert_eq!( - source - .query_row("SELECT count(*) FROM sys_effects", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 128 - ); -} - -#[test] -fn byte_limit_rejection_does_not_consume_the_next_effect_ordinal() { - let source_target = source_target(); - let destination = CellTarget::new( - source_target.tenant(), - source_target.application(), - NamespaceId::from_bytes([5; 16]), - b"destination", - ) - .unwrap(); - let mut source = source_connection(); - let transaction = source.transaction().unwrap(); - let mut effects = EffectBatch::new(&transaction, &source_target, 1, 0).unwrap(); - let large = vec![0; 9_000]; - let mut successful = 0_u32; - while effects - .insert_command( - &transaction, - &command_intent(destination.clone(), &large, 10_000), - ) - .is_ok() - { - successful += 1; - } - let effect_id = effects - .insert_command( - &transaction, - &command_intent(destination, b"after-limit", 10_000), - ) - .unwrap(); - let operation: Vec = transaction - .query_row( - "SELECT operation FROM sys_effects WHERE effect_id = ?1", - [effect_id.as_slice()], - |row| row.get(0), - ) - .unwrap(); - let request = peer_wire::EffectRequest::decode(operation.as_slice()).unwrap(); - assert_eq!(request.identity.unwrap().ordinal, successful); - transaction.commit().unwrap(); -} - -#[test] -fn manual_retry_preserves_identity_and_never_reopens_terminal_effect() { - let source_target = source_target(); - let destination = CellTarget::new( - source_target.tenant(), - source_target.application(), - NamespaceId::from_bytes([5; 16]), - b"destination", - ) - .unwrap(); - let mut source = source_connection(); - let transaction = source.transaction().unwrap(); - EffectBatch::new(&transaction, &source_target, 1, 0) - .unwrap() - .insert_command(&transaction, &command_intent(destination, b"work", 5_100)) - .unwrap(); - let mut tokens = Tokens(0); - let claim = effect_claim(&transaction, 0, 1, 5_000, &mut tokens) - .unwrap() - .remove(0); - assert_eq!( - crate::primitives::effects::effect_retry(&transaction, 4_999, &claim).unwrap(), - EffectLeaseOutcome::Failed - ); - assert!( - effect_claim(&transaction, 4_999, 1, 5_000, &mut tokens) - .unwrap() - .is_empty() - ); - transaction.commit().unwrap(); -} - -#[test] -fn effect_expiry_accepts_the_seven_day_boundary() { - let source_target = source_target(); - let destination = CellTarget::new( - source_target.tenant(), - source_target.application(), - NamespaceId::from_bytes([5; 16]), - b"destination", - ) - .unwrap(); - let mut source = source_connection(); - let transaction = source.transaction().unwrap(); - let mut effects = EffectBatch::new(&transaction, &source_target, 1, 0).unwrap(); - assert!( - effects - .insert_command( - &transaction, - &command_intent(destination, b"work", EFFECT_LIFETIME_MS), - ) - .is_ok() - ); -} - -#[test] -fn effect_expiry_past_seven_days_is_rejected() { - let source_target = source_target(); - let destination = CellTarget::new( - source_target.tenant(), - source_target.application(), - NamespaceId::from_bytes([5; 16]), - b"destination", - ) - .unwrap(); - let mut source = source_connection(); - let transaction = source.transaction().unwrap(); - let mut effects = EffectBatch::new(&transaction, &source_target, 1, 0).unwrap(); - assert!( - effects - .insert_command( - &transaction, - &command_intent(destination, b"work", EFFECT_LIFETIME_MS + 1), - ) - .is_err() - ); -} - -#[test] -fn claim_rejects_limits_outside_one_to_32() { - let mut source = source_connection(); - let transaction = source.transaction().unwrap(); - let mut tokens = Tokens(0); - assert!(effect_claim(&transaction, 0, 0, MIN_LEASE_MS, &mut tokens).is_err()); - assert!( - effect_claim( - &transaction, - 0, - MAX_CLAIM_ITEMS + 1, - MIN_LEASE_MS, - &mut tokens - ) - .is_err() - ); -} - -#[test] -fn claim_rejects_leases_outside_five_to_three_hundred_seconds() { - let mut source = source_connection(); - let transaction = source.transaction().unwrap(); - let mut tokens = Tokens(0); - assert!(effect_claim(&transaction, 0, 1, MIN_LEASE_MS - 1, &mut tokens).is_err()); - assert!(effect_claim(&transaction, 0, 1, MAX_LEASE_MS + 1, &mut tokens).is_err()); -} - -#[test] -fn extend_rejects_bounds_outside_five_to_three_hundred_seconds() { - let (claim, mut source) = claimed_effect(); - let transaction = source.transaction().unwrap(); - assert!(effect_extend(&transaction, 0, &claim, MIN_LEASE_MS - 1).is_err()); - assert!(effect_extend(&transaction, 0, &claim, MAX_LEASE_MS + 1).is_err()); -} - -#[test] -fn retry_fails_at_the_attempt_cap() { - let (claim, mut source) = claimed_effect(); - let transaction = source.transaction().unwrap(); - transaction - .execute( - "UPDATE sys_effects SET attempt = ?1 WHERE effect_id = ?2", - (i64::from(MAX_ATTEMPTS), claim.effect_id.as_slice()), - ) - .unwrap(); - let mut capped = claim; - capped.attempt = MAX_ATTEMPTS; - assert_eq!( - effect_retry(&transaction, 0, &capped).unwrap(), - EffectLeaseOutcome::Failed - ); -} - -fn claimed_effect() -> ( - crate::primitives::effects::EffectClaim, - crab_ltx::rusqlite::Connection, -) { - let source_target = source_target(); - let destination = CellTarget::new( - source_target.tenant(), - source_target.application(), - NamespaceId::from_bytes([5; 16]), - b"destination", - ) - .unwrap(); - let mut source = source_connection(); - let transaction = source.transaction().unwrap(); - EffectBatch::new(&transaction, &source_target, 1, 0) - .unwrap() - .insert_command(&transaction, &command_intent(destination, b"work", 5_100)) - .unwrap(); - let mut tokens = Tokens(0); - let claim = effect_claim(&transaction, 0, 1, MIN_LEASE_MS, &mut tokens) - .unwrap() - .remove(0); - transaction.commit().unwrap(); - (claim, source) -} diff --git a/crates/crab-cell-runtime/src/primitives/kv.rs b/crates/crab-cell-runtime/src/primitives/kv.rs deleted file mode 100644 index e74e1eb54..000000000 --- a/crates/crab-cell-runtime/src/primitives/kv.rs +++ /dev/null @@ -1,475 +0,0 @@ -//! KV primitive: keyed values with compare-and-swap. -use std::collections::HashSet; - -use rusqlite::{Connection, OptionalExtension, Transaction, types::Value}; - -use crate::{Error, Result}; - -mod api; - -pub use api::{ - KvAtomicCommand, KvGetQuery, KvGetRequest, KvListQuery, KvListRequest, KvModule, KvNamespace, - register_kv, -}; - -const KV_SCHEMA: &str = include_str!("../migrations/kv.sql"); -pub(crate) const KV_TABLE: &str = "kv_entries"; -const MAX_SCOPE_BYTES: usize = 1_024; -const MAX_KEY_BYTES: usize = 1_024; -const MAX_VALUE_BYTES: usize = 4 * 1024 * 1024; -const MAX_ATOMIC_ITEMS: usize = 128; -const MAX_OPERATION_BYTES: usize = crate::codec::MAX_WIRE_BYTES; -const MAX_LIST_ITEMS: usize = 1_000; -// Preserve the normal page size; a single larger entry occupies one page. -const MAX_PAGE_BYTES: usize = 1 << 20; -const VERSION_BYTES: usize = 28; -const CLEANUP_ITEMS: usize = 128; - -type RawEntry = (Vec, Vec, Vec, Option); - -/// One condition checked against the logical live value before any KV write. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum KvCondition { - /// Applies only while the key has no live value. - Absent, - /// Applies only while the live version matches. - Version([u8; VERSION_BYTES]), -} - -/// One key precondition in a scoped atomic operation. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct KvCheck { - /// Key the condition applies to. - pub key: Vec, - /// Condition the live value must satisfy. - pub condition: KvCondition, -} - -/// One ordered mutation in a scoped atomic operation. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum KvMutation { - /// Writes one value. - Put { - /// Key to write. - key: Vec, - /// Value bytes to store. - value: Vec, - /// Logical time the value expires, when it should. - expires_at_ms: Option, - }, - /// Removes one key. - Delete { - /// Key to remove. - key: Vec, - }, -} - -impl KvMutation { - fn key(&self) -> &[u8] { - match self { - Self::Put { key, .. } | Self::Delete { key } => key, - } - } -} - -/// All checks and ordered writes applied by one runtime command transaction. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct KvAtomicRequest { - /// Scope the checks and mutations apply to. - pub scope: Vec, - /// Conditions checked before any write. - pub checks: Vec, - /// Ordered writes applied when every check passes. - pub mutations: Vec, -} - -/// Result for one applied mutation, preserving request order. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct KvMutationResult { - /// Key the mutation wrote. - pub key: Vec, - /// Version after the write, absent for a delete. - pub version: Option<[u8; VERSION_BYTES]>, - /// Whether the mutation removed the key. - pub deleted: bool, -} - -/// Business outcome produced inside the runtime's application savepoint. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum KvAtomicOutcome { - /// Every check passed and the writes were applied in order. - Applied(Vec), - /// A check failed; no write was applied. - PreconditionFailed { - /// Key whose condition failed. - key: Vec, - }, -} - -/// One live KV entry returned by get or list. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct KvEntry { - /// Entry key. - pub key: Vec, - /// Live value bytes. - pub value: Vec, - /// Current version. - pub version: [u8; VERSION_BYTES], - /// Logical time the entry expires, when it does. - pub expires_at_ms: Option, -} - -/// One bounded, current-read page within a single scope. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct KvPage { - /// Entries in key order. - pub entries: Vec, - /// Key to continue after when the page filled its limit. - pub next_after: Option>, -} - -/// Installs the exact version-one KV schema inside bootstrap or migration SQL. -pub fn install_kv_schema(transaction: &Transaction<'_>) -> Result<()> { - transaction.execute_batch(KV_SCHEMA)?; - Ok(()) -} - -/// Checks and mutates one scope atomically inside the caller's runtime command. -pub fn kv_atomic( - transaction: &Transaction<'_>, - now_ms: i64, - request: &KvAtomicRequest, -) -> Result { - validate_atomic(now_ms, request)?; - let (incarnation, prior_sequence) = transaction.query_row( - "SELECT incarnation, commit_sequence FROM sys_meta WHERE singleton = 1", - [], - |row| Ok((row.get::<_, Vec>(0)?, row.get::<_, i64>(1)?)), - )?; - let incarnation: [u8; 16] = incarnation - .try_into() - .map_err(|_| Error::Command("invalid KV runtime incarnation"))?; - let sequence = prior_sequence - .checked_add(1) - .filter(|value| *value > 0) - .ok_or(Error::Command("KV commit sequence overflow"))?; - - for check in &request.checks { - let version = live_version(transaction, &request.scope, &check.key, now_ms)?; - let satisfied = match (&check.condition, version) { - (KvCondition::Absent, None) => true, - (KvCondition::Version(expected), Some(actual)) => expected.as_slice() == actual, - _ => false, - }; - if !satisfied { - return Ok(KvAtomicOutcome::PreconditionFailed { - key: check.key.clone(), - }); - } - } - - let mut results = Vec::with_capacity(request.mutations.len()); - for (ordinal, mutation) in request.mutations.iter().enumerate() { - match mutation { - KvMutation::Put { - key, - value, - expires_at_ms, - } => { - let ordinal = - u32::try_from(ordinal).map_err(|_| Error::Command("too many KV mutations"))?; - let version = kv_version(incarnation, sequence, ordinal)?; - transaction.execute( - "INSERT INTO kv_entries(scope, key, version, value, expires_at_ms) VALUES (?1, ?2, ?3, ?4, ?5) ON CONFLICT(scope, key) DO UPDATE SET version = excluded.version, value = excluded.value, expires_at_ms = excluded.expires_at_ms", - ( - request.scope.as_slice(), - key.as_slice(), - version.as_slice(), - value.as_slice(), - expires_at_ms, - ), - )?; - results.push(KvMutationResult { - key: key.clone(), - version: Some(version), - deleted: false, - }); - } - KvMutation::Delete { key } => { - transaction.execute( - "DELETE FROM kv_entries WHERE scope = ?1 AND key = ?2", - (request.scope.as_slice(), key.as_slice()), - )?; - results.push(KvMutationResult { - key: key.clone(), - version: None, - deleted: true, - }); - } - } - } - Ok(KvAtomicOutcome::Applied(results)) -} - -/// Reads one logically live value from a scope at the supplied logical time. -pub fn kv_get( - connection: &Connection, - scope: &[u8], - key: &[u8], - now_ms: i64, -) -> Result> { - validate_scope(scope)?; - validate_key(key)?; - validate_now(now_ms)?; - let row = connection - .query_row( - "SELECT key, value, version, expires_at_ms FROM kv_entries WHERE scope = ?1 AND key = ?2 AND (expires_at_ms IS NULL OR expires_at_ms > ?3)", - (scope, key, now_ms), - decode_entry, - ) - .optional()?; - row.map(validate_entry).transpose() -} - -/// Lists a bounded binary-ordered page within one scope and prefix. -pub fn kv_list( - connection: &Connection, - scope: &[u8], - prefix: &[u8], - after_key: Option<&[u8]>, - limit: usize, - now_ms: i64, -) -> Result { - validate_scope(scope)?; - if prefix.len() > MAX_KEY_BYTES { - return Err(Error::Command("KV prefix exceeds 1024 bytes")); - } - if let Some(after_key) = after_key { - validate_key(after_key)?; - } - if !(1..=MAX_LIST_ITEMS).contains(&limit) { - return Err(Error::Command("KV list limit must be in 1..=1000")); - } - validate_now(now_ms)?; - - let mut sql = String::from( - "SELECT key, value, version, expires_at_ms FROM kv_entries WHERE scope = ?1 AND (expires_at_ms IS NULL OR expires_at_ms > ?2)", - ); - let mut parameters = vec![Value::Blob(scope.to_vec()), Value::Integer(now_ms)]; - if let Some(after_key) = after_key { - parameters.push(Value::Blob(after_key.to_vec())); - sql.push_str(&format!(" AND key > ?{}", parameters.len())); - } - if !prefix.is_empty() { - parameters.push(Value::Blob(prefix.to_vec())); - sql.push_str(&format!(" AND key >= ?{}", parameters.len())); - if let Some(end) = prefix_successor(prefix) { - parameters.push(Value::Blob(end)); - sql.push_str(&format!(" AND key < ?{}", parameters.len())); - } - } - parameters.push(Value::Integer( - i64::try_from(limit + 1).map_err(|_| Error::Command("KV list limit overflow"))?, - )); - sql.push_str(&format!(" ORDER BY key LIMIT ?{}", parameters.len())); - - let mut statement = connection.prepare(&sql)?; - let rows = statement.query_map(rusqlite::params_from_iter(parameters), decode_entry)?; - let mut entries = Vec::with_capacity(limit.min(16)); - let mut page_bytes = 0_usize; - let mut has_more = false; - for row in rows { - let entry = validate_entry(row?)?; - let entry_bytes = entry - .key - .len() - .checked_add(entry.value.len()) - .and_then(|bytes| bytes.checked_add(VERSION_BYTES + 64)) - .ok_or(Error::Command("KV page byte count overflow"))?; - if entries.len() == limit - || (!entries.is_empty() - && page_bytes - .checked_add(entry_bytes) - .is_none_or(|bytes| bytes > MAX_PAGE_BYTES)) - { - has_more = true; - break; - } - page_bytes += entry_bytes; - entries.push(entry); - } - let next_after = if has_more { - entries.last().map(|entry| entry.key.clone()) - } else { - None - }; - Ok(KvPage { - entries, - next_after, - }) -} - -/// Deletes at most 128 physically expired entries in one internal command. -pub fn kv_cleanup_expired(transaction: &Transaction<'_>, now_ms: i64) -> Result { - kv_cleanup_expired_bounded(transaction, now_ms, CLEANUP_ITEMS) -} - -pub(crate) fn kv_cleanup_expired_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - validate_now(now_ms)?; - if limit > CLEANUP_ITEMS { - return Err(Error::Command("KV cleanup limit exceeds 128")); - } - if limit == 0 { - return Ok(0); - } - let changed = transaction.execute( - "DELETE FROM kv_entries WHERE (scope, key) IN (SELECT scope, key FROM kv_entries INDEXED BY kv_expiry WHERE expires_at_ms IS NOT NULL AND expires_at_ms <= ?1 ORDER BY expires_at_ms, scope, key LIMIT ?2)", - (now_ms, limit as i64), - )?; - Ok(changed) -} - -fn validate_atomic(now_ms: i64, request: &KvAtomicRequest) -> Result<()> { - validate_now(now_ms)?; - validate_scope(&request.scope)?; - let item_count = request - .checks - .len() - .checked_add(request.mutations.len()) - .ok_or(Error::Command("KV atomic item count overflow"))?; - if item_count == 0 || item_count > MAX_ATOMIC_ITEMS { - return Err(Error::Command("KV atomic requires 1..=128 items")); - } - let mut operation_bytes = request.scope.len(); - for check in &request.checks { - validate_key(&check.key)?; - operation_bytes = operation_bytes - .checked_add(check.key.len() + VERSION_BYTES + 8) - .ok_or(Error::Command("KV atomic byte count overflow"))?; - } - let mut mutation_keys = HashSet::with_capacity(request.mutations.len()); - for mutation in &request.mutations { - let key = mutation.key(); - validate_key(key)?; - if !mutation_keys.insert(key) { - return Err(Error::Command("duplicate KV mutation key")); - } - operation_bytes = operation_bytes - .checked_add(key.len() + 16) - .ok_or(Error::Command("KV atomic byte count overflow"))?; - if let KvMutation::Put { - value, - expires_at_ms, - .. - } = mutation - { - if value.len() > MAX_VALUE_BYTES { - return Err(Error::Command("KV value exceeds 4 MiB")); - } - operation_bytes = operation_bytes - .checked_add(value.len()) - .ok_or(Error::Command("KV atomic byte count overflow"))?; - if expires_at_ms.is_some_and(|expiry| expiry <= now_ms) { - return Err(Error::Command("KV expiry must be after logical time")); - } - } - } - if operation_bytes > MAX_OPERATION_BYTES { - return Err(Error::Command("KV atomic operation exceeds byte budget")); - } - Ok(()) -} - -fn validate_scope(scope: &[u8]) -> Result<()> { - if scope.len() > MAX_SCOPE_BYTES { - return Err(Error::Command("KV scope exceeds 1024 bytes")); - } - Ok(()) -} - -fn validate_key(key: &[u8]) -> Result<()> { - if key.is_empty() || key.len() > MAX_KEY_BYTES { - return Err(Error::Command("KV key must contain 1..=1024 bytes")); - } - Ok(()) -} - -fn validate_now(now_ms: i64) -> Result<()> { - if now_ms < 0 { - return Err(Error::Command("negative KV logical time")); - } - Ok(()) -} - -fn live_version( - transaction: &Transaction<'_>, - scope: &[u8], - key: &[u8], - now_ms: i64, -) -> Result>> { - transaction - .query_row( - "SELECT version FROM kv_entries WHERE scope = ?1 AND key = ?2 AND (expires_at_ms IS NULL OR expires_at_ms > ?3)", - (scope, key, now_ms), - |row| row.get(0), - ) - .optional() - .map_err(Error::from) -} - -fn kv_version(incarnation: [u8; 16], sequence: i64, ordinal: u32) -> Result<[u8; VERSION_BYTES]> { - let sequence = u64::try_from(sequence).map_err(|_| Error::Command("invalid KV sequence"))?; - let mut version = [0; VERSION_BYTES]; - version[..16].copy_from_slice(&incarnation); - version[16..24].copy_from_slice(&sequence.to_be_bytes()); - version[24..].copy_from_slice(&ordinal.to_be_bytes()); - Ok(version) -} - -fn decode_entry(row: &rusqlite::Row<'_>) -> rusqlite::Result { - Ok((row.get(0)?, row.get(1)?, row.get(2)?, row.get(3)?)) -} - -fn validate_entry((key, value, version, expires_at_ms): RawEntry) -> Result { - validate_key(&key)?; - if value.len() > MAX_VALUE_BYTES { - return Err(Error::Command("stored KV value exceeds limit")); - } - let version = version - .try_into() - .map_err(|_| Error::Command("stored KV version has invalid length"))?; - Ok(KvEntry { - key, - value, - version, - expires_at_ms, - }) -} - -fn prefix_successor(prefix: &[u8]) -> Option> { - let index = prefix.iter().rposition(|byte| *byte != u8::MAX)?; - let mut end = prefix[..=index].to_vec(); - end[index] += 1; - Some(end) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn embedded_kv_migration_matches_normative_contract() { - assert_eq!(KV_SCHEMA, include_str!("../../docs/contracts/kv.sql")); - } - - #[test] - fn all_ff_prefix_has_no_exclusive_successor() { - assert_eq!(prefix_successor(&[]), None); - assert_eq!(prefix_successor(&[0xff, 0xff]), None); - assert_eq!(prefix_successor(&[0x12, 0xff]), Some(vec![0x13])); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/kv/api.rs b/crates/crab-cell-runtime/src/primitives/kv/api.rs deleted file mode 100644 index feae59952..000000000 --- a/crates/crab-cell-runtime/src/primitives/kv/api.rs +++ /dev/null @@ -1,514 +0,0 @@ -use std::marker::PhantomData; - -use crate::cell::catalog::CatalogRole; -use crate::client::{CellClient, Committed, InvocationError, Observed, Receipt}; -use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; -use crate::identity::{ - ApplicationId, CellTarget, NamespaceId, TenantId, partition_for_shard, shard_for_scope, -}; -use crate::registry::{Command, Query, RegistryBuilder}; -use crate::registry::{CommandContext, CommandResult, QueryContext}; - -use super::{ - KvAtomicOutcome, KvAtomicRequest, KvCheck, KvCondition, KvEntry, KvMutation, KvMutationResult, - KvPage, MAX_ATOMIC_ITEMS, MAX_LIST_ITEMS, VERSION_BYTES, kv_atomic, kv_get, kv_list, -}; - -const ABSENT_TAG: u8 = 0; -const VERSION_TAG: u8 = 1; -const PUT_TAG: u8 = 0; -const DELETE_TAG: u8 = 1; -const APPLIED_TAG: u8 = 0; -const PRECONDITION_FAILED_TAG: u8 = 1; - -/// Compile-time operation identifiers for one native KV module. -pub trait KvModule: Send + Sync + 'static { - /// Module the KV surfaces register under. - const MODULE: &'static str; - /// Codec version of the KV operations. - const CODEC_VERSION: u32 = 1; - /// Command id for atomic check-and-mutate. - const ATOMIC_COMMAND_ID: u32; - /// Query id for single-key reads. - const GET_QUERY_ID: u32; - /// Query id for scoped listing. - const LIST_QUERY_ID: u32; -} - -/// Registers the three typed KV bindings contributed by one compiled module. -pub fn register_kv(registry: &mut RegistryBuilder) -> crate::Result<()> { - registry.bind_command::>()?; - registry.bind_query::>()?; - registry.bind_query::>() -} - -/// Typed KV atomic command bound to its module's immutable operation IDs. -pub struct KvAtomicCommand(PhantomData M>); - -impl Command for KvAtomicCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::ATOMIC_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = KvAtomicRequest; - type Output = KvAtomicOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let outcome = kv_atomic(context.primitive_transaction(), context.now_ms(), &input)?; - Ok(match outcome { - KvAtomicOutcome::Applied(_) => CommandResult::Success(outcome), - KvAtomicOutcome::PreconditionFailed { .. } => CommandResult::Rejected(outcome), - }) - } -} - -/// One typed KV point-read input. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct KvGetRequest { - /// Scope the key lives in. - pub scope: Vec, - /// Key to read. - pub key: Vec, -} - -/// Typed KV point query bound to its module's immutable operation IDs. -pub struct KvGetQuery(PhantomData M>); - -impl Query for KvGetQuery { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::GET_QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = KvGetRequest; - type Output = Option; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - kv_get( - context.primitive_connection(), - &input.scope, - &input.key, - context.now_ms(), - ) - } -} - -/// One typed bounded KV list input. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct KvListRequest { - /// Scope to list. - pub scope: Vec, - /// Key prefix to match. - pub prefix: Vec, - /// Key to continue after, from a previous page. - pub after_key: Option>, - /// Maximum entries to return. - pub limit: u32, -} - -/// Typed KV list query bound to its module's immutable operation IDs. -pub struct KvListQuery(PhantomData M>); - -impl Query for KvListQuery { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::LIST_QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = KvListRequest; - type Output = KvPage; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - kv_list( - context.primitive_connection(), - &input.scope, - &input.prefix, - input.after_key.as_deref(), - usize::try_from(input.limit) - .map_err(|_| crate::Error::Command("KV list limit overflow"))?, - context.now_ms(), - ) - } -} - -/// Authorized native KV capability that derives stable shard Cells from scope. -pub struct KvNamespace { - client: CellClient, - tenant: TenantId, - application: ApplicationId, - namespace: NamespaceId, - shards: u32, - module: PhantomData M>, -} - -impl Clone for KvNamespace { - fn clone(&self) -> Self { - Self { - client: self.client.clone(), - tenant: self.tenant, - application: self.application, - namespace: self.namespace, - shards: self.shards, - module: PhantomData, - } - } -} - -impl KvNamespace { - /// Creates a scoped KV capability after validating the compiled namespace role. - pub fn new( - client: CellClient, - tenant: TenantId, - application: ApplicationId, - namespace: NamespaceId, - ) -> crate::Result { - let shards = client.require_namespace(namespace, M::MODULE, CatalogRole::Kv)?; - Ok(Self { - client, - tenant, - application, - namespace, - shards, - module: PhantomData, - }) - } - - /// Atomically checks and mutates one scope-derived shard. - pub async fn atomic( - &self, - identity: crate::cell::executor::MutationIdentity, - request: KvAtomicRequest, - ) -> std::result::Result, InvocationError> { - let target = self - .target(&request.scope) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>(&target, identity, request) - .await - } - - /// Reads one live key from its scope-derived shard. - pub async fn get( - &self, - scope: Vec, - key: Vec, - minimum: Option, - ) -> std::result::Result>, InvocationError>> { - let target = self.target(&scope).map_err(InvocationError::NotStarted)?; - self.client - .query::>(&target, minimum, KvGetRequest { scope, key }) - .await - } - - /// Lists one bounded current-read page from its scope-derived shard. - pub async fn list( - &self, - request: KvListRequest, - minimum: Option, - ) -> std::result::Result, InvocationError> { - let target = self - .target(&request.scope) - .map_err(InvocationError::NotStarted)?; - self.client - .query::>(&target, minimum, request) - .await - } - - fn target(&self, scope: &[u8]) -> crate::Result { - let shard = shard_for_scope(self.namespace, scope, self.shards)?; - CellTarget::new( - self.tenant, - self.application, - self.namespace, - &partition_for_shard(shard), - ) - } -} - -impl WireValue for KvAtomicRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.scope)?; - encoder.write_count(self.checks.len())?; - for check in &self.checks { - check.encode(encoder)?; - } - encoder.write_count(self.mutations.len())?; - for mutation in &self.mutations { - mutation.encode(encoder)?; - } - Ok(()) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let scope = decoder.read_bytes()?.to_vec(); - let checks = decode_bounded(decoder, MAX_ATOMIC_ITEMS, KvCheck::decode)?; - let remaining = MAX_ATOMIC_ITEMS - .checked_sub(checks.len()) - .ok_or(CodecError::Invalid("too many KV checks"))?; - let mutations = decode_bounded(decoder, remaining, KvMutation::decode)?; - Ok(Self { - scope, - checks, - mutations, - }) - } -} - -impl WireValue for KvCheck { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.key)?; - match self.condition { - KvCondition::Absent => encoder.write_u8(ABSENT_TAG), - KvCondition::Version(version) => { - encoder.write_u8(VERSION_TAG)?; - encoder.write_bytes(&version) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let key = decoder.read_bytes()?.to_vec(); - let condition = match decoder.read_u8()? { - ABSENT_TAG => KvCondition::Absent, - VERSION_TAG => KvCondition::Version(read_version(decoder)?), - _ => return Err(CodecError::Invalid("invalid KV condition tag")), - }; - Ok(Self { key, condition }) - } -} - -impl WireValue for KvMutation { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Put { - key, - value, - expires_at_ms, - } => { - encoder.write_u8(PUT_TAG)?; - encoder.write_bytes(key)?; - encoder.write_bytes(value)?; - expires_at_ms.encode(encoder) - } - Self::Delete { key } => { - encoder.write_u8(DELETE_TAG)?; - encoder.write_bytes(key) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - PUT_TAG => Ok(Self::Put { - key: decoder.read_bytes()?.to_vec(), - value: decoder.read_bytes()?.to_vec(), - expires_at_ms: Option::::decode(decoder)?, - }), - DELETE_TAG => Ok(Self::Delete { - key: decoder.read_bytes()?.to_vec(), - }), - _ => Err(CodecError::Invalid("invalid KV mutation tag")), - } - } -} - -impl WireValue for KvAtomicOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Applied(results) => { - encoder.write_u8(APPLIED_TAG)?; - encoder.write_count(results.len())?; - for result in results { - result.encode(encoder)?; - } - Ok(()) - } - Self::PreconditionFailed { key } => { - encoder.write_u8(PRECONDITION_FAILED_TAG)?; - encoder.write_bytes(key) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - APPLIED_TAG => Ok(Self::Applied(decode_bounded( - decoder, - MAX_ATOMIC_ITEMS, - KvMutationResult::decode, - )?)), - PRECONDITION_FAILED_TAG => Ok(Self::PreconditionFailed { - key: decoder.read_bytes()?.to_vec(), - }), - _ => Err(CodecError::Invalid("invalid KV atomic outcome tag")), - } - } -} - -impl WireValue for KvMutationResult { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.key)?; - match self.version { - None => encoder.write_u8(0)?, - Some(version) => { - encoder.write_u8(1)?; - encoder.write_bytes(&version)?; - } - } - encoder.write_bool(self.deleted) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let key = decoder.read_bytes()?.to_vec(); - let version = match decoder.read_u8()? { - 0 => None, - 1 => Some(read_version(decoder)?), - _ => return Err(CodecError::Invalid("invalid KV result version tag")), - }; - let deleted = decoder.read_bool()?; - if deleted == version.is_some() { - return Err(CodecError::Invalid("inconsistent KV mutation result")); - } - Ok(Self { - key, - version, - deleted, - }) - } -} - -impl WireValue for KvGetRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.scope)?; - encoder.write_bytes(&self.key) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - scope: decoder.read_bytes()?.to_vec(), - key: decoder.read_bytes()?.to_vec(), - }) - } -} - -impl WireValue for KvListRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.scope)?; - encoder.write_bytes(&self.prefix)?; - self.after_key.encode(encoder)?; - encoder.write_u32(self.limit) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - scope: decoder.read_bytes()?.to_vec(), - prefix: decoder.read_bytes()?.to_vec(), - after_key: Option::>::decode(decoder)?, - limit: decoder.read_u32()?, - }) - } -} - -impl WireValue for KvEntry { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.key)?; - encoder.write_bytes(&self.value)?; - encoder.write_bytes(&self.version)?; - self.expires_at_ms.encode(encoder) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - key: decoder.read_bytes()?.to_vec(), - value: decoder.read_bytes()?.to_vec(), - version: read_version(decoder)?, - expires_at_ms: Option::::decode(decoder)?, - }) - } -} - -impl WireValue for KvPage { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_count(self.entries.len())?; - for entry in &self.entries { - entry.encode(encoder)?; - } - self.next_after.encode(encoder) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - entries: decode_bounded(decoder, MAX_LIST_ITEMS, KvEntry::decode)?, - next_after: Option::>::decode(decoder)?, - }) - } -} - -fn decode_bounded( - decoder: &mut BoundedDecoder<'_>, - maximum: usize, - mut decode: impl FnMut(&mut BoundedDecoder<'_>) -> Result, -) -> Result, CodecError> { - let count = decoder.read_count()?; - if count > maximum { - return Err(CodecError::Invalid("KV collection count exceeds limit")); - } - let mut values = Vec::with_capacity(count); - for _ in 0..count { - values.push(decode(decoder)?); - } - Ok(values) -} - -fn read_version(decoder: &mut BoundedDecoder<'_>) -> Result<[u8; VERSION_BYTES], CodecError> { - decoder - .read_bytes()? - .try_into() - .map_err(|_| CodecError::Invalid("invalid KV version length")) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::codec::roundtrip; - - #[test] - fn kv_command_and_page_codecs_roundtrip_exactly() { - roundtrip(KvAtomicRequest { - scope: b"scope".to_vec(), - checks: vec![KvCheck { - key: b"old".to_vec(), - condition: KvCondition::Version([7; VERSION_BYTES]), - }], - mutations: vec![ - KvMutation::Put { - key: b"new".to_vec(), - value: b"value".to_vec(), - expires_at_ms: Some(99), - }, - KvMutation::Delete { - key: b"old".to_vec(), - }, - ], - }); - roundtrip(KvPage { - entries: vec![KvEntry { - key: b"key".to_vec(), - value: b"value".to_vec(), - version: [8; VERSION_BYTES], - expires_at_ms: None, - }], - next_after: Some(b"key".to_vec()), - }); - } - - #[test] - fn kv_decoder_rejects_oversized_collections_before_allocation() { - let mut bytes = Vec::new(); - bytes.extend_from_slice(&0_u32.to_be_bytes()); - bytes.extend_from_slice(&129_u32.to_be_bytes()); - let mut decoder = BoundedDecoder::new(&bytes, bytes.len() as u32).unwrap(); - assert!(matches!( - KvAtomicRequest::decode(&mut decoder), - Err(CodecError::Invalid("KV collection count exceeds limit")) - )); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/maintenance.rs b/crates/crab-cell-runtime/src/primitives/maintenance.rs deleted file mode 100644 index 4e77c250d..000000000 --- a/crates/crab-cell-runtime/src/primitives/maintenance.rs +++ /dev/null @@ -1,623 +0,0 @@ -//! Persisted and transferable work inventory used by the scheduler. -use std::marker::PhantomData; - -use crab_ltx::rusqlite::Connection; - -use crate::Error; -use crate::cell::catalog::CatalogRole; -use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; -use crate::fleet::scheduler::SchedulerTickOutcome; -use crate::fleet::scheduler::scheduler_tick_at; -use crate::primitives::cron::CronTarget; -use crate::primitives::queue::QueueDeadLetterTarget; -use crate::primitives::workflow::WorkflowDefinition; -use crate::registry::{Command, RegistryBuilder}; -use crate::registry::{CommandContext, CommandResult}; - -const REQUESTS: u8 = 1 << 0; -const INBOX: u8 = 1 << 1; -const EFFECTS: u8 = 1 << 2; -const QUEUE_MESSAGES: u8 = 1 << 3; -const QUEUE_DEDUP: u8 = 1 << 4; -const WORKFLOWS: u8 = 1 << 5; -const BLOBS: u8 = 1 << 6; -const CRON_SCHEDULES: u8 = 1 << 7; - -const TRANSFER_EFFECTS: u8 = 1 << 0; -const TRANSFER_QUEUE: u8 = 1 << 1; -const TRANSFER_WORKFLOW: u8 = 1 << 2; -const TRANSFER_CRON: u8 = 1 << 3; - -/// Conservative inventory of rows that can retain executable release contracts. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct PersistedWorkInventory { - bits: u8, - unknown: bool, -} - -impl PersistedWorkInventory { - pub(crate) const fn unknown() -> Self { - Self { - bits: 0, - unknown: true, - } - } - - /// Reports whether contract removal can proceed without transforming durable work. - #[must_use] - pub const fn is_empty(self) -> bool { - !self.unknown && self.bits == 0 - } - - /// Reports whether durable rows need no source-side execution after an - /// exact-root handoff. Retained outcomes, Blob metadata, and producer - /// identities are restored with the root; live primitive rows still block. - pub(crate) const fn is_transfer_settled(self) -> bool { - !self.unknown && self.bits & (EFFECTS | QUEUE_MESSAGES | WORKFLOWS | CRON_SCHEDULES) == 0 - } - - pub(crate) const fn is_unknown(self) -> bool { - self.unknown - } - - /// Names the first durable work class blocking contract removal. - #[must_use] - pub const fn first_blocker(self) -> Option<&'static str> { - if self.unknown { - Some("maintenance release is blocked by unknown persisted work") - } else if self.bits & REQUESTS != 0 { - Some("maintenance release is blocked by retained request outcomes") - } else if self.bits & INBOX != 0 { - Some("maintenance release is blocked by retained effect inbox outcomes") - } else if self.bits & EFFECTS != 0 { - Some("maintenance release is blocked by retained source effects") - } else if self.bits & QUEUE_MESSAGES != 0 { - Some("maintenance release is blocked by retained Queue messages") - } else if self.bits & QUEUE_DEDUP != 0 { - Some("maintenance release is blocked by retained Queue producer identities") - } else if self.bits & WORKFLOWS != 0 { - Some("maintenance release is blocked by retained Workflow runs") - } else if self.bits & BLOBS != 0 { - Some("maintenance release is blocked by retained Blob objects") - } else if self.bits & CRON_SCHEDULES != 0 { - Some("maintenance release is blocked by retained Cron schedules") - } else { - None - } - } - - pub(crate) fn encode(self) -> Vec { - if self.unknown { - Vec::new() - } else { - vec![self.bits] - } - } - - pub(crate) fn decode(bytes: &[u8]) -> crate::Result { - match bytes { - [bits] => Ok(Self { - bits: *bits, - unknown: false, - }), - _ => Err(Error::Command("invalid persisted-work inventory")), - } - } -} - -/// Transfer-only view of durable work that still needs the current owner. -/// -/// This deliberately has a separate contract from [`PersistedWorkInventory`]: -/// retained outcomes and immutable state can travel with an exact root, while -/// executable work must remain on its current owner until it settles. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub(crate) struct TransferWorkInventory { - bits: u8, -} - -impl TransferWorkInventory { - pub(crate) const fn is_settled(self) -> bool { - self.bits == 0 - } - - #[cfg(test)] - const fn bits(self) -> u8 { - self.bits - } -} - -pub(crate) fn inspect_persisted_work( - connection: &Connection, - role: CatalogRole, -) -> crate::Result { - let mut bits = 0; - bits |= exists(connection, "SELECT EXISTS(SELECT 1 FROM sys_requests)")? * REQUESTS; - bits |= exists(connection, "SELECT EXISTS(SELECT 1 FROM sys_inbox)")? * INBOX; - bits |= exists(connection, "SELECT EXISTS(SELECT 1 FROM sys_effects)")? * EFFECTS; - if role == CatalogRole::Queue { - bits |= exists(connection, "SELECT EXISTS(SELECT 1 FROM queue_messages)")? * QUEUE_MESSAGES; - bits |= exists(connection, "SELECT EXISTS(SELECT 1 FROM queue_dedup)")? * QUEUE_DEDUP; - } - if role == CatalogRole::Workflow { - bits |= exists(connection, "SELECT EXISTS(SELECT 1 FROM workflow_runs)")? * WORKFLOWS; - } - if role == CatalogRole::Blob { - bits |= exists(connection, "SELECT EXISTS(SELECT 1 FROM blob_objects)")? * BLOBS; - } - if role == CatalogRole::Cron { - bits |= exists(connection, "SELECT EXISTS(SELECT 1 FROM cron_schedules)")? * CRON_SCHEDULES; - } - Ok(PersistedWorkInventory { - bits, - unknown: false, - }) -} - -/// Inspects only executable rows that cannot safely move with an exact root. -/// -/// Every probe is an indexed `EXISTS` query and therefore has bounded result -/// memory. Any SQL/schema error is returned to the actor, which refuses the -/// transfer rather than treating an unknown inspection as settled. -pub(crate) fn inspect_transfer_work( - connection: &Connection, - role: CatalogRole, - now_ms: i64, -) -> crate::Result { - if now_ms < 0 { - return Err(Error::Command("invalid transfer inspection time")); - } - let effects = connection.query_row( - "SELECT EXISTS(SELECT 1 FROM sys_effects INDEXED BY sys_effects_due WHERE state = 1 OR (state = 0 AND due_at_ms <= ?1) LIMIT 1)", - [now_ms], - |row| row.get::<_, i64>(0), - )?; - let mut bits = match effects { - 0 => 0, - 1 => TRANSFER_EFFECTS, - _ => return Err(Error::Command("invalid transfer effect existence result")), - }; - if role == CatalogRole::Queue { - bits |= exists( - connection, - "SELECT EXISTS(SELECT 1 FROM queue_messages INDEXED BY queue_ready WHERE state IN (0, 1) LIMIT 1)", - )? * TRANSFER_QUEUE; - } - if role == CatalogRole::Workflow { - // A running workflow can be waiting on an external event without a - // materialized activity or timer. Its durable state is still an - // executable contract for the current owner, so the exact-root - // handoff must remain deferred until the run is terminal or paused. - bits |= exists( - connection, - "SELECT EXISTS(SELECT 1 FROM workflow_runs WHERE status = 0 LIMIT 1)", - )? * TRANSFER_WORKFLOW; - bits |= exists( - connection, - "SELECT EXISTS(SELECT 1 FROM workflow_activities INDEXED BY activities_due WHERE state IN (0, 1) LIMIT 1)", - )? * TRANSFER_WORKFLOW; - bits |= exists( - connection, - "SELECT EXISTS(SELECT 1 FROM workflow_timers INDEXED BY timers_due WHERE state = 0 LIMIT 1)", - )? * TRANSFER_WORKFLOW; - } - if role == CatalogRole::Cron { - let due = connection.query_row( - "SELECT EXISTS(SELECT 1 FROM cron_schedules INDEXED BY cron_due WHERE enabled = 1 AND next_due_ms <= ?1 LIMIT 1)", - [now_ms], - |row| row.get::<_, i64>(0), - )?; - bits |= match due { - 0 => 0, - 1 => TRANSFER_CRON, - _ => return Err(Error::Command("invalid transfer cron existence result")), - }; - } - Ok(TransferWorkInventory { bits }) -} - -fn exists(connection: &Connection, sql: &str) -> crate::Result { - let exists = connection.query_row(sql, [], |row| row.get::<_, i64>(0))?; - match exists { - 0 => Ok(0), - 1 => Ok(1), - _ => Err(Error::Command("invalid persisted-work existence result")), - } -} - -/// Compile-time binding for the internal maintenance command of one module. -pub trait MaintenanceModule: Send + Sync + 'static { - /// Module the internal scheduler Tick registers under. - const MODULE: &'static str; - /// Codec version of the Tick command. - const CODEC_VERSION: u32 = 1; - /// Command id of the module's internal Tick. - const TICK_COMMAND_ID: u32; - /// Workflow definitions this module's Tick may advance. - const WORKFLOW_DEFINITIONS: &'static [&'static dyn WorkflowDefinition] = &[]; - /// Queue dead-letter target this module's Tick may redrive. - const QUEUE_DEAD_LETTER: Option = None; - /// Cron targets this module's Tick may fire. - const CRON_TARGETS: &'static [CronTarget] = &[]; -} - -/// Registers one module's internal scheduler Tick command. -pub fn register_maintenance( - registry: &mut RegistryBuilder, -) -> crate::Result<()> { - registry.bind_maintenance_module(M::MODULE, M::QUEUE_DEAD_LETTER)?; - registry.bind_maintenance_runner::()?; - registry.bind_command::>() -} - -/// Published root position from which a due-Cell scan scheduled this Tick. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct MaintenanceTickRequest { - /// Commit sequence the scan expected the Cell to be at. - pub expected_commit_sequence: u64, -} - -/// Durable result of one scheduled Tick attempt. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum MaintenanceTickOutcome { - /// The Tick advanced due maintenance items. - Applied { - /// Items the Tick processed. - processed: u32, - }, - /// The Cell moved past the expected sequence, so the Tick did nothing. - Stale, -} - -/// Typed system command that advances bounded maintenance through the actor. -pub struct MaintenanceTickCommand(PhantomData M>); - -impl Command for MaintenanceTickCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::TICK_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = MaintenanceTickRequest; - type Output = MaintenanceTickOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let expected_next = input - .expected_commit_sequence - .checked_add(1) - .ok_or(Error::Command("maintenance expected sequence overflow"))?; - if expected_next != context.sequence() { - return Ok(CommandResult::Success(MaintenanceTickOutcome::Stale)); - } - let SchedulerTickOutcome { processed } = scheduler_tick_at( - context.primitive_transaction(), - context.target(), - context.sequence(), - context.now_ms(), - M::WORKFLOW_DEFINITIONS, - M::QUEUE_DEAD_LETTER, - M::CRON_TARGETS, - )?; - Ok(CommandResult::Success(MaintenanceTickOutcome::Applied { - processed, - })) - } -} - -impl WireValue for MaintenanceTickRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_u64(self.expected_commit_sequence) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - expected_commit_sequence: decoder.read_u64()?, - }) - } -} - -impl WireValue for MaintenanceTickOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Applied { processed } if *processed <= 128 => { - encoder.write_u8(0)?; - encoder.write_u32(*processed) - } - Self::Applied { .. } => { - Err(CodecError::Invalid("maintenance result exceeds 128 items")) - } - Self::Stale => encoder.write_u8(1), - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => { - let processed = decoder.read_u32()?; - if processed > 128 { - return Err(CodecError::Invalid("maintenance result exceeds 128 items")); - } - Ok(Self::Applied { processed }) - } - 1 => Ok(Self::Stale), - _ => Err(CodecError::Invalid("invalid maintenance result tag")), - } - } -} - -#[cfg(test)] -mod tests { - use crab_ltx::rusqlite::Connection; - - use super::*; - use crate::cell::schema::install_runtime_schema; - use crate::identity::CellId; - use crate::identity::IncarnationId; - use crate::primitives::cron::install_cron_schema; - use crate::primitives::queue::install_queue_schema; - use crate::primitives::workflow::install_workflow_schema; - - #[test] - fn maintenance_codecs_reject_unbounded_results() { - let mut encoder = BoundedEncoder::new(16).unwrap(); - assert!( - MaintenanceTickOutcome::Applied { processed: 129 } - .encode(&mut encoder) - .is_err() - ); - } - - #[test] - fn retained_runtime_outcome_blocks_contract_removal() { - let mut connection = Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - CellId::from_bytes([1; 32]), - IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - assert!( - inspect_persisted_work(&connection, CatalogRole::Repository) - .unwrap() - .is_empty() - ); - connection - .execute( - "INSERT INTO sys_requests VALUES (?1, ?2, 1, X'', 1, 1, 1)", - ([3_u8; 16].as_slice(), [4_u8; 32].as_slice()), - ) - .unwrap(); - - assert_eq!( - inspect_persisted_work(&connection, CatalogRole::Repository) - .unwrap() - .first_blocker(), - Some("maintenance release is blocked by retained request outcomes") - ); - } - - #[test] - fn unknown_inventory_stays_fail_closed_without_changing_wire_width() { - let unknown = PersistedWorkInventory::unknown(); - assert!(unknown.is_unknown()); - assert!(!unknown.is_empty()); - assert_eq!( - unknown.first_blocker(), - Some("maintenance release is blocked by unknown persisted work") - ); - assert!(PersistedWorkInventory::decode(&unknown.encode()).is_err()); - - let empty = PersistedWorkInventory::decode(&[0]).unwrap(); - assert!(empty.is_empty()); - assert!(!empty.is_unknown()); - } - - #[test] - fn queue_and_workflow_roles_inventory_primitive_rows() { - let mut queue = Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut queue, - CellId::from_bytes([5; 32]), - IncarnationId::from_bytes([6; 16]), - 1, - ) - .unwrap(); - let transaction = queue.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - queue - .execute( - "INSERT INTO queue_dedup VALUES (?1, ?2, ?3, 1)", - ( - [7_u8; 16].as_slice(), - [8_u8; 32].as_slice(), - [9_u8; 16].as_slice(), - ), - ) - .unwrap(); - assert_eq!( - inspect_persisted_work(&queue, CatalogRole::Queue) - .unwrap() - .first_blocker(), - Some("maintenance release is blocked by retained Queue producer identities") - ); - - let mut workflow = Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut workflow, - CellId::from_bytes([10; 32]), - IncarnationId::from_bytes([11; 16]), - 1, - ) - .unwrap(); - let transaction = workflow.transaction().unwrap(); - install_workflow_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - workflow - .execute( - "INSERT INTO workflow_runs VALUES (X'01', ?1, ?2, 0, X'', 0, NULL, NULL)", - ([12_u8; 16].as_slice(), [13_u8; 32].as_slice()), - ) - .unwrap(); - assert_eq!( - inspect_persisted_work(&workflow, CatalogRole::Workflow) - .unwrap() - .first_blocker(), - Some("maintenance release is blocked by retained Workflow runs") - ); - } - - #[test] - fn transfer_inventory_allows_settled_rows_and_future_cron() { - let mut queue = Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut queue, - CellId::from_bytes([14; 32]), - IncarnationId::from_bytes([15; 16]), - 1, - ) - .unwrap(); - let transaction = queue.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - install_cron_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - queue - .execute( - "INSERT INTO queue_messages(message_id, payload, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, result_code) VALUES (?1, X'', 2, 0, 0, 100, NULL, NULL, 0)", - [[16_u8; 16].as_slice()], - ) - .unwrap(); - queue - .execute( - "INSERT INTO queue_dedup VALUES (?1, ?2, ?3, 100)", - ( - [17_u8; 16].as_slice(), - [18_u8; 32].as_slice(), - [16_u8; 16].as_slice(), - ), - ) - .unwrap(); - queue - .execute( - "INSERT INTO cron_schedules VALUES (?1, 0, X'', X'', 1000, 2000, 0, 1, 1, 0)", - [[19_u8; 16].as_slice()], - ) - .unwrap(); - - assert!( - inspect_transfer_work(&queue, CatalogRole::Queue, 1000) - .unwrap() - .is_settled() - ); - assert!( - inspect_transfer_work(&queue, CatalogRole::Cron, 1000) - .unwrap() - .is_settled() - ); - } - - #[test] - fn transfer_inventory_blocks_effects_and_queue_leases() { - let mut connection = Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - CellId::from_bytes([20; 32]), - IncarnationId::from_bytes([21; 16]), - 1, - ) - .unwrap(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - connection - .execute( - "INSERT INTO sys_effects(effect_id, destination, operation, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, created_sequence, result) VALUES (?1, ?2, X'01', 0, 0, 50, 100, NULL, NULL, 1, NULL)", - ([22_u8; 32].as_slice(), [23_u8; 32].as_slice()), - ) - .unwrap(); - assert!( - inspect_transfer_work(&connection, CatalogRole::Repository, 10) - .unwrap() - .is_settled() - ); - connection - .execute( - "UPDATE sys_effects SET due_at_ms = 0 WHERE effect_id = ?1", - [[22_u8; 32].as_slice()], - ) - .unwrap(); - connection - .execute( - "INSERT INTO queue_messages(message_id, payload, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, result_code) VALUES (?1, X'', 1, 1, 0, 100, ?2, 50, NULL)", - ([24_u8; 16].as_slice(), [25_u8; 16].as_slice()), - ) - .unwrap(); - - let inventory = inspect_transfer_work(&connection, CatalogRole::Queue, 10).unwrap(); - assert_eq!(inventory.bits(), TRANSFER_EFFECTS | TRANSFER_QUEUE); - assert!(!inventory.is_settled()); - } - - #[test] - fn transfer_inventory_blocks_pending_workflow_and_due_cron() { - let mut workflow = Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut workflow, - CellId::from_bytes([26; 32]), - IncarnationId::from_bytes([27; 16]), - 1, - ) - .unwrap(); - let transaction = workflow.transaction().unwrap(); - install_workflow_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - workflow - .execute( - "INSERT INTO workflow_runs VALUES (X'01', ?1, ?2, 0, X'', 0, NULL, NULL)", - ([28_u8; 16].as_slice(), [29_u8; 32].as_slice()), - ) - .unwrap(); - assert_eq!( - inspect_transfer_work(&workflow, CatalogRole::Workflow, 10) - .unwrap() - .bits(), - TRANSFER_WORKFLOW - ); - workflow - .execute( - "INSERT INTO workflow_activities(run_id, activity_id, activity_type, input, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, completion_token, completion_digest, result) VALUES (?1, ?2, 'test', X'', 0, 0, 0, 100, NULL, NULL, NULL, NULL, NULL)", - ([28_u8; 16].as_slice(), [30_u8; 16].as_slice()), - ) - .unwrap(); - assert_eq!( - inspect_transfer_work(&workflow, CatalogRole::Workflow, 10) - .unwrap() - .bits(), - TRANSFER_WORKFLOW - ); - - let mut cron = Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut cron, - CellId::from_bytes([31; 32]), - IncarnationId::from_bytes([32; 16]), - 1, - ) - .unwrap(); - let transaction = cron.transaction().unwrap(); - install_cron_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - cron.execute( - "INSERT INTO cron_schedules VALUES (?1, 0, X'', X'', 1000, 10, 0, 1, 1, 0)", - [[33_u8; 16].as_slice()], - ) - .unwrap(); - assert_eq!( - inspect_transfer_work(&cron, CatalogRole::Cron, 10) - .unwrap() - .bits(), - TRANSFER_CRON - ); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/queue.rs b/crates/crab-cell-runtime/src/primitives/queue.rs deleted file mode 100644 index 1f0644753..000000000 --- a/crates/crab-cell-runtime/src/primitives/queue.rs +++ /dev/null @@ -1,959 +0,0 @@ -//! Queue primitive: lease-based messages, retries, and dead letters. -use rand::RngCore; -use rusqlite::{Connection, OptionalExtension, Transaction}; - -use crate::codec::{BoundedEncoder, WireValue}; -use crate::identity::IncarnationId; -use crate::identity::{CellId, CellTarget, NamespaceId, partition_for_shard, shard_for_scope}; -use crate::primitives::effects::EffectBatch; -use crate::primitives::effects::EffectCommandIntent; -use crate::{Error, Result}; - -mod api; - -pub use api::{ - QueueClaimCommand, QueueClaimRequest, QueueControlCommand, QueueDeadLetterTarget, - QueueInfoQuery, QueueInfoRequest, QueueLeaseCommand, QueueLeaseRequest, QueueModule, - QueueNamespace, QueueSendCommand, QueueValidateClaimQuery, QueueValidateRequest, - register_queue, -}; - -const QUEUE_SCHEMA: &str = include_str!("../migrations/queue.sql"); -pub(crate) const QUEUE_TABLE: &str = "queue_messages"; -const MAX_PAYLOAD_BYTES: usize = 256 * 1024; -pub(crate) const QUEUE_SEND_MAX_INPUT_BYTES: u32 = MAX_PAYLOAD_BYTES as u32 + 32; -const MAX_CLAIM_BYTES: usize = 512 * 1024; -const MAX_CLAIM_ITEMS: usize = 32; -pub(crate) const MAX_ATTEMPTS: u32 = 20; -const MAX_RECLAIM_ITEMS: usize = 128; -const MIN_LEASE_MS: u32 = 5_000; -const MAX_LEASE_MS: u32 = 300_000; -const MAX_RETRY_DELAY_MS: u32 = 3_600_000; -const MAX_AVAILABLE_DELAY_MS: i64 = 7 * 24 * 60 * 60 * 1_000; -const RETENTION_MS: i64 = 30 * 24 * 60 * 60 * 1_000; -const DELIVERY_MARGIN_MS: i64 = 1_000; -const DEAD_LETTER_EFFECT_LIFETIME_MS: i64 = 7 * 24 * 60 * 60 * 1_000; - -pub(crate) struct QueueDeadLetterWriter<'a> { - target: QueueDeadLetterTarget, - effects: &'a mut EffectBatch, -} - -impl<'a> QueueDeadLetterWriter<'a> { - pub(crate) const fn new(target: QueueDeadLetterTarget, effects: &'a mut EffectBatch) -> Self { - Self { target, effects } - } - - fn insert( - &mut self, - transaction: &Transaction<'_>, - message_id: [u8; 16], - payload: &[u8], - now_ms: i64, - ) -> Result<[u8; 32]> { - let source = self.effects.source_target(); - let producer_id = dead_letter_producer_id( - self.target.namespace(), - source.cell_id(), - self.effects.source_incarnation(), - message_id, - ); - let shard = shard_for_scope(self.target.namespace(), &producer_id, self.target.shards())?; - let target = CellTarget::new( - source.tenant(), - source.application(), - self.target.namespace(), - &partition_for_shard(shard), - )?; - let request = QueueSendRequest { - producer_id, - payload: payload.to_vec(), - available_at_ms: now_ms, - }; - let mut encoder = BoundedEncoder::new(QUEUE_SEND_MAX_INPUT_BYTES)?; - request.encode(&mut encoder)?; - let expires_at_ms = now_ms - .checked_add(DEAD_LETTER_EFFECT_LIFETIME_MS) - .ok_or(Error::Command("queue dead-letter effect expiry overflow"))?; - self.effects.insert_command( - transaction, - &EffectCommandIntent { - target, - command_id: self.target.send_command_id(), - codec_version: self.target.codec_version(), - input: encoder.finish(), - expires_at_ms, - }, - ) - } -} - -/// Source of unpredictable queue lease tokens. -pub trait QueueTokenSource { - /// Returns an unpredictable lease token. - fn next_token(&mut self) -> Result<[u8; 16]>; -} - -/// Cryptographically seeded process-local queue token source. -pub struct SystemQueueTokens; - -impl QueueTokenSource for SystemQueueTokens { - fn next_token(&mut self) -> Result<[u8; 16]> { - let mut token = [0; 16]; - rand::rng().fill_bytes(&mut token); - Ok(token) - } -} - -/// Durable queue lifecycle state. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum QueueState { - /// The message is available for its next delivery. - Ready, - /// Delivery is leased to one consumer token. - Leased, - /// A consumer acknowledged the message. - Acked, - /// The message exhausted its attempts and waits for redrive or purge. - Dead, -} - -impl QueueState { - fn encode(self) -> i64 { - match self { - Self::Ready => 0, - Self::Leased => 1, - Self::Acked => 2, - Self::Dead => 3, - } - } - - fn decode(value: i64) -> Result { - match value { - 0 => Ok(Self::Ready), - 1 => Ok(Self::Leased), - 2 => Ok(Self::Acked), - 3 => Ok(Self::Dead), - _ => Err(Error::Command("invalid stored queue state")), - } - } -} - -/// Identity and scheduling input for an idempotent queue send. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct QueueSendRequest { - /// Producer identity that deduplicates the send. - pub producer_id: [u8; 16], - /// Message bytes. - pub payload: Vec, - /// Logical time the message becomes visible. - pub available_at_ms: i64, -} - -/// Result of producer-level queue deduplication. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum QueueSendOutcome { - /// The message was stored under this stable identity. - Sent { - /// Message identity, reused for a repeated producer send. - message_id: [u8; 16], - }, - /// This producer already sent a message with different bytes. - ProducerConflict, -} - -/// A published queue delivery lease. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct QueueMessage { - /// Stable message identity. - pub message_id: [u8; 16], - /// Message bytes. - pub payload: Vec, - /// Lease token required to acknowledge, retry, or extend. - pub token: [u8; 16], - /// Delivery attempt this lease represents. - pub attempt: u32, - /// Logical time the lease expires. - pub lease_until_ms: i64, -} - -/// Conditional mutation of one current queue lease. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum QueueLeaseAction { - /// Acknowledges the message and finishes it. - Ack, - /// Returns the message for another delivery. - Retry { - /// Delay before the message is visible again. - delay_ms: u32, - }, - /// Extends the current lease. - Extend { - /// Additional lease time to grant. - extension_ms: u32, - }, -} - -/// Business outcome of a queue lease mutation. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum QueueLeaseOutcome { - /// The mutation was applied to the current lease. - Applied { - /// Message state after the mutation. - state: QueueState, - /// New lease expiry, when the message is still leased. - lease_until_ms: Option, - }, - /// The token no longer matches the current lease. - LeaseLost, -} - -/// Administrative mutation applied to one queue shard. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum QueueControlAction { - /// Stops the shard from leasing messages. - Pause, - /// Allows leasing again and starts a new generation. - Resume, - /// Removes up to `limit` messages from the shard. - Purge { - /// Maximum messages to remove. - limit: u32, - }, - /// Returns up to `limit` dead messages to the ready state. - Redrive { - /// Maximum dead messages to return. - limit: u32, - }, -} - -/// Result of one queue control mutation. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum QueueControlOutcome { - /// The shard is paused at this generation. - Paused { - /// Pause/resume generation the shard counted. - generation: u64, - }, - /// The shard resumed with a new generation. - Resumed { - /// Pause/resume generation the shard counted. - generation: u64, - }, - /// The purge removed messages. - Purged { - /// Number of messages removed. - messages: u32, - }, - /// The redrive returned dead messages to ready. - Redriven { - /// Number of messages returned. - messages: u32, - }, -} - -/// Bounded queue shard state returned to operators. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct QueueInfo { - /// Whether the shard currently leases messages. - pub paused: bool, - /// Pause/resume generation the shard counted. - pub generation: u64, - /// Messages available for delivery. - pub ready: u64, - /// Messages currently leased to a consumer. - pub leased: u64, - /// Messages acknowledged. - pub acked: u64, - /// Messages that exhausted their attempts. - pub dead: u64, -} - -/// Installs the exact version-one Queue schema inside bootstrap or migration SQL. -pub fn install_queue_schema(transaction: &Transaction<'_>) -> Result<()> { - transaction.execute_batch(QUEUE_SCHEMA)?; - Ok(()) -} - -/// Inserts or deduplicates one producer send inside a runtime command. -pub fn queue_send( - transaction: &Transaction<'_>, - namespace: NamespaceId, - now_ms: i64, - request: &QueueSendRequest, -) -> Result { - validate_now(now_ms)?; - if request.payload.len() > MAX_PAYLOAD_BYTES { - return Err(Error::Command("queue payload exceeds 256 KiB")); - } - let latest = now_ms - .checked_add(MAX_AVAILABLE_DELAY_MS) - .ok_or(Error::Command("queue available time overflow"))?; - if request.available_at_ms < 0 || request.available_at_ms > latest { - return Err(Error::Command( - "queue available time outside seven-day window", - )); - } - let due_at_ms = request.available_at_ms.max(now_ms); - let digest = send_digest(&request.payload, request.available_at_ms); - let existing = transaction - .query_row( - "SELECT payload_digest, message_id FROM queue_dedup WHERE producer_id = ?1", - [request.producer_id.as_slice()], - |row| Ok((row.get::<_, Vec>(0)?, row.get::<_, Vec>(1)?)), - ) - .optional()?; - if let Some((stored_digest, stored_id)) = existing { - if stored_digest.as_slice() != digest { - return Ok(QueueSendOutcome::ProducerConflict); - } - let message_id = stored_id - .try_into() - .map_err(|_| Error::Command("invalid stored queue message ID"))?; - return Ok(QueueSendOutcome::Sent { message_id }); - } - - let message_id = message_id(namespace, request.producer_id); - let expires_at_ms = now_ms - .checked_add(RETENTION_MS) - .ok_or(Error::Command("queue expiry overflow"))?; - transaction.execute( - "INSERT INTO queue_messages(message_id, payload, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, result_code) VALUES (?1, ?2, 0, 0, ?3, ?4, NULL, NULL, NULL)", - ( - message_id.as_slice(), - request.payload.as_slice(), - due_at_ms, - expires_at_ms, - ), - )?; - transaction.execute( - "INSERT INTO queue_dedup(producer_id, payload_digest, message_id, retain_until_ms) VALUES (?1, ?2, ?3, ?4)", - ( - request.producer_id.as_slice(), - digest.as_slice(), - message_id.as_slice(), - expires_at_ms, - ), - )?; - Ok(QueueSendOutcome::Sent { message_id }) -} - -/// Reclaims expired leases and claims a bounded ready batch. -pub fn queue_claim( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, - lease_ms: u32, - tokens: &mut impl QueueTokenSource, -) -> Result> { - queue_claim_with_dead_letter(transaction, now_ms, limit, lease_ms, tokens, None) -} - -pub(crate) fn queue_claim_with_dead_letter( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, - lease_ms: u32, - tokens: &mut impl QueueTokenSource, - dead_letter: Option<&mut QueueDeadLetterWriter<'_>>, -) -> Result> { - validate_now(now_ms)?; - if !(1..=MAX_CLAIM_ITEMS).contains(&limit) { - return Err(Error::Command("queue claim limit must be in 1..=32")); - } - if !(MIN_LEASE_MS..=MAX_LEASE_MS).contains(&lease_ms) { - return Err(Error::Command("queue lease must be in 5..=300 seconds")); - } - if queue_paused(transaction)? { - return Ok(Vec::new()); - } - queue_reclaim_expired_bounded_with_dead_letter( - transaction, - now_ms, - MAX_RECLAIM_ITEMS, - dead_letter, - )?; - - let mut statement = transaction.prepare( - "SELECT message_id, payload, attempt, expires_at_ms FROM queue_messages INDEXED BY queue_ready WHERE state = 0 AND due_at_ms <= ?1 AND expires_at_ms > ?1 AND attempt < ?2 ORDER BY due_at_ms, message_id LIMIT ?3", - )?; - let rows = statement.query_map( - (now_ms, i64::from(MAX_ATTEMPTS), (limit + 1) as i64), - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, i64>(3)?, - )) - }, - )?; - let mut candidates = Vec::with_capacity(limit); - let mut payload_bytes = 0_usize; - for row in rows { - if candidates.len() == limit { - break; - } - let (message_id, payload, attempt, expires_at_ms) = row?; - let next_bytes = payload_bytes - .checked_add(payload.len()) - .ok_or(Error::Command("queue claim byte count overflow"))?; - if next_bytes > MAX_CLAIM_BYTES { - break; - } - let message_id: [u8; 16] = message_id - .try_into() - .map_err(|_| Error::Command("invalid stored queue message ID"))?; - if payload.len() > MAX_PAYLOAD_BYTES || attempt < 0 { - return Err(Error::Command("invalid stored queue message")); - } - let attempt = - u32::try_from(attempt).map_err(|_| Error::Command("invalid stored queue attempt"))?; - payload_bytes = next_bytes; - candidates.push((message_id, payload, attempt, expires_at_ms)); - } - drop(statement); - - let requested_deadline = now_ms - .checked_add(i64::from(lease_ms)) - .ok_or(Error::Command("queue lease deadline overflow"))?; - let mut claimed = Vec::with_capacity(candidates.len()); - for (message_id, payload, attempt, expires_at_ms) in candidates { - let token = tokens.next_token()?; - let lease_until_ms = requested_deadline.min(expires_at_ms); - let changed = transaction.execute( - "UPDATE queue_messages SET state = 1, attempt = attempt + 1, token = ?1, lease_until_ms = ?2 WHERE message_id = ?3 AND state = 0 AND due_at_ms <= ?4 AND expires_at_ms > ?4 AND attempt < ?5", - ( - token.as_slice(), - lease_until_ms, - message_id.as_slice(), - now_ms, - i64::from(MAX_ATTEMPTS), - ), - )?; - if changed != 1 { - return Err(Error::Command("queue claim lost selected ready row")); - } - claimed.push(QueueMessage { - message_id, - payload, - token, - attempt: attempt + 1, - lease_until_ms, - }); - } - Ok(claimed) -} - -/// Rechecks published claim tokens immediately before task emission. -pub fn queue_validate_claim( - connection: &Connection, - now_ms: i64, - claimed: &[QueueMessage], -) -> Result { - validate_now(now_ms)?; - if queue_paused(connection)? { - return Ok(false); - } - for message in claimed { - let state = connection - .query_row( - "SELECT state, attempt, token, lease_until_ms FROM queue_messages WHERE message_id = ?1", - [message.message_id.as_slice()], - |row| { - Ok(( - row.get::<_, i64>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, Option>>(2)?, - row.get::<_, Option>(3)?, - )) - }, - ) - .optional()?; - let Some((state, attempt, token, lease_until_ms)) = state else { - return Ok(false); - }; - let margin = now_ms - .checked_add(DELIVERY_MARGIN_MS) - .ok_or(Error::Command("queue delivery margin overflow"))?; - if QueueState::decode(state)? != QueueState::Leased - || attempt != i64::from(message.attempt) - || token.as_deref() != Some(message.token.as_slice()) - || lease_until_ms != Some(message.lease_until_ms) - || message.lease_until_ms < margin - { - return Ok(false); - } - } - Ok(true) -} - -/// Applies a bounded administrative action to one queue shard. -pub fn queue_control( - transaction: &Transaction<'_>, - now_ms: i64, - action: QueueControlAction, -) -> Result { - validate_now(now_ms)?; - match action { - QueueControlAction::Pause => { - let generation = set_queue_paused(transaction, now_ms, true)?; - Ok(QueueControlOutcome::Paused { generation }) - } - QueueControlAction::Resume => { - let generation = set_queue_paused(transaction, now_ms, false)?; - Ok(QueueControlOutcome::Resumed { generation }) - } - QueueControlAction::Purge { limit } => { - let messages = purge_queue(transaction, limit)?; - Ok(QueueControlOutcome::Purged { messages }) - } - QueueControlAction::Redrive { limit } => { - let messages = redrive_queue(transaction, now_ms, limit)?; - Ok(QueueControlOutcome::Redriven { messages }) - } - } -} - -/// Returns aggregate state for one queue shard. -pub fn queue_info(connection: &Connection) -> Result { - let (paused, generation, ready, leased, acked, dead) = connection.query_row( - "SELECT paused, generation, ready_count, leased_count, acked_count, dead_count FROM queue_control WHERE singleton = 1", - [], - |row| { - Ok(( - row.get::<_, i64>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, i64>(3)?, - row.get::<_, i64>(4)?, - row.get::<_, i64>(5)?, - )) - }, - )?; - Ok(QueueInfo { - paused: paused != 0, - generation: nonnegative_u64(generation, "invalid queue control generation")?, - ready: nonnegative_u64(ready, "invalid ready queue count")?, - leased: nonnegative_u64(leased, "invalid leased queue count")?, - acked: nonnegative_u64(acked, "invalid acked queue count")?, - dead: nonnegative_u64(dead, "invalid dead queue count")?, - }) -} - -/// Verifies Queue's transactionally maintained state counters without repairing them. -pub fn verify_queue_counts(connection: &Connection) -> Result<()> { - let stored = connection.query_row( - "SELECT ready_count, leased_count, acked_count, dead_count FROM queue_control WHERE singleton = 1", - [], - |row| { - Ok(( - row.get::<_, i64>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, i64>(3)?, - )) - }, - )?; - let computed = connection.query_row( - "SELECT count(*) FILTER (WHERE state = 0), count(*) FILTER (WHERE state = 1), count(*) FILTER (WHERE state = 2), count(*) FILTER (WHERE state = 3) FROM queue_messages", - [], - |row| { - Ok(( - row.get::<_, i64>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, i64>(3)?, - )) - }, - )?; - if stored != computed { - return Err(Error::Command("queue state counters do not match messages")); - } - Ok(()) -} - -/// Applies an ack, retry or extension only to the exact live lease token. -pub fn queue_apply_lease( - transaction: &Transaction<'_>, - now_ms: i64, - message_id: [u8; 16], - token: [u8; 16], - action: QueueLeaseAction, -) -> Result { - queue_apply_lease_with_dead_letter(transaction, now_ms, message_id, token, action, None) -} - -pub(crate) fn queue_apply_lease_with_dead_letter( - transaction: &Transaction<'_>, - now_ms: i64, - message_id: [u8; 16], - token: [u8; 16], - action: QueueLeaseAction, - dead_letter: Option<&mut QueueDeadLetterWriter<'_>>, -) -> Result { - validate_now(now_ms)?; - let current = transaction - .query_row( - "SELECT state, attempt, token, lease_until_ms, expires_at_ms, payload FROM queue_messages WHERE message_id = ?1", - [message_id.as_slice()], - |row| { - Ok(( - row.get::<_, i64>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, Option>>(2)?, - row.get::<_, Option>(3)?, - row.get::<_, i64>(4)?, - row.get::<_, Vec>(5)?, - )) - }, - ) - .optional()?; - let Some((state, attempt, stored_token, lease_until_ms, expires_at_ms, payload)) = current - else { - return Ok(QueueLeaseOutcome::LeaseLost); - }; - let live = QueueState::decode(state)? == QueueState::Leased - && stored_token.as_deref() == Some(token.as_slice()) - && lease_until_ms.is_some_and(|deadline| deadline > now_ms); - if !live { - return Ok(QueueLeaseOutcome::LeaseLost); - } - let prior_deadline = lease_until_ms.ok_or(Error::Command("leased queue row lacks deadline"))?; - let (next_state, next_deadline, due_at_ms) = match action { - QueueLeaseAction::Ack => (QueueState::Acked, None, None), - QueueLeaseAction::Retry { delay_ms } => { - if delay_ms > MAX_RETRY_DELAY_MS { - return Err(Error::Command("queue retry delay exceeds one hour")); - } - let due_at_ms = now_ms - .checked_add(i64::from(delay_ms)) - .ok_or(Error::Command("queue retry time overflow"))?; - if attempt >= i64::from(MAX_ATTEMPTS) || due_at_ms >= expires_at_ms { - (QueueState::Dead, None, None) - } else { - (QueueState::Ready, None, Some(due_at_ms)) - } - } - QueueLeaseAction::Extend { extension_ms } => { - if !(MIN_LEASE_MS..=MAX_LEASE_MS).contains(&extension_ms) { - return Err(Error::Command("queue extension must be in 5..=300 seconds")); - } - let requested = now_ms - .checked_add(i64::from(extension_ms)) - .ok_or(Error::Command("queue extension overflow"))?; - let deadline = prior_deadline.max(requested.min(expires_at_ms)); - (QueueState::Leased, Some(deadline), None) - } - }; - let dead_letter_effect_id = if next_state == QueueState::Dead { - dead_letter - .map(|writer| writer.insert(transaction, message_id, &payload, now_ms)) - .transpose()? - } else { - None - }; - let changed = match next_state { - QueueState::Leased => transaction.execute( - "UPDATE queue_messages SET lease_until_ms = ?1 WHERE message_id = ?2 AND state = 1 AND token = ?3 AND lease_until_ms > ?4", - (next_deadline, message_id.as_slice(), token.as_slice(), now_ms), - )?, - QueueState::Ready => transaction.execute( - "UPDATE queue_messages SET state = 0, due_at_ms = ?1, token = NULL, lease_until_ms = NULL WHERE message_id = ?2 AND state = 1 AND token = ?3 AND lease_until_ms > ?4", - (due_at_ms, message_id.as_slice(), token.as_slice(), now_ms), - )?, - QueueState::Acked | QueueState::Dead => transaction.execute( - "UPDATE queue_messages SET state = ?1, token = NULL, lease_until_ms = NULL, dead_letter_effect_id = ?5 WHERE message_id = ?2 AND state = 1 AND token = ?3 AND lease_until_ms > ?4", - ( - next_state.encode(), - message_id.as_slice(), - token.as_slice(), - now_ms, - dead_letter_effect_id.as_ref().map(<[u8; 32]>::as_slice), - ), - )?, - }; - if changed != 1 { - return Err(Error::Command("queue lease changed after validation")); - } - Ok(QueueLeaseOutcome::Applied { - state: next_state, - lease_until_ms: next_deadline, - }) -} - -/// Deletes bounded expired dedup and terminal message rows. -pub fn queue_cleanup_expired(transaction: &Transaction<'_>, now_ms: i64) -> Result { - queue_cleanup_expired_bounded(transaction, now_ms, MAX_RECLAIM_ITEMS) -} - -pub(crate) fn queue_cleanup_expired_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - validate_now(now_ms)?; - validate_maintenance_limit(limit)?; - if limit == 0 { - return Ok(0); - } - let messages = transaction.execute( - "DELETE FROM queue_messages WHERE message_id IN (SELECT message_id FROM queue_messages INDEXED BY queue_retention WHERE expires_at_ms <= ?1 AND (state = 2 OR (state = 3 AND (dead_letter_effect_id IS NULL OR NOT EXISTS (SELECT 1 FROM sys_effects WHERE effect_id = queue_messages.dead_letter_effect_id AND state IN (0, 1, 3))))) ORDER BY expires_at_ms, message_id LIMIT ?2)", - (now_ms, limit as i64), - )?; - let remaining = limit.saturating_sub(messages); - if remaining == 0 { - return Ok(messages); - } - let dedup = transaction.execute( - "DELETE FROM queue_dedup WHERE producer_id IN (SELECT producer_id FROM queue_dedup INDEXED BY queue_dedup_expiry WHERE retain_until_ms <= ?1 AND NOT EXISTS (SELECT 1 FROM queue_messages WHERE message_id = queue_dedup.message_id) ORDER BY retain_until_ms, producer_id LIMIT ?2)", - (now_ms, remaining as i64), - )?; - messages - .checked_add(dedup) - .ok_or(Error::Command("queue cleanup count overflow")) -} - -pub(crate) fn queue_reclaim_expired_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - queue_reclaim_expired_bounded_with_dead_letter(transaction, now_ms, limit, None) -} - -pub(crate) fn queue_reclaim_expired_bounded_with_dead_letter( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, - mut dead_letter: Option<&mut QueueDeadLetterWriter<'_>>, -) -> Result { - validate_now(now_ms)?; - validate_maintenance_limit(limit)?; - if limit == 0 { - return Ok(0); - } - let mut statement = transaction.prepare( - "SELECT message_id, attempt, expires_at_ms, payload FROM queue_messages INDEXED BY queue_leases WHERE state = 1 AND lease_until_ms <= ?1 ORDER BY lease_until_ms, message_id LIMIT ?2", - )?; - let rows = statement.query_map((now_ms, limit as i64), |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, Vec>(3)?, - )) - })?; - let mut expired = Vec::new(); - for row in rows { - expired.push(row?); - } - drop(statement); - let count = expired.len(); - for (message_id, attempt, expires_at_ms, payload) in expired { - let message_id: [u8; 16] = message_id - .try_into() - .map_err(|_| Error::Command("invalid stored queue message ID"))?; - let next_state = if attempt >= i64::from(MAX_ATTEMPTS) || expires_at_ms <= now_ms { - QueueState::Dead - } else { - QueueState::Ready - }; - let dead_letter_effect_id = if next_state == QueueState::Dead { - dead_letter - .as_deref_mut() - .map(|writer| writer.insert(transaction, message_id, &payload, now_ms)) - .transpose()? - } else { - None - }; - let changed = transaction.execute( - "UPDATE queue_messages SET state = ?1, due_at_ms = CASE WHEN ?1 = 0 THEN ?2 ELSE due_at_ms END, token = NULL, lease_until_ms = NULL, dead_letter_effect_id = ?4 WHERE message_id = ?3 AND state = 1 AND lease_until_ms <= ?2", - ( - next_state.encode(), - now_ms, - message_id.as_slice(), - dead_letter_effect_id.as_ref().map(<[u8; 32]>::as_slice), - ), - )?; - if changed != 1 { - return Err(Error::Command("queue reclaim lost selected lease")); - } - } - Ok(count) -} - -pub(crate) fn queue_expire_ready_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - queue_expire_ready_bounded_with_dead_letter(transaction, now_ms, limit, None) -} - -pub(crate) fn queue_expire_ready_bounded_with_dead_letter( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, - mut dead_letter: Option<&mut QueueDeadLetterWriter<'_>>, -) -> Result { - validate_now(now_ms)?; - validate_maintenance_limit(limit)?; - if limit == 0 { - return Ok(0); - } - let mut statement = transaction.prepare( - "SELECT message_id, payload FROM queue_messages INDEXED BY queue_ready WHERE state = 0 AND (expires_at_ms <= ?1 OR attempt >= ?2) ORDER BY due_at_ms, message_id LIMIT ?3", - )?; - let rows = statement.query_map((now_ms, i64::from(MAX_ATTEMPTS), limit as i64), |row| { - Ok((row.get::<_, Vec>(0)?, row.get::<_, Vec>(1)?)) - })?; - let expired = rows.collect::, _>>()?; - drop(statement); - for (message_id, payload) in &expired { - let message_id: [u8; 16] = message_id - .as_slice() - .try_into() - .map_err(|_| Error::Command("invalid stored queue message ID"))?; - let dead_letter_effect_id = dead_letter - .as_deref_mut() - .map(|writer| writer.insert(transaction, message_id, payload, now_ms)) - .transpose()?; - let changed = transaction.execute( - "UPDATE queue_messages SET state = 3, dead_letter_effect_id = ?1 WHERE message_id = ?2 AND state = 0 AND (expires_at_ms <= ?3 OR attempt >= ?4)", - ( - dead_letter_effect_id.as_ref().map(<[u8; 32]>::as_slice), - message_id.as_slice(), - now_ms, - i64::from(MAX_ATTEMPTS), - ), - )?; - if changed != 1 { - return Err(Error::Command("queue expiry lost selected ready row")); - } - } - Ok(expired.len()) -} - -fn queue_paused(connection: &Connection) -> Result { - let paused = connection.query_row( - "SELECT paused FROM queue_control WHERE singleton = 1", - [], - |row| row.get::<_, i64>(0), - )?; - Ok(paused != 0) -} - -fn set_queue_paused(transaction: &Transaction<'_>, now_ms: i64, paused: bool) -> Result { - let paused_value = if paused { 1_i64 } else { 0_i64 }; - transaction.execute( - "UPDATE queue_control SET paused = ?1, generation = generation + CASE WHEN paused = ?1 THEN 0 ELSE 1 END, updated_at_ms = ?2 WHERE singleton = 1", - (paused_value, now_ms), - )?; - let generation = transaction.query_row( - "SELECT generation FROM queue_control WHERE singleton = 1", - [], - |row| row.get::<_, i64>(0), - )?; - nonnegative_u64(generation, "invalid queue control generation") -} - -fn purge_queue(transaction: &Transaction<'_>, limit: u32) -> Result { - let limit = control_limit(limit)?; - let mut statement = transaction.prepare( - "SELECT message_id FROM queue_messages WHERE state != 1 ORDER BY message_id LIMIT ?1", - )?; - let rows = statement.query_map([i64::from(limit)], |row| row.get::<_, Vec>(0))?; - let ids = rows.collect::, _>>()?; - drop(statement); - for id in &ids { - transaction.execute("DELETE FROM queue_dedup WHERE message_id = ?1", [id])?; - let changed = transaction.execute( - "DELETE FROM queue_messages WHERE message_id = ?1 AND state != 1", - [id], - )?; - if changed != 1 { - return Err(Error::Command("queue purge lost selected message")); - } - } - u32::try_from(ids.len()).map_err(|_| Error::Command("queue purge count overflow")) -} - -fn redrive_queue(transaction: &Transaction<'_>, now_ms: i64, limit: u32) -> Result { - let limit = control_limit(limit)?; - let changed = transaction.execute( - "UPDATE queue_messages SET state = 0, attempt = 0, due_at_ms = ?1, expires_at_ms = ?2, token = NULL, lease_until_ms = NULL, result_code = NULL, dead_letter_effect_id = NULL WHERE message_id IN (SELECT message_id FROM queue_messages WHERE state = 3 AND (dead_letter_effect_id IS NULL OR NOT EXISTS (SELECT 1 FROM sys_effects WHERE effect_id = queue_messages.dead_letter_effect_id AND state IN (0, 1))) ORDER BY message_id LIMIT ?3)", - ( - now_ms, - now_ms - .checked_add(RETENTION_MS) - .ok_or(Error::Command("queue redrive expiry overflow"))?, - i64::from(limit), - ), - )?; - u32::try_from(changed).map_err(|_| Error::Command("queue redrive count overflow")) -} - -fn control_limit(limit: u32) -> Result { - if !(1..=MAX_RECLAIM_ITEMS as u32).contains(&limit) { - return Err(Error::Command("queue control limit must be in 1..=128")); - } - Ok(limit) -} - -fn nonnegative_u64(value: i64, message: &'static str) -> Result { - u64::try_from(value).map_err(|_| Error::Command(message)) -} - -fn validate_maintenance_limit(limit: usize) -> Result<()> { - if limit > MAX_RECLAIM_ITEMS { - return Err(Error::Command("queue maintenance limit exceeds 128")); - } - Ok(()) -} - -fn message_id(namespace: NamespaceId, producer_id: [u8; 16]) -> [u8; 16] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.queue-message.v1\0"); - hasher.update(namespace.as_bytes()); - hasher.update(&producer_id); - let mut id = [0; 16]; - id.copy_from_slice(&hasher.finalize().as_bytes()[..16]); - id -} - -fn dead_letter_producer_id( - target: NamespaceId, - source: CellId, - incarnation: IncarnationId, - message_id: [u8; 16], -) -> [u8; 16] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.queue-dead-letter.v1\0"); - hasher.update(target.as_bytes()); - hasher.update(source.as_bytes()); - hasher.update(incarnation.as_bytes()); - hasher.update(&message_id); - let mut id = [0; 16]; - id.copy_from_slice(&hasher.finalize().as_bytes()[..16]); - id -} - -fn send_digest(payload: &[u8], available_at_ms: i64) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.queue-send.v1\0"); - hasher.update(&(payload.len() as u32).to_be_bytes()); - hasher.update(payload); - hasher.update(&available_at_ms.to_be_bytes()); - *hasher.finalize().as_bytes() -} - -fn validate_now(now_ms: i64) -> Result<()> { - if now_ms < 0 { - return Err(Error::Command("negative queue logical time")); - } - Ok(()) -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/primitives/queue/api.rs b/crates/crab-cell-runtime/src/primitives/queue/api.rs deleted file mode 100644 index a5c239fef..000000000 --- a/crates/crab-cell-runtime/src/primitives/queue/api.rs +++ /dev/null @@ -1,558 +0,0 @@ -use std::marker::PhantomData; - -use crate::cell::catalog::CatalogRole; -use crate::client::{CellClient, Committed, InvocationError, Observed, Receipt}; -use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; -use crate::identity::{ - ApplicationId, CellTarget, NamespaceId, TenantId, partition_for_shard, shard_for_scope, -}; -use crate::primitives::maintenance::{MaintenanceModule, register_maintenance}; -use crate::registry::{Command, Query, RegistryBuilder}; -use crate::registry::{CommandContext, CommandResult, QueryContext}; - -mod codec; - -use super::{ - MAX_ATTEMPTS, MAX_CLAIM_ITEMS, MAX_PAYLOAD_BYTES, QueueControlAction, QueueControlOutcome, - QueueDeadLetterWriter, QueueInfo, QueueLeaseAction, QueueLeaseOutcome, QueueMessage, - QueueSendOutcome, QueueSendRequest, QueueState, SystemQueueTokens, queue_apply_lease, - queue_apply_lease_with_dead_letter, queue_claim, queue_claim_with_dead_letter, queue_control, - queue_info, queue_send, queue_validate_claim, -}; - -/// Compile-time routing contract for one Queue namespace's dead-letter target. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct QueueDeadLetterTarget { - module: &'static str, - namespace: NamespaceId, - shards: u32, - send_command_id: u32, - codec_version: u32, -} - -impl QueueDeadLetterTarget { - /// Declares one module's dead-letter target: its module name, namespace, - /// shard count, send command id, and codec version. - #[must_use] - pub const fn new( - module: &'static str, - namespace: NamespaceId, - shards: u32, - send_command_id: u32, - codec_version: u32, - ) -> Self { - Self { - module, - namespace, - shards, - send_command_id, - codec_version, - } - } - - pub(crate) const fn module(self) -> &'static str { - self.module - } - - pub(crate) const fn namespace(self) -> NamespaceId { - self.namespace - } - - pub(crate) const fn shards(self) -> u32 { - self.shards - } - - pub(crate) const fn send_command_id(self) -> u32 { - self.send_command_id - } - - pub(crate) const fn codec_version(self) -> u32 { - self.codec_version - } -} - -/// Compile-time namespace and operation identifiers for one native Queue module. -pub trait QueueModule: MaintenanceModule { - /// Namespace that owns this Queue module. - const NAMESPACE: NamespaceId; - /// Command id that sends a message. - const SEND_COMMAND_ID: u32; - /// Command id that claims a delivery lease. - const CLAIM_COMMAND_ID: u32; - /// Command id that acknowledges, retries, or extends a lease. - const LEASE_COMMAND_ID: u32; - /// Query id that revalidates claimed leases. - const VALIDATE_QUERY_ID: u32; - /// Command id that pauses, resumes, purges, or redrives the shard. - const CONTROL_COMMAND_ID: u32; - /// Query id that reads shard state. - const INFO_QUERY_ID: u32; -} - -/// Registers all typed Queue bindings contributed by one compiled module. -pub fn register_queue(registry: &mut RegistryBuilder) -> crate::Result<()> { - registry.bind_queue_module( - M::MODULE, - M::NAMESPACE, - M::SEND_COMMAND_ID, - M::CODEC_VERSION, - M::QUEUE_DEAD_LETTER, - )?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_query::>()?; - registry.bind_query::>()?; - register_maintenance::(registry) -} - -/// Typed queue control command bound to immutable module operation IDs. -pub struct QueueControlCommand(PhantomData M>); - -impl Command for QueueControlCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::CONTROL_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = QueueControlAction; - type Output = QueueControlOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - Ok(CommandResult::Success(queue_control( - context.primitive_transaction(), - context.now_ms(), - input, - )?)) - } -} - -/// Empty request for queue shard information. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct QueueInfoRequest; - -/// Typed queue shard information query. -pub struct QueueInfoQuery(PhantomData M>); - -impl Query for QueueInfoQuery { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::INFO_QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = QueueInfoRequest; - type Output = QueueInfo; - - fn execute(context: &mut QueryContext<'_>, _: Self::Input) -> crate::Result { - queue_info(context.primitive_connection()) - } -} - -/// Typed Queue send bound to immutable module operation IDs. -pub struct QueueSendCommand(PhantomData M>); - -impl Command for QueueSendCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::SEND_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = QueueSendRequest; - type Output = QueueSendOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let outcome = queue_send( - context.primitive_transaction(), - M::NAMESPACE, - context.now_ms(), - &input, - )?; - Ok(match outcome { - QueueSendOutcome::Sent { .. } => CommandResult::Success(outcome), - QueueSendOutcome::ProducerConflict => CommandResult::Rejected(outcome), - }) - } -} - -/// Bounded claim parameters for one explicitly selected Queue shard. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct QueueClaimRequest { - /// Maximum messages to claim. - pub limit: u32, - /// Lease duration granted to each claim. - pub lease_ms: u32, -} - -/// Typed Queue claim bound to immutable module operation IDs. -pub struct QueueClaimCommand(PhantomData M>); - -impl Command for QueueClaimCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::CLAIM_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = QueueClaimRequest; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let limit = usize::try_from(input.limit) - .map_err(|_| crate::Error::Command("queue claim limit overflow"))?; - let mut tokens = SystemQueueTokens; - let claimed = if let Some(target) = M::QUEUE_DEAD_LETTER { - let now_ms = context.now_ms(); - let (transaction, effects) = context.primitive_effects()?; - let mut writer = QueueDeadLetterWriter::new(target, effects); - queue_claim_with_dead_letter( - transaction, - now_ms, - limit, - input.lease_ms, - &mut tokens, - Some(&mut writer), - )? - } else { - queue_claim( - context.primitive_transaction(), - context.now_ms(), - limit, - input.lease_ms, - &mut tokens, - )? - }; - Ok(CommandResult::Success(claimed)) - } -} - -/// Exact lease identity and transition for one Queue message. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct QueueLeaseRequest { - /// Message the lease belongs to. - pub message_id: [u8; 16], - /// Lease token that must match the current lease. - pub token: [u8; 16], - /// Transition to apply. - pub action: QueueLeaseAction, -} - -/// Typed Queue lease transition bound to immutable module operation IDs. -pub struct QueueLeaseCommand(PhantomData M>); - -impl Command for QueueLeaseCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::LEASE_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = QueueLeaseRequest; - type Output = QueueLeaseOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let outcome = if let Some(target) = M::QUEUE_DEAD_LETTER { - let now_ms = context.now_ms(); - let (transaction, effects) = context.primitive_effects()?; - let mut writer = QueueDeadLetterWriter::new(target, effects); - queue_apply_lease_with_dead_letter( - transaction, - now_ms, - input.message_id, - input.token, - input.action, - Some(&mut writer), - )? - } else { - queue_apply_lease( - context.primitive_transaction(), - context.now_ms(), - input.message_id, - input.token, - input.action, - )? - }; - Ok(match outcome { - QueueLeaseOutcome::Applied { .. } => CommandResult::Success(outcome), - QueueLeaseOutcome::LeaseLost => CommandResult::Rejected(outcome), - }) - } -} - -/// Exact published claims to revalidate before native task emission. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct QueueValidateRequest { - /// Claims to revalidate before native task emission. - pub claimed: Vec, -} - -/// Typed Queue claim validation bound to immutable module operation IDs. -pub struct QueueValidateClaimQuery(PhantomData M>); - -impl Query for QueueValidateClaimQuery { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::VALIDATE_QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = QueueValidateRequest; - type Output = bool; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - queue_validate_claim( - context.primitive_connection(), - context.now_ms(), - &input.claimed, - ) - } -} - -/// Authorized native Queue capability with deterministic producer sharding. -pub struct QueueNamespace { - client: CellClient, - tenant: TenantId, - application: ApplicationId, - shards: u32, - module: PhantomData M>, -} - -impl Clone for QueueNamespace { - fn clone(&self) -> Self { - Self { - client: self.client.clone(), - tenant: self.tenant, - application: self.application, - shards: self.shards, - module: PhantomData, - } - } -} - -impl QueueNamespace { - /// Creates a Queue capability after validating the compiled namespace role. - pub fn new( - client: CellClient, - tenant: TenantId, - application: ApplicationId, - ) -> crate::Result { - let shards = client.require_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Queue)?; - Ok(Self { - client, - tenant, - application, - shards, - module: PhantomData, - }) - } - - /// Sends one producer-deduplicated message to its deterministic shard. - pub async fn send( - &self, - identity: crate::cell::executor::MutationIdentity, - request: QueueSendRequest, - ) -> std::result::Result, InvocationError> { - let target = self - .producer_target(&request.producer_id) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>(&target, identity, request) - .await - } - - /// Claims a bounded message batch from one explicitly selected shard. - pub async fn claim( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - request: QueueClaimRequest, - ) -> std::result::Result>, InvocationError>> { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>(&target, identity, request) - .await - } - - /// Revalidates exact published leases before their payloads are emitted. - pub async fn validate_claim( - &self, - shard: u32, - claimed: Vec, - minimum: Option, - ) -> std::result::Result, InvocationError> { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - // Lease checks gate external work; a stale snapshot cannot prove that - // the owner has not revoked or replaced the claim. - self.client - .with_read_policy(crate::client::ReadPolicy::CurrentOwner) - .query::>(&target, minimum, QueueValidateRequest { claimed }) - .await - } - - /// Acknowledges one exact live message lease on its claimed shard. - pub async fn ack( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - message_id: [u8; 16], - token: [u8; 16], - ) -> std::result::Result, InvocationError> { - self.apply_lease(identity, shard, message_id, token, QueueLeaseAction::Ack) - .await - } - - /// Returns one exact live lease to its shard after a bounded delay. - pub async fn retry( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - message_id: [u8; 16], - token: [u8; 16], - delay_ms: u32, - ) -> std::result::Result, InvocationError> { - self.apply_lease( - identity, - shard, - message_id, - token, - QueueLeaseAction::Retry { delay_ms }, - ) - .await - } - - /// Extends one exact live lease without shortening its current deadline. - pub async fn extend( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - message_id: [u8; 16], - token: [u8; 16], - extension_ms: u32, - ) -> std::result::Result, InvocationError> { - self.apply_lease( - identity, - shard, - message_id, - token, - QueueLeaseAction::Extend { extension_ms }, - ) - .await - } - - /// Pauses new claims while allowing live lease settlement. - pub async fn pause( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - ) -> std::result::Result, InvocationError> - { - self.control(identity, shard, QueueControlAction::Pause) - .await - } - - /// Resumes claims on one queue shard. - pub async fn resume( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - ) -> std::result::Result, InvocationError> - { - self.control(identity, shard, QueueControlAction::Resume) - .await - } - - /// Deletes a bounded batch of non-leased messages. - pub async fn purge( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - limit: u32, - ) -> std::result::Result, InvocationError> - { - self.control(identity, shard, QueueControlAction::Purge { limit }) - .await - } - - /// Returns a bounded batch of dead messages to ready state. - pub async fn redrive( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - limit: u32, - ) -> std::result::Result, InvocationError> - { - self.control(identity, shard, QueueControlAction::Redrive { limit }) - .await - } - - /// Reads aggregate control and lifecycle state for one queue shard. - pub async fn info( - &self, - shard: u32, - minimum: Option, - ) -> std::result::Result, InvocationError> { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - self.client - .query::>(&target, minimum, QueueInfoRequest) - .await - } - - async fn control( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - action: QueueControlAction, - ) -> std::result::Result, InvocationError> - { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>(&target, identity, action) - .await - } - - async fn apply_lease( - &self, - identity: crate::cell::executor::MutationIdentity, - shard: u32, - message_id: [u8; 16], - token: [u8; 16], - action: QueueLeaseAction, - ) -> std::result::Result, InvocationError> { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>( - &target, - identity, - QueueLeaseRequest { - message_id, - token, - action, - }, - ) - .await - } - - fn producer_target(&self, producer_id: &[u8; 16]) -> crate::Result { - let shard = shard_for_scope(M::NAMESPACE, producer_id, self.shards)?; - self.shard_target(shard) - } - - fn shard_target(&self, shard: u32) -> crate::Result { - if shard >= self.shards { - return Err(crate::Error::Identity("queue shard outside namespace")); - } - CellTarget::new( - self.tenant, - self.application, - M::NAMESPACE, - &partition_for_shard(shard), - ) - } -} diff --git a/crates/crab-cell-runtime/src/primitives/queue/api/codec.rs b/crates/crab-cell-runtime/src/primitives/queue/api/codec.rs deleted file mode 100644 index 59116c788..000000000 --- a/crates/crab-cell-runtime/src/primitives/queue/api/codec.rs +++ /dev/null @@ -1,440 +0,0 @@ -//! Queue wire codecs for send, claim, lease, control, and info payloads. - -use super::*; -use crate::codec::read_fixed; - -const SENT_TAG: u8 = 0; -const PRODUCER_CONFLICT_TAG: u8 = 1; -const ACK_TAG: u8 = 0; -const RETRY_TAG: u8 = 1; -const EXTEND_TAG: u8 = 2; -const APPLIED_TAG: u8 = 0; -const LEASE_LOST_TAG: u8 = 1; -const PAUSE_TAG: u8 = 0; -const RESUME_TAG: u8 = 1; -const PURGE_TAG: u8 = 2; -const REDRIVE_TAG: u8 = 3; - -impl WireValue for QueueSendRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.producer_id)?; - encoder.write_bytes(&self.payload)?; - encoder.write_i64(self.available_at_ms) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - producer_id: read_fixed(decoder, "queue producer ID length")?, - payload: decoder.read_bytes()?.to_vec(), - available_at_ms: decoder.read_i64()?, - }) - } -} - -impl WireValue for QueueSendOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Sent { message_id } => { - encoder.write_u8(SENT_TAG)?; - encoder.write_bytes(message_id) - } - Self::ProducerConflict => encoder.write_u8(PRODUCER_CONFLICT_TAG), - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - SENT_TAG => Ok(Self::Sent { - message_id: read_fixed(decoder, "queue message ID length")?, - }), - PRODUCER_CONFLICT_TAG => Ok(Self::ProducerConflict), - _ => Err(CodecError::Invalid("invalid queue send outcome tag")), - } - } -} - -impl WireValue for QueueClaimRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_u32(self.limit)?; - encoder.write_u32(self.lease_ms) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - limit: decoder.read_u32()?, - lease_ms: decoder.read_u32()?, - }) - } -} - -impl WireValue for QueueMessage { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - validate_message(self)?; - encoder.write_bytes(&self.message_id)?; - encoder.write_bytes(&self.payload)?; - encoder.write_bytes(&self.token)?; - encoder.write_u32(self.attempt)?; - encoder.write_i64(self.lease_until_ms) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let message = Self { - message_id: read_fixed(decoder, "queue message ID length")?, - payload: decoder.read_bytes()?.to_vec(), - token: read_fixed(decoder, "queue lease token length")?, - attempt: decoder.read_u32()?, - lease_until_ms: decoder.read_i64()?, - }; - validate_message(&message)?; - Ok(message) - } -} - -impl WireValue for Vec { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if self.len() > MAX_CLAIM_ITEMS { - return Err(CodecError::Invalid("too many queue messages")); - } - encoder.write_count(self.len())?; - for message in self { - message.encode(encoder)?; - } - Ok(()) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let count = decoder.read_count()?; - if count > MAX_CLAIM_ITEMS { - return Err(CodecError::Invalid("too many queue messages")); - } - let mut messages = Vec::with_capacity(count); - for _ in 0..count { - messages.push(QueueMessage::decode(decoder)?); - } - Ok(messages) - } -} - -impl WireValue for QueueLeaseRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.message_id)?; - encoder.write_bytes(&self.token)?; - match self.action { - QueueLeaseAction::Ack => encoder.write_u8(ACK_TAG), - QueueLeaseAction::Retry { delay_ms } => { - encoder.write_u8(RETRY_TAG)?; - encoder.write_u32(delay_ms) - } - QueueLeaseAction::Extend { extension_ms } => { - encoder.write_u8(EXTEND_TAG)?; - encoder.write_u32(extension_ms) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let message_id = read_fixed(decoder, "queue message ID length")?; - let token = read_fixed(decoder, "queue lease token length")?; - let action = match decoder.read_u8()? { - ACK_TAG => QueueLeaseAction::Ack, - RETRY_TAG => QueueLeaseAction::Retry { - delay_ms: decoder.read_u32()?, - }, - EXTEND_TAG => QueueLeaseAction::Extend { - extension_ms: decoder.read_u32()?, - }, - _ => return Err(CodecError::Invalid("invalid queue lease action tag")), - }; - Ok(Self { - message_id, - token, - action, - }) - } -} - -impl WireValue for QueueLeaseOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Applied { - state, - lease_until_ms, - } => { - validate_lease_outcome(*state, *lease_until_ms)?; - encoder.write_u8(APPLIED_TAG)?; - encode_state(*state, encoder)?; - lease_until_ms.encode(encoder) - } - Self::LeaseLost => encoder.write_u8(LEASE_LOST_TAG), - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - APPLIED_TAG => { - let state = decode_state(decoder)?; - let lease_until_ms = Option::::decode(decoder)?; - validate_lease_outcome(state, lease_until_ms)?; - Ok(Self::Applied { - state, - lease_until_ms, - }) - } - LEASE_LOST_TAG => Ok(Self::LeaseLost), - _ => Err(CodecError::Invalid("invalid queue lease outcome tag")), - } - } -} - -impl WireValue for QueueValidateRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - self.claimed.encode(encoder) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - claimed: Vec::::decode(decoder)?, - }) - } -} - -impl WireValue for QueueControlAction { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Pause => encoder.write_u8(PAUSE_TAG), - Self::Resume => encoder.write_u8(RESUME_TAG), - Self::Purge { limit } => { - encoder.write_u8(PURGE_TAG)?; - encoder.write_u32(*limit) - } - Self::Redrive { limit } => { - encoder.write_u8(REDRIVE_TAG)?; - encoder.write_u32(*limit) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - PAUSE_TAG => Ok(Self::Pause), - RESUME_TAG => Ok(Self::Resume), - PURGE_TAG => Ok(Self::Purge { - limit: decoder.read_u32()?, - }), - REDRIVE_TAG => Ok(Self::Redrive { - limit: decoder.read_u32()?, - }), - _ => Err(CodecError::Invalid("invalid queue control action tag")), - } - } -} - -impl WireValue for QueueControlOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Paused { generation } => { - encoder.write_u8(PAUSE_TAG)?; - encoder.write_u64(*generation) - } - Self::Resumed { generation } => { - encoder.write_u8(RESUME_TAG)?; - encoder.write_u64(*generation) - } - Self::Purged { messages } => { - encoder.write_u8(PURGE_TAG)?; - encoder.write_u32(*messages) - } - Self::Redriven { messages } => { - encoder.write_u8(REDRIVE_TAG)?; - encoder.write_u32(*messages) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - PAUSE_TAG => Ok(Self::Paused { - generation: decoder.read_u64()?, - }), - RESUME_TAG => Ok(Self::Resumed { - generation: decoder.read_u64()?, - }), - PURGE_TAG => Ok(Self::Purged { - messages: decoder.read_u32()?, - }), - REDRIVE_TAG => Ok(Self::Redriven { - messages: decoder.read_u32()?, - }), - _ => Err(CodecError::Invalid("invalid queue control outcome tag")), - } - } -} - -impl WireValue for QueueInfoRequest { - fn encode(&self, _: &mut BoundedEncoder) -> Result<(), CodecError> { - Ok(()) - } - - fn decode(_: &mut BoundedDecoder<'_>) -> Result { - Ok(Self) - } -} - -impl WireValue for QueueInfo { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - self.paused.encode(encoder)?; - encoder.write_u64(self.generation)?; - encoder.write_u64(self.ready)?; - encoder.write_u64(self.leased)?; - encoder.write_u64(self.acked)?; - encoder.write_u64(self.dead) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - paused: bool::decode(decoder)?, - generation: decoder.read_u64()?, - ready: decoder.read_u64()?, - leased: decoder.read_u64()?, - acked: decoder.read_u64()?, - dead: decoder.read_u64()?, - }) - } -} - -fn encode_state(state: QueueState, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_u8(match state { - QueueState::Ready => 0, - QueueState::Leased => 1, - QueueState::Acked => 2, - QueueState::Dead => 3, - }) -} - -fn decode_state(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(QueueState::Ready), - 1 => Ok(QueueState::Leased), - 2 => Ok(QueueState::Acked), - 3 => Ok(QueueState::Dead), - _ => Err(CodecError::Invalid("invalid queue state tag")), - } -} - -fn validate_message(message: &QueueMessage) -> Result<(), CodecError> { - if message.payload.len() > MAX_PAYLOAD_BYTES - || !(1..=MAX_ATTEMPTS).contains(&message.attempt) - || message.lease_until_ms < 0 - { - return Err(CodecError::Invalid("invalid queue claim")); - } - Ok(()) -} - -fn validate_lease_outcome( - state: QueueState, - lease_until_ms: Option, -) -> Result<(), CodecError> { - if (state == QueueState::Leased) != lease_until_ms.is_some() - || lease_until_ms.is_some_and(|deadline| deadline < 0) - { - return Err(CodecError::Invalid("inconsistent queue lease outcome")); - } - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::codec::roundtrip; - - #[test] - fn queue_codecs_roundtrip_every_operation_shape() { - roundtrip(QueueSendRequest { - producer_id: [1; 16], - payload: vec![0, 255], - available_at_ms: i64::MAX, - }); - roundtrip(QueueSendOutcome::Sent { - message_id: [2; 16], - }); - roundtrip(QueueSendOutcome::ProducerConflict); - roundtrip(QueueClaimRequest { - limit: 32, - lease_ms: 300_000, - }); - let message = QueueMessage { - message_id: [3; 16], - payload: b"job".to_vec(), - token: [4; 16], - attempt: 20, - lease_until_ms: i64::MAX, - }; - roundtrip(vec![message.clone()]); - for action in [ - QueueLeaseAction::Ack, - QueueLeaseAction::Retry { delay_ms: 100 }, - QueueLeaseAction::Extend { - extension_ms: 5_000, - }, - ] { - roundtrip(QueueLeaseRequest { - message_id: message.message_id, - token: message.token, - action, - }); - } - roundtrip(QueueLeaseOutcome::Applied { - state: QueueState::Leased, - lease_until_ms: Some(10), - }); - roundtrip(QueueLeaseOutcome::LeaseLost); - roundtrip(QueueValidateRequest { - claimed: vec![message], - }); - roundtrip(QueueControlAction::Purge { limit: 128 }); - roundtrip(QueueControlOutcome::Paused { generation: 7 }); - roundtrip(QueueInfo { - paused: true, - generation: 7, - ready: 1, - leased: 2, - acked: 3, - dead: 4, - }); - } - - #[test] - fn queue_decoder_rejects_invalid_fixed_ids_and_unbounded_claims() { - let mut bad_id = BoundedEncoder::new(32).unwrap(); - bad_id.write_bytes(&[1; 15]).unwrap(); - bad_id.write_bytes(&[]).unwrap(); - bad_id.write_i64(0).unwrap(); - let bytes = bad_id.finish(); - let mut decoder = BoundedDecoder::new(&bytes, 32).unwrap(); - assert!(matches!( - QueueSendRequest::decode(&mut decoder), - Err(CodecError::Invalid("queue producer ID length")) - )); - - let mut too_many = BoundedEncoder::new(16).unwrap(); - too_many.write_count(MAX_CLAIM_ITEMS + 1).unwrap(); - let bytes = too_many.finish(); - let mut decoder = BoundedDecoder::new(&bytes, 16).unwrap(); - assert!(matches!( - Vec::::decode(&mut decoder), - Err(CodecError::Invalid("too many queue messages")) - )); - - let mut inconsistent = BoundedEncoder::new(16).unwrap(); - inconsistent.write_u8(APPLIED_TAG).unwrap(); - encode_state(QueueState::Acked, &mut inconsistent).unwrap(); - Some(10_i64).encode(&mut inconsistent).unwrap(); - let bytes = inconsistent.finish(); - let mut decoder = BoundedDecoder::new(&bytes, 16).unwrap(); - assert!(matches!( - QueueLeaseOutcome::decode(&mut decoder), - Err(CodecError::Invalid("inconsistent queue lease outcome")) - )); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/queue/tests.rs b/crates/crab-cell-runtime/src/primitives/queue/tests.rs deleted file mode 100644 index 26e44c6ee..000000000 --- a/crates/crab-cell-runtime/src/primitives/queue/tests.rs +++ /dev/null @@ -1,585 +0,0 @@ -use prost::Message; - -use super::*; - -fn connection() -> Connection { - let mut connection = Connection::open_in_memory().unwrap(); - crate::cell::schema::install_runtime_schema( - &mut connection, - source_target().cell_id(), - IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - connection -} - -fn source_target() -> CellTarget { - CellTarget::new( - crate::TenantId::from_bytes([3; 16]), - crate::ApplicationId::from_bytes([4; 16]), - NamespaceId::from_bytes([5; 16]), - &0_u32.to_be_bytes(), - ) - .unwrap() -} - -fn dead_letter_target(shards: u32) -> QueueDeadLetterTarget { - QueueDeadLetterTarget::new( - "dead-letter", - NamespaceId::from_bytes([6; 16]), - shards, - 7, - 1, - ) -} - -fn insert_leased_message(connection: &mut Connection) -> [u8; 16] { - let transaction = connection.transaction().unwrap(); - let outcome = queue_send( - &transaction, - source_target().namespace(), - 0, - &QueueSendRequest { - producer_id: [7; 16], - payload: b"original".to_vec(), - available_at_ms: 0, - }, - ) - .unwrap(); - let QueueSendOutcome::Sent { message_id } = outcome else { - panic!("first queue send must insert") - }; - transaction - .execute( - "UPDATE queue_messages SET state = 1, attempt = 20, token = ?1, lease_until_ms = 100 WHERE message_id = ?2", - ([8_u8; 16].as_slice(), message_id.as_slice()), - ) - .unwrap(); - transaction.commit().unwrap(); - message_id -} - -#[test] -fn embedded_queue_migration_matches_normative_contract() { - assert_eq!( - QUEUE_SCHEMA, - include_str!("../../../docs/contracts/queue.sql") - ); -} - -#[test] -fn message_identity_binds_namespace_and_producer() { - assert_ne!( - message_id(NamespaceId::from_bytes([1; 16]), [2; 16]), - message_id(NamespaceId::from_bytes([3; 16]), [2; 16]) - ); -} - -#[test] -fn extend_never_shortens_a_live_lease() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - transaction - .execute( - "UPDATE queue_messages SET lease_until_ms = 50000, expires_at_ms = 60000 WHERE message_id = ?1", - [message_id.as_slice()], - ) - .unwrap(); - assert_eq!( - queue_apply_lease( - &transaction, - 10_000, - message_id, - [8; 16], - QueueLeaseAction::Extend { - extension_ms: 5_000 - }, - ) - .unwrap(), - QueueLeaseOutcome::Applied { - state: QueueState::Leased, - lease_until_ms: Some(50_000), - } - ); - transaction.commit().unwrap(); -} - -#[test] -fn extend_stops_at_the_message_expiry() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - transaction - .execute( - "UPDATE queue_messages SET lease_until_ms = 50000, expires_at_ms = 60000 WHERE message_id = ?1", - [message_id.as_slice()], - ) - .unwrap(); - assert_eq!( - queue_apply_lease( - &transaction, - 10_000, - message_id, - [8; 16], - QueueLeaseAction::Extend { - extension_ms: 300_000, - }, - ) - .unwrap(), - QueueLeaseOutcome::Applied { - state: QueueState::Leased, - lease_until_ms: Some(60_000), - } - ); - transaction.commit().unwrap(); -} - -#[test] -fn claim_validation_rejects_a_lease_inside_the_delivery_margin() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - transaction - .execute( - "UPDATE queue_messages SET lease_until_ms = 10500 WHERE message_id = ?1", - [message_id.as_slice()], - ) - .unwrap(); - transaction.commit().unwrap(); - let claimed = QueueMessage { - message_id, - payload: b"original".to_vec(), - token: [8; 16], - attempt: 20, - lease_until_ms: 10_500, - }; - assert!(!queue_validate_claim(&connection, 10_000, &[claimed]).unwrap()); -} - -#[test] -fn dead_transition_atomically_links_one_canonical_queue_effect() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - let mut effects = EffectBatch::new(&transaction, &source_target(), 1, 10).unwrap(); - let mut dead_letter = QueueDeadLetterWriter::new(dead_letter_target(1), &mut effects); - assert_eq!( - queue_apply_lease_with_dead_letter( - &transaction, - 10, - message_id, - [8; 16], - QueueLeaseAction::Retry { delay_ms: 0 }, - Some(&mut dead_letter), - ) - .unwrap(), - QueueLeaseOutcome::Applied { - state: QueueState::Dead, - lease_until_ms: None, - } - ); - let (linked, operation): (Vec, Vec) = transaction - .query_row( - "SELECT dead_letter_effect_id, operation FROM queue_messages JOIN sys_effects ON effect_id = dead_letter_effect_id WHERE message_id = ?1", - [message_id.as_slice()], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .unwrap(); - assert_eq!(linked.len(), 32); - let effect = crate::peer::wire::EffectRequest::decode(operation.as_slice()).unwrap(); - assert!(effect.destination_incarnation.is_empty()); - let command = match effect.operation { - Some(crate::peer::wire::effect_request::Operation::CellCommand(command)) => command, - None => panic!("dead letter must use the typed Queue command"), - }; - assert_eq!(command.command_id, 7); - let mut decoder = - crate::codec::BoundedDecoder::new(&command.input, QUEUE_SEND_MAX_INPUT_BYTES).unwrap(); - let request = QueueSendRequest::decode(&mut decoder).unwrap(); - decoder.finish().unwrap(); - assert_eq!(request.payload, b"original"); - assert_eq!(request.available_at_ms, 10); - transaction.commit().unwrap(); -} - -#[test] -fn dead_payload_is_retained_until_its_effect_is_terminal() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - let mut effects = EffectBatch::new(&transaction, &source_target(), 1, 10).unwrap(); - let mut dead_letter = QueueDeadLetterWriter::new(dead_letter_target(1), &mut effects); - queue_apply_lease_with_dead_letter( - &transaction, - 10, - message_id, - [8; 16], - QueueLeaseAction::Retry { delay_ms: 0 }, - Some(&mut dead_letter), - ) - .unwrap(); - transaction - .execute( - "UPDATE queue_messages SET expires_at_ms = 10 WHERE message_id = ?1", - [message_id.as_slice()], - ) - .unwrap(); - transaction - .execute("UPDATE queue_dedup SET retain_until_ms = 10", []) - .unwrap(); - assert_eq!(queue_cleanup_expired(&transaction, 10).unwrap(), 0); - let retained: (i64, i64) = transaction - .query_row( - "SELECT (SELECT count(*) FROM queue_messages), (SELECT count(*) FROM queue_dedup)", - [], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .unwrap(); - assert_eq!(retained, (1, 1)); - transaction - .execute( - "UPDATE sys_effects SET state = 2 WHERE effect_id = (SELECT dead_letter_effect_id FROM queue_messages WHERE message_id = ?1)", - [message_id.as_slice()], - ) - .unwrap(); - assert_eq!(queue_cleanup_expired(&transaction, 10).unwrap(), 2); - transaction.commit().unwrap(); -} - -#[test] -fn queue_state_counters_follow_insert_update_delete_and_report_drift() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - for id in [1_u8, 2] { - transaction - .execute( - "INSERT INTO queue_messages(message_id, payload, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, result_code) VALUES (?1, X'01', 0, 0, 0, 100, NULL, NULL, NULL)", - [vec![id; 16]], - ) - .unwrap(); - } - assert_eq!(queue_info(&transaction).unwrap().ready, 2); - transaction - .execute( - "UPDATE queue_messages SET state = 1, token = zeroblob(16), lease_until_ms = 10 WHERE message_id = ?1", - [vec![1_u8; 16]], - ) - .unwrap(); - transaction - .execute( - "UPDATE queue_messages SET state = 2, token = NULL, lease_until_ms = NULL WHERE message_id = ?1", - [vec![1_u8; 16]], - ) - .unwrap(); - transaction - .execute( - "DELETE FROM queue_messages WHERE message_id = ?1", - [vec![2_u8; 16]], - ) - .unwrap(); - verify_queue_counts(&transaction).unwrap(); - assert_eq!(queue_info(&transaction).unwrap().acked, 1); - transaction - .execute( - "UPDATE queue_control SET ready_count = ready_count + 1 WHERE singleton = 1", - [], - ) - .unwrap(); - assert!(verify_queue_counts(&transaction).is_err()); -} - -#[test] -fn repeated_ready_expiry_does_not_duplicate_dead_letter() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - let QueueSendOutcome::Sent { message_id } = queue_send( - &transaction, - source_target().namespace(), - 0, - &QueueSendRequest { - producer_id: [9; 16], - payload: b"expire".to_vec(), - available_at_ms: 0, - }, - ) - .unwrap() else { - panic!("first queue send must insert") - }; - transaction - .execute( - "UPDATE queue_messages SET attempt = 20 WHERE message_id = ?1", - [message_id.as_slice()], - ) - .unwrap(); - let mut effects = EffectBatch::new(&transaction, &source_target(), 1, 1).unwrap(); - let mut dead_letter = QueueDeadLetterWriter::new(dead_letter_target(1), &mut effects); - assert_eq!( - queue_expire_ready_bounded_with_dead_letter(&transaction, 1, 128, Some(&mut dead_letter),) - .unwrap(), - 1 - ); - assert_eq!( - queue_expire_ready_bounded_with_dead_letter(&transaction, 1, 128, Some(&mut dead_letter),) - .unwrap(), - 0 - ); - assert_eq!( - transaction - .query_row("SELECT count(*) FROM sys_effects", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 1 - ); - transaction.commit().unwrap(); -} - -#[test] -fn failed_dead_letter_insert_rolls_back_queue_transition() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - let mut effects = EffectBatch::new(&transaction, &source_target(), 1, 10).unwrap(); - let mut dead_letter = QueueDeadLetterWriter::new(dead_letter_target(0), &mut effects); - assert!( - queue_apply_lease_with_dead_letter( - &transaction, - 10, - message_id, - [8; 16], - QueueLeaseAction::Retry { delay_ms: 0 }, - Some(&mut dead_letter), - ) - .is_err() - ); - drop(transaction); - let state: i64 = connection - .query_row( - "SELECT state FROM queue_messages WHERE message_id = ?1", - [message_id.as_slice()], - |row| row.get(0), - ) - .unwrap(); - let effects: i64 = connection - .query_row("SELECT count(*) FROM sys_effects", [], |row| row.get(0)) - .unwrap(); - assert_eq!((state, effects), (1, 0)); -} - -#[test] -fn pause_blocks_claims_but_preserves_messages_for_resume() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - queue_send( - &transaction, - source_target().namespace(), - 0, - &QueueSendRequest { - producer_id: [10; 16], - payload: b"work".to_vec(), - available_at_ms: 0, - }, - ) - .unwrap(); - assert_eq!( - queue_control(&transaction, 1, QueueControlAction::Pause).unwrap(), - QueueControlOutcome::Paused { generation: 1 } - ); - assert!( - queue_claim(&transaction, 1, 1, 5_000, &mut SystemQueueTokens) - .unwrap() - .is_empty() - ); - let info = queue_info(&transaction).unwrap(); - assert!(info.paused); - assert_eq!(info.ready, 1); - assert_eq!( - queue_control(&transaction, 2, QueueControlAction::Resume).unwrap(), - QueueControlOutcome::Resumed { generation: 2 } - ); - assert_eq!( - queue_claim(&transaction, 2, 1, 5_000, &mut SystemQueueTokens) - .unwrap() - .len(), - 1 - ); -} - -#[test] -fn purge_never_deletes_a_live_lease() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - assert_eq!( - queue_control(&transaction, 1, QueueControlAction::Purge { limit: 128 },).unwrap(), - QueueControlOutcome::Purged { messages: 0 } - ); - assert_eq!( - transaction - .query_row( - "SELECT count(*) FROM queue_messages WHERE message_id = ?1", - [message_id.as_slice()], - |row| row.get::<_, i64>(0), - ) - .unwrap(), - 1 - ); -} - -#[test] -fn redrive_waits_for_a_terminal_dead_letter_effect() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - let mut effects = EffectBatch::new(&transaction, &source_target(), 1, 10).unwrap(); - let mut dead_letter = QueueDeadLetterWriter::new(dead_letter_target(1), &mut effects); - queue_apply_lease_with_dead_letter( - &transaction, - 10, - message_id, - [8; 16], - QueueLeaseAction::Retry { delay_ms: 0 }, - Some(&mut dead_letter), - ) - .unwrap(); - assert_eq!( - queue_control(&transaction, 11, QueueControlAction::Redrive { limit: 1 },).unwrap(), - QueueControlOutcome::Redriven { messages: 0 } - ); - transaction - .execute( - "UPDATE sys_effects SET state = 3 WHERE effect_id = (SELECT dead_letter_effect_id FROM queue_messages WHERE message_id = ?1)", - [message_id.as_slice()], - ) - .unwrap(); - assert_eq!( - queue_control(&transaction, 12, QueueControlAction::Redrive { limit: 1 },).unwrap(), - QueueControlOutcome::Redriven { messages: 1 } - ); - assert_eq!( - transaction - .query_row( - "SELECT state, attempt FROM queue_messages WHERE message_id = ?1", - [message_id.as_slice()], - |row| Ok((row.get::<_, i64>(0)?, row.get::<_, i64>(1)?)), - ) - .unwrap(), - (0, 0) - ); -} - -#[test] -fn send_accepts_a_payload_at_the_documented_limit() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - let outcome = queue_send( - &transaction, - source_target().namespace(), - 0, - &QueueSendRequest { - producer_id: [9; 16], - payload: vec![b'p'; MAX_PAYLOAD_BYTES], - available_at_ms: 0, - }, - ) - .unwrap(); - assert!(matches!(outcome, QueueSendOutcome::Sent { .. })); -} - -#[test] -fn send_rejects_a_payload_past_the_documented_limit() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - let outcome = queue_send( - &transaction, - source_target().namespace(), - 0, - &QueueSendRequest { - producer_id: [10; 16], - payload: vec![b'p'; MAX_PAYLOAD_BYTES + 1], - available_at_ms: 0, - }, - ); - assert!(outcome.is_err()); -} - -#[test] -fn send_retains_a_message_for_thirty_days() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - let QueueSendOutcome::Sent { message_id } = queue_send( - &transaction, - source_target().namespace(), - 0, - &QueueSendRequest { - producer_id: [11; 16], - payload: b"payload".to_vec(), - available_at_ms: 0, - }, - ) - .unwrap() else { - panic!("first send must insert"); - }; - let expires_at_ms = transaction - .query_row( - "SELECT expires_at_ms FROM queue_messages WHERE message_id = ?1", - [message_id.as_slice()], - |row| row.get::<_, i64>(0), - ) - .unwrap(); - assert_eq!(expires_at_ms, RETENTION_MS); -} - -#[test] -fn extend_rejects_bounds_outside_five_to_three_hundred_seconds() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - assert!( - queue_apply_lease( - &transaction, - 10, - message_id, - [8; 16], - QueueLeaseAction::Extend { - extension_ms: 4_999 - }, - ) - .is_err() - ); - assert!( - queue_apply_lease( - &transaction, - 10, - message_id, - [8; 16], - QueueLeaseAction::Extend { - extension_ms: 300_001, - }, - ) - .is_err() - ); -} - -#[test] -fn retry_rejects_a_delay_past_one_hour() { - let mut connection = connection(); - let message_id = insert_leased_message(&mut connection); - let transaction = connection.transaction().unwrap(); - assert!( - queue_apply_lease( - &transaction, - 10, - message_id, - [8; 16], - QueueLeaseAction::Retry { - delay_ms: MAX_RETRY_DELAY_MS + 1, - }, - ) - .is_err() - ); -} diff --git a/crates/crab-cell-runtime/src/primitives/sql.rs b/crates/crab-cell-runtime/src/primitives/sql.rs deleted file mode 100644 index f373af4eb..000000000 --- a/crates/crab-cell-runtime/src/primitives/sql.rs +++ /dev/null @@ -1,443 +0,0 @@ -//! SQL primitive: typed statement batches over the Cell database. -use rusqlite::{ - Connection, Transaction, - hooks::{AuthAction, AuthContext, Authorization}, - params_from_iter, - types::{Value, ValueRef}, -}; - -use crate::{Error, Result}; - -mod api; - -pub use api::{SqlBatchCommand, SqlBatchQuery, SqlCell, SqlModule, register_sql}; - -const MAX_STATEMENTS: usize = 128; -const MAX_ROWS: usize = 1_000; -const MAX_PARAMETERS: usize = 32_766; -const MAX_OPERATION_BYTES: usize = 1 << 20; -const MAX_RESULT_BYTES: usize = 1 << 20; - -/// One typed SQLite parameter or result value. -#[derive(Clone, Debug, PartialEq)] -pub enum SqlValue { - /// SQL NULL. - Null, - /// A 64-bit signed integer. - Integer(i64), - /// A floating-point value. - Real(f64), - /// UTF-8 text. - Text(String), - /// Binary blob. - Blob(Vec), -} - -/// One parameterized application SQL statement. -#[derive(Clone, Debug, PartialEq)] -pub struct SqlStatement { - /// SQL text with `?` placeholders. - pub sql: String, - /// Values bound to the placeholders, in order. - pub parameters: Vec, -} - -/// A bounded group of application SQL statements executed in order. -#[derive(Clone, Debug, PartialEq)] -pub struct SqlBatch { - /// Statements executed in order. - pub statements: Vec, -} - -/// Materialized result of one SQL statement. -#[derive(Clone, Debug, PartialEq)] -pub struct SqlResultSet { - /// Column names in selection order. - pub columns: Vec, - /// Result rows. - pub rows: Vec>, - /// Rows the statement changed. - pub rows_affected: u64, -} - -#[derive(Clone, Copy)] -enum AccessMode { - ReadOnly, - ReadWrite, -} - -/// Executes a bounded application batch inside the caller's command transaction. -/// -/// Runtime and primitive tables, schema changes, connection configuration and -/// transaction control are unavailable through this interface. Any failure -/// leaves rollback of the surrounding application savepoint to the runtime. -pub fn sql_batch(transaction: &Transaction<'_>, batch: &SqlBatch) -> Result> { - execute_batch(transaction, batch, AccessMode::ReadWrite) -} - -/// Executes and materializes a bounded read-only application batch. -pub fn sql_query_batch(connection: &Connection, batch: &SqlBatch) -> Result> { - execute_batch(connection, batch, AccessMode::ReadOnly) -} - -pub(crate) fn write_blob( - transaction: &Transaction<'_>, - table: &str, - column: &str, - row_id: i64, - offset: usize, - bytes: &[u8], -) -> Result<()> { - // SQLite incremental I/O bypasses the SQL authorizer. Check the literal - // table name here; no raw handle or database selector escapes this call. - if is_protected_name(table) || starts_with_ignore_ascii_case(table, "sqlite_") { - return Err(Error::Command("SQL blob targets a protected table")); - } - let mut size = 16; - for len in [table.len(), column.len(), bytes.len()] { - add_size( - &mut size, - encoded_bytes(len), - MAX_OPERATION_BYTES, - "SQL blob write exceeds 1 MiB", - )?; - } - let mut blob = - transaction.blob_open(rusqlite::DatabaseName::Main, table, column, row_id, false)?; - let written = blob.write_at(bytes, offset); - let closed = blob.close(); - written?; - closed?; - Ok(()) -} - -fn execute_batch( - connection: &Connection, - batch: &SqlBatch, - mode: AccessMode, -) -> Result> { - validate_batch(batch)?; - let _authorizer = AuthorizerGuard::install(connection, mode); - let mut result_sets = Vec::with_capacity(batch.statements.len()); - let mut row_count = 0_usize; - let mut result_bytes = 0_usize; - - for item in &batch.statements { - let values = item - .parameters - .iter() - .map(to_sqlite_value) - .collect::>>()?; - let mut statement = connection.prepare(&item.sql)?; - if statement.parameter_count() != values.len() { - return Err(Error::Command( - "SQL parameter count does not match statement", - )); - } - let readonly = statement.readonly(); - if matches!(mode, AccessMode::ReadOnly) && !readonly { - return Err(Error::Command("SQL query batch contains a mutation")); - } - - if !readonly { - if statement.column_count() != 0 { - return Err(Error::Command("mutating SQL cannot use RETURNING")); - } - drop(statement); - let changed = connection.execute(&item.sql, params_from_iter(values))?; - let rows_affected = u64::try_from(changed) - .map_err(|_| Error::Command("SQL affected-row count overflow"))?; - add_size( - &mut result_bytes, - 8, - MAX_RESULT_BYTES, - "SQL result exceeds 1 MiB", - )?; - result_sets.push(SqlResultSet { - columns: Vec::new(), - rows: Vec::new(), - rows_affected, - }); - continue; - } - - let column_count = statement.column_count(); - let columns = statement - .column_names() - .into_iter() - .map(str::to_owned) - .collect::>(); - for column in &columns { - add_size( - &mut result_bytes, - encoded_bytes(column.len()), - MAX_RESULT_BYTES, - "SQL result exceeds 1 MiB", - )?; - } - - let mut rows = statement.query(params_from_iter(values))?; - let mut materialized = Vec::new(); - while let Some(row) = rows.next()? { - row_count = row_count - .checked_add(1) - .ok_or(Error::Command("SQL result row count overflow"))?; - if row_count > MAX_ROWS { - return Err(Error::Command("SQL result exceeds 1000 rows")); - } - let mut values = Vec::with_capacity(column_count); - for column in 0..column_count { - let value = from_sqlite_value(row.get_ref(column)?)?; - add_size( - &mut result_bytes, - value.encoded_len()?, - MAX_RESULT_BYTES, - "SQL result exceeds 1 MiB", - )?; - values.push(value); - } - materialized.push(values); - } - result_sets.push(SqlResultSet { - columns, - rows: materialized, - rows_affected: 0, - }); - } - Ok(result_sets) -} - -fn validate_batch(batch: &SqlBatch) -> Result<()> { - if !(1..=MAX_STATEMENTS).contains(&batch.statements.len()) { - return Err(Error::Command("SQL batch must contain 1..=128 statements")); - } - let mut bytes = 0_usize; - for statement in &batch.statements { - if statement.sql.trim().is_empty() { - return Err(Error::Command("SQL statement cannot be empty")); - } - if statement.parameters.len() > MAX_PARAMETERS { - return Err(Error::Command( - "SQL statement exceeds SQLite variable limit", - )); - } - if has_unquoted_semicolon(&statement.sql) { - return Err(Error::Command( - "SQL statement cannot contain a statement separator", - )); - } - add_size( - &mut bytes, - encoded_bytes(statement.sql.len()), - MAX_OPERATION_BYTES, - "SQL batch exceeds 1 MiB", - )?; - for value in &statement.parameters { - add_size( - &mut bytes, - value.encoded_len()?, - MAX_OPERATION_BYTES, - "SQL batch exceeds 1 MiB", - )?; - } - } - Ok(()) -} - -impl SqlValue { - fn encoded_len(&self) -> Result { - match self { - Self::Null => Ok(1), - Self::Integer(_) | Self::Real(_) => Ok(9), - Self::Text(value) => Ok(encoded_bytes(value.len())), - Self::Blob(value) => Ok(encoded_bytes(value.len())), - } - } -} - -fn encoded_bytes(payload: usize) -> usize { - payload.saturating_add(5) -} - -fn add_size( - current: &mut usize, - added: usize, - maximum: usize, - message: &'static str, -) -> Result<()> { - *current = current - .checked_add(added) - .filter(|total| *total <= maximum) - .ok_or(Error::Command(message))?; - Ok(()) -} - -fn to_sqlite_value(value: &SqlValue) -> Result { - Ok(match value { - SqlValue::Null => Value::Null, - SqlValue::Integer(value) => Value::Integer(*value), - SqlValue::Real(value) if value.is_finite() => Value::Real(*value), - SqlValue::Real(_) => return Err(Error::Command("SQL real parameter must be finite")), - SqlValue::Text(value) => Value::Text(value.clone()), - SqlValue::Blob(value) => Value::Blob(value.clone()), - }) -} - -fn from_sqlite_value(value: ValueRef<'_>) -> Result { - Ok(match value { - ValueRef::Null => SqlValue::Null, - ValueRef::Integer(value) => SqlValue::Integer(value), - ValueRef::Real(value) if value.is_finite() => SqlValue::Real(value), - ValueRef::Real(_) => return Err(Error::Command("SQL result contains a non-finite real")), - ValueRef::Text(value) => SqlValue::Text(std::str::from_utf8(value)?.to_owned()), - ValueRef::Blob(value) => SqlValue::Blob(value.to_vec()), - }) -} - -struct AuthorizerGuard<'a> { - connection: &'a Connection, -} - -impl<'a> AuthorizerGuard<'a> { - fn install(connection: &'a Connection, mode: AccessMode) -> Self { - let authorizer: fn(AuthContext<'_>) -> Authorization = match mode { - AccessMode::ReadOnly => authorize_read, - AccessMode::ReadWrite => authorize_write, - }; - connection.authorizer(Some(authorizer)); - Self { connection } - } -} - -impl Drop for AuthorizerGuard<'_> { - fn drop(&mut self) { - self.connection - .authorizer(None::) -> Authorization>); - } -} - -fn authorize_read(context: AuthContext<'_>) -> Authorization { - authorize(context, AccessMode::ReadOnly) -} - -fn authorize_write(context: AuthContext<'_>) -> Authorization { - authorize(context, AccessMode::ReadWrite) -} - -fn authorize(context: AuthContext<'_>, mode: AccessMode) -> Authorization { - if context.database_name.is_some_and(|name| name != "main") { - return Authorization::Deny; - } - if context.accessor.is_some_and(is_protected_name) { - return Authorization::Deny; - } - - match context.action { - AuthAction::Select | AuthAction::Recursive => Authorization::Allow, - AuthAction::Read { table_name, .. } if !is_protected_name(table_name) => { - Authorization::Allow - } - AuthAction::Insert { table_name } - | AuthAction::Delete { table_name } - | AuthAction::Update { table_name, .. } - if matches!(mode, AccessMode::ReadWrite) && !is_protected_name(table_name) => - { - Authorization::Allow - } - AuthAction::Function { function_name } - if !function_name.eq_ignore_ascii_case("load_extension") => - { - Authorization::Allow - } - _ => Authorization::Deny, - } -} - -fn is_protected_name(name: &str) -> bool { - [ - "sys_", - "kv_", - "queue_", - "workflow_", - "blob_", - "cron_", - "capacity_", - ] - .iter() - .any(|prefix| starts_with_ignore_ascii_case(name, prefix)) -} - -fn starts_with_ignore_ascii_case(value: &str, prefix: &str) -> bool { - value - .get(..prefix.len()) - .is_some_and(|candidate| candidate.eq_ignore_ascii_case(prefix)) -} - -fn has_unquoted_semicolon(sql: &str) -> bool { - #[derive(Clone, Copy)] - enum State { - Normal, - Single, - Double, - Backtick, - Bracket, - LineComment, - BlockComment, - } - - let bytes = sql.as_bytes(); - let mut state = State::Normal; - let mut index = 0_usize; - while index < bytes.len() { - let current = bytes[index]; - let next = bytes.get(index + 1).copied(); - match state { - State::Normal => match (current, next) { - (b';', _) => return true, - (b'\'', _) => state = State::Single, - (b'"', _) => state = State::Double, - (b'`', _) => state = State::Backtick, - (b'[', _) => state = State::Bracket, - (b'-', Some(b'-')) => { - state = State::LineComment; - index += 1; - } - (b'/', Some(b'*')) => { - state = State::BlockComment; - index += 1; - } - _ => {} - }, - State::Single if current == b'\'' => { - if next == Some(b'\'') { - index += 1; - } else { - state = State::Normal; - } - } - State::Double if current == b'"' => { - if next == Some(b'"') { - index += 1; - } else { - state = State::Normal; - } - } - State::Backtick if current == b'`' => { - if next == Some(b'`') { - index += 1; - } else { - state = State::Normal; - } - } - State::Bracket if current == b']' => state = State::Normal, - State::LineComment if current == b'\n' || current == b'\r' => state = State::Normal, - State::BlockComment if current == b'*' && next == Some(b'/') => { - state = State::Normal; - index += 1; - } - _ => {} - } - index += 1; - } - false -} diff --git a/crates/crab-cell-runtime/src/primitives/sql/api.rs b/crates/crab-cell-runtime/src/primitives/sql/api.rs deleted file mode 100644 index 8c11fa33b..000000000 --- a/crates/crab-cell-runtime/src/primitives/sql/api.rs +++ /dev/null @@ -1,334 +0,0 @@ -use std::marker::PhantomData; - -use crate::cell::catalog::CatalogRole; -use crate::client::{CellClient, Committed, InvocationError, Observed, Receipt}; -use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; -use crate::identity::CellTarget; -use crate::registry::{Command, Query, RegistryBuilder}; -use crate::registry::{CommandContext, CommandResult, QueryContext}; - -use super::{ - MAX_PARAMETERS, MAX_ROWS, MAX_STATEMENTS, SqlBatch, SqlResultSet, SqlStatement, SqlValue, -}; - -const NULL_TAG: u8 = 0; -const INTEGER_TAG: u8 = 1; -const REAL_TAG: u8 = 2; -const TEXT_TAG: u8 = 3; -const BLOB_TAG: u8 = 4; -const MAX_COLUMNS: usize = 2_000; - -/// Compile-time operation identifiers for one native SQL module. -pub trait SqlModule: Send + Sync + 'static { - /// Module the SQL surfaces register under. - const MODULE: &'static str; - /// Codec version of the SQL batch surfaces. - const CODEC_VERSION: u32 = 1; - /// Command id that executes a read-write batch. - const BATCH_COMMAND_ID: u32; - /// Query id that runs a read-only batch. - const BATCH_QUERY_ID: u32; -} - -/// Registers the typed read-write and read-only SQL bindings for one module. -pub fn register_sql(registry: &mut RegistryBuilder) -> crate::Result<()> { - registry.bind_command::>()?; - registry.bind_query::>() -} - -/// Typed read-write SQL batch bound to immutable module operation IDs. -pub struct SqlBatchCommand(PhantomData M>); - -impl Command for SqlBatchCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::BATCH_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = SqlBatch; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - Ok(CommandResult::Success(context.sql(&input)?)) - } -} - -/// Typed read-only SQL batch bound to immutable module operation IDs. -pub struct SqlBatchQuery(PhantomData M>); - -impl Query for SqlBatchQuery { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::BATCH_QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = SqlBatch; - type Output = Vec; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - context.sql(&input) - } -} - -/// Authorized native SQL capability for one explicit-key Cell. -#[derive(Clone)] -pub struct SqlCell { - client: CellClient, - target: CellTarget, - module: PhantomData M>, -} - -impl SqlCell { - /// Creates a SQL capability after validating its compiled namespace role. - pub fn new(client: CellClient, target: CellTarget) -> crate::Result { - let _ = client.require_namespace(target.namespace(), M::MODULE, CatalogRole::Sql)?; - Ok(Self { - client, - target, - module: PhantomData, - }) - } - - /// Executes one bounded parameterized batch and publishes its receipt. - pub async fn batch( - &self, - identity: crate::cell::executor::MutationIdentity, - batch: SqlBatch, - ) -> std::result::Result>, InvocationError>> { - self.client - .command::>(&self.target, identity, batch) - .await - } - - /// Executes one bounded read-only batch at an optional minimum receipt. - pub async fn query( - &self, - minimum: Option, - batch: SqlBatch, - ) -> std::result::Result>, InvocationError>> { - self.client - .query::>(&self.target, minimum, batch) - .await - } -} - -impl WireValue for SqlValue { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Null => encoder.write_u8(NULL_TAG), - Self::Integer(value) => { - encoder.write_u8(INTEGER_TAG)?; - encoder.write_i64(*value) - } - Self::Real(value) => { - encoder.write_u8(REAL_TAG)?; - encoder.write_f64(*value) - } - Self::Text(value) => { - encoder.write_u8(TEXT_TAG)?; - encoder.write_text(value) - } - Self::Blob(value) => { - encoder.write_u8(BLOB_TAG)?; - encoder.write_bytes(value) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - NULL_TAG => Ok(Self::Null), - INTEGER_TAG => Ok(Self::Integer(decoder.read_i64()?)), - REAL_TAG => Ok(Self::Real(decoder.read_f64()?)), - TEXT_TAG => Ok(Self::Text(decoder.read_text()?.to_owned())), - BLOB_TAG => Ok(Self::Blob(decoder.read_bytes()?.to_vec())), - _ => Err(CodecError::Invalid("invalid SQL value tag")), - } - } -} - -impl WireValue for SqlStatement { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_text(&self.sql)?; - encoder.write_count(self.parameters.len())?; - for parameter in &self.parameters { - parameter.encode(encoder)?; - } - Ok(()) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let sql = decoder.read_text()?.to_owned(); - let count = bounded_count(decoder, MAX_PARAMETERS, "too many SQL parameters")?; - let mut parameters = Vec::with_capacity(count.min(64)); - for _ in 0..count { - parameters.push(SqlValue::decode(decoder)?); - } - Ok(Self { sql, parameters }) - } -} - -impl WireValue for SqlBatch { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_count(self.statements.len())?; - for statement in &self.statements { - statement.encode(encoder)?; - } - Ok(()) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let count = bounded_count(decoder, MAX_STATEMENTS, "too many SQL statements")?; - let mut statements = Vec::with_capacity(count); - for _ in 0..count { - statements.push(SqlStatement::decode(decoder)?); - } - Ok(Self { statements }) - } -} - -impl WireValue for Vec { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_count(self.len())?; - for result in self { - encode_result(result, encoder)?; - } - Ok(()) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let count = bounded_count(decoder, MAX_STATEMENTS, "too many SQL result sets")?; - let mut results = Vec::with_capacity(count); - let mut remaining_rows = MAX_ROWS; - for _ in 0..count { - let result = decode_result(decoder, remaining_rows)?; - remaining_rows = remaining_rows - .checked_sub(result.rows.len()) - .ok_or(CodecError::Invalid("too many SQL result rows"))?; - results.push(result); - } - Ok(results) - } -} - -fn encode_result(result: &SqlResultSet, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_count(result.columns.len())?; - for column in &result.columns { - encoder.write_text(column)?; - } - encoder.write_count(result.rows.len())?; - for row in &result.rows { - encoder.write_count(row.len())?; - for value in row { - value.encode(encoder)?; - } - } - encoder.write_u64(result.rows_affected) -} - -fn decode_result( - decoder: &mut BoundedDecoder<'_>, - remaining_rows: usize, -) -> Result { - let column_count = bounded_count(decoder, MAX_COLUMNS, "too many SQL result columns")?; - let mut columns = Vec::with_capacity(column_count); - for _ in 0..column_count { - columns.push(decoder.read_text()?.to_owned()); - } - let row_count = bounded_count(decoder, remaining_rows, "too many SQL result rows")?; - let mut rows = Vec::with_capacity(row_count.min(64)); - for _ in 0..row_count { - let value_count = bounded_count(decoder, MAX_COLUMNS, "too many SQL row values")?; - if value_count != column_count { - return Err(CodecError::Invalid("SQL row width does not match columns")); - } - let mut row = Vec::with_capacity(value_count); - for _ in 0..value_count { - row.push(SqlValue::decode(decoder)?); - } - rows.push(row); - } - let rows_affected = decoder.read_u64()?; - if (columns.is_empty() && !rows.is_empty()) || (!columns.is_empty() && rows_affected != 0) { - return Err(CodecError::Invalid("inconsistent SQL result shape")); - } - Ok(SqlResultSet { - columns, - rows, - rows_affected, - }) -} - -fn bounded_count( - decoder: &mut BoundedDecoder<'_>, - maximum: usize, - message: &'static str, -) -> Result { - let count = decoder.read_count()?; - if count > maximum { - return Err(CodecError::Invalid(message)); - } - Ok(count) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::codec::roundtrip; - - #[test] - fn sql_batch_and_results_roundtrip_every_value_kind() { - roundtrip(SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT ?1, ?2, ?3, ?4, ?5".into(), - parameters: vec![ - SqlValue::Null, - SqlValue::Integer(i64::MIN), - SqlValue::Real(1.5), - SqlValue::Text("crab".into()), - SqlValue::Blob(vec![0, 255]), - ], - }], - }); - roundtrip(vec![SqlResultSet { - columns: vec!["value".into()], - rows: vec![vec![SqlValue::Integer(i64::MAX)]], - rows_affected: 0, - }]); - } - - #[test] - fn sql_decoder_rejects_counts_and_inconsistent_rows_before_allocation() { - let mut too_many = BoundedEncoder::new(16).unwrap(); - too_many.write_count(MAX_STATEMENTS + 1).unwrap(); - let bytes = too_many.finish(); - let mut decoder = BoundedDecoder::new(&bytes, 16).unwrap(); - assert!(matches!( - SqlBatch::decode(&mut decoder), - Err(CodecError::Invalid("too many SQL statements")) - )); - - let mut too_many_parameters = BoundedEncoder::new(32).unwrap(); - too_many_parameters.write_text("SELECT 1").unwrap(); - too_many_parameters.write_count(MAX_PARAMETERS + 1).unwrap(); - let bytes = too_many_parameters.finish(); - let mut decoder = BoundedDecoder::new(&bytes, 32).unwrap(); - assert!(matches!( - SqlStatement::decode(&mut decoder), - Err(CodecError::Invalid("too many SQL parameters")) - )); - - let mut inconsistent = BoundedEncoder::new(64).unwrap(); - inconsistent.write_count(1).unwrap(); - inconsistent.write_count(1).unwrap(); - inconsistent.write_text("one").unwrap(); - inconsistent.write_count(1).unwrap(); - inconsistent.write_count(0).unwrap(); - inconsistent.write_u64(0).unwrap(); - let bytes = inconsistent.finish(); - let mut decoder = BoundedDecoder::new(&bytes, 64).unwrap(); - assert!(matches!( - Vec::::decode(&mut decoder), - Err(CodecError::Invalid("SQL row width does not match columns")) - )); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/workflow.rs b/crates/crab-cell-runtime/src/primitives/workflow.rs deleted file mode 100644 index 2af3dc051..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow.rs +++ /dev/null @@ -1,806 +0,0 @@ -//! Workflow primitive: runs, timers, activities, signals, and controls. -use rusqlite::{Connection, OptionalExtension, Transaction}; - -use crate::identity::RequestId; -use crate::identity::{CellTarget, Digest, NamespaceId}; -use crate::primitives::effects::EffectBatch; -use crate::primitives::effects::EffectCommandIntent; -use crate::primitives::effects::validate_effect_command_intent; -use crate::{Error, Result}; - -mod activity; -mod activity_api; -mod activity_codec; -mod api; -mod commands; -mod maintenance; - -pub use commands::*; -use maintenance::next_effect_batch; -pub use maintenance::workflow_fire_timer; - -pub(crate) use maintenance::{workflow_fail_one_expired_activity, workflow_fire_one_due_timer}; - -pub use activity::{ - ActivityClaim, ActivityCompletion, ActivityCompletionOutcome, ActivityLeaseOutcome, - ActivitySupport, ActivityTokenSource, MAX_ACTIVITY_PAYLOAD_BYTES, SystemActivityTokens, - workflow_claim_activities, workflow_cleanup_terminal, workflow_complete_activity, - workflow_extend_activity, workflow_validate_activity_claim, -}; -pub(crate) use activity::{workflow_cleanup_terminal_bounded, workflow_reclaim_expired_bounded}; -pub use activity_api::{ - ActivityCancellation, ActivityContext, ActivityExecution, ActivityHandler, ActivityRunOutcome, - ActivitySupervisor, ActivitySupervisorError, BlockingActivityHandler, WorkflowActivities, - WorkflowActivityClaimCommand, WorkflowActivityClaimRequest, WorkflowActivityCompleteCommand, - WorkflowActivityExtendCommand, WorkflowActivityExtendRequest, WorkflowActivityModule, - WorkflowActivityValidateQuery, WorkflowActivityValidateRequest, register_activity, - register_blocking_activity, register_workflow_activities, -}; -pub use api::{ - WorkflowCancelCommand, WorkflowControlCommand, WorkflowGetQuery, WorkflowGetRequest, - WorkflowModule, WorkflowNamespace, WorkflowSignalCommand, WorkflowStartCommand, - register_workflow, -}; - -const WORKFLOW_SCHEMA: &str = include_str!("../migrations/workflow.sql"); -pub(crate) const WORKFLOW_TABLE: &str = "workflow_activities"; -const MAX_WORKFLOW_BYTES: usize = 1 << 20; -const MAX_ACTIONS: usize = 128; -const MAX_EVENTS_PER_CELL: u64 = 100_000; -const MAX_ACTIVITY_LIFETIME_MS: i64 = 7 * 24 * 60 * 60 * 1_000; - -/// Durable workflow run state. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum WorkflowStatus { - /// The run accepts events and schedules actions. - Running, - /// The run finished with a result. - Completed, - /// The run ended in failure. - Failed, - /// The run was cancelled. - Cancelled, - /// The run holds its state and accepts no events until resumed. - Paused, -} - -impl WorkflowStatus { - fn encode(self) -> i64 { - match self { - Self::Running => 0, - Self::Completed => 1, - Self::Failed => 2, - Self::Cancelled => 3, - Self::Paused => 4, - } - } - - fn decode(value: i64) -> Result { - match value { - 0 => Ok(Self::Running), - 1 => Ok(Self::Completed), - 2 => Ok(Self::Failed), - 3 => Ok(Self::Cancelled), - 4 => Ok(Self::Paused), - _ => Err(Error::Command("invalid stored workflow status")), - } - } -} - -/// One deterministic scheduling intention returned by a workflow definition. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum WorkflowAction { - /// Schedules one activity for the run. - Activity { - /// Registered activity type to schedule. - activity_type: String, - /// Deterministic input handed to the activity handler. - input: Vec, - /// Logical time the activity becomes claimable. - due_at_ms: i64, - /// Logical time after which the activity is abandoned. - expires_at_ms: i64, - }, - /// Wakes the run at a logical time. - Timer { - /// Logical time the run is woken. - due_at_ms: i64, - }, - /// Emits one effect for this transition. - Effect { - /// Effect intent the runtime submits. - intent: EffectCommandIntent, - }, -} - -/// Complete result of one pure workflow transition. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowDecision { - /// Run status after the transition. - pub status: WorkflowStatus, - /// Replacement definition-owned state. - pub state: Vec, - /// Terminal result, once the transition completed the run. - pub result: Option>, - /// Actions the transition scheduled, in ordinal order. - pub actions: Vec, -} - -/// Stable inputs available to a workflow definition. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowContext { - source: CellTarget, - run_id: [u8; 16], - event_sequence: u64, - now_ms: i64, -} - -impl WorkflowContext { - /// Returns the source Cell so effects can derive same-tenant application targets. - #[must_use] - pub const fn source(&self) -> &CellTarget { - &self.source - } - - /// Returns the run this context belongs to. - #[must_use] - pub const fn run_id(&self) -> [u8; 16] { - self.run_id - } - - /// Returns the sequence of the event being applied. - #[must_use] - pub const fn event_sequence(&self) -> u64 { - self.event_sequence - } - - /// Returns the logical time the transition runs at. - #[must_use] - pub const fn now_ms(&self) -> i64 { - self.now_ms - } - - /// Derives the ID that the action at `ordinal` will receive if committed. - #[must_use] - pub fn action_id(&self, ordinal: u32) -> [u8; 16] { - action_id(self.run_id, self.event_sequence, ordinal) - } -} - -/// A statically registered, deterministic workflow state machine. -pub trait WorkflowDefinition: Send + Sync + 'static { - /// Returns the digest that pins this definition's transition logic. - fn digest(&self) -> Digest; - - /// Declares every namespace this definition may target with an effect. - fn effect_targets(&self) -> &'static [NamespaceId] { - &[] - } - - /// Applies one event to `state` and returns the next decision. - fn transition( - &self, - state: &[u8], - event: &[u8], - context: WorkflowContext, - ) -> Result; -} - -/// Inputs that create one new workflow run. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowStart { - /// Caller-chosen workflow identity. - pub workflow_id: Vec, - /// Request identity that makes the start idempotent. - pub request_id: RequestId, - /// First event handed to the definition. - pub event: Vec, -} - -/// Idempotent external signal for one exact run. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowSignal { - /// Workflow the signal targets. - pub workflow_id: Vec, - /// Exact run the signal must match. - pub run_id: [u8; 16], - /// Signal identity that makes delivery idempotent. - pub signal_id: [u8; 16], - /// Event handed to the definition. - pub event: Vec, -} - -/// Business outcome of a workflow operation. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum WorkflowOutcome { - /// The event advanced the run. - Applied { - /// Run the event was applied to. - run_id: [u8; 16], - /// Run status after the event. - status: WorkflowStatus, - /// Sequence the applied event received. - event_sequence: u64, - }, - /// The same idempotency key was delivered before. - Duplicate { - /// Run the duplicate delivery belongs to. - run_id: [u8; 16], - /// Status the run holds now. - status: WorkflowStatus, - /// Sequence recorded for the original delivery. - event_sequence: u64, - }, - /// A workflow with this ID was started with different bytes. - AlreadyExists, - /// A workflow or run ID is already pinned to different identity bytes. - IdentityConflict, - /// The request targets a different run than the workflow's current one. - RunMismatch, - /// The run does not accept this operation in its current status. - NotRunning, - /// The requested action is not due yet. - NotDue, - /// Another transition currently holds this run. - Busy, -} - -/// Administrative transition for one exact workflow run. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowControl { - /// Workflow the control targets. - pub workflow_id: Vec, - /// Exact run the control must match. - pub run_id: [u8; 16], - /// Operator action to apply. - pub action: WorkflowControlAction, -} - -/// Durable workflow operator action. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum WorkflowControlAction { - /// Stops accepting events until the run is resumed. - Pause, - /// Returns a paused run to running. - Resume, - /// Starts a fresh run sequence for the workflow. - Restart { - /// Request identity that makes the restart idempotent. - request_id: RequestId, - /// Event handed to the restarted run. - event: Vec, - }, -} - -/// Materialized current state of one workflow run. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowRun { - /// Caller-chosen workflow identity. - pub workflow_id: Vec, - /// Current run identity. - pub run_id: [u8; 16], - /// Definition digest the run is pinned to. - pub definition_digest: Digest, - /// Current run status. - pub status: WorkflowStatus, - /// Definition-owned state bytes. - pub state: Vec, - /// Highest applied event sequence. - pub event_sequence: u64, - /// Terminal result, once the run completed. - pub result: Option>, -} - -fn exact_array(value: Vec, message: &'static str) -> Result<[u8; N]> { - value.try_into().map_err(|_| Error::Command(message)) -} - -/// Reads one bounded current workflow state without exposing primitive tables. -pub fn workflow_state(connection: &Connection, workflow_id: &[u8]) -> Result> { - validate_identifier(workflow_id)?; - let row = connection - .query_row( - "SELECT run_id, definition_digest, status, state, event_sequence, result FROM workflow_runs WHERE workflow_id = ?1", - [workflow_id], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, Vec>(3)?, - row.get::<_, i64>(4)?, - row.get::<_, Option>>(5)?, - )) - }, - ) - .optional()?; - let Some((run_id, definition_digest, status, state, event_sequence, result)) = row else { - return Ok(None); - }; - let run_id = run_id - .try_into() - .map_err(|_| Error::Command("invalid stored workflow run ID"))?; - let definition_digest = Digest::from_bytes( - definition_digest - .try_into() - .map_err(|_| Error::Command("invalid stored workflow definition digest"))?, - ); - let status = WorkflowStatus::decode(status)?; - let event_sequence = u64::try_from(event_sequence) - .map_err(|_| Error::Command("invalid stored workflow event sequence"))?; - let result_bytes = result.as_ref().map_or(0, Vec::len); - if state.len() > MAX_WORKFLOW_BYTES - || result_bytes > MAX_WORKFLOW_BYTES - || state.len().saturating_add(result_bytes) > MAX_WORKFLOW_BYTES - { - return Err(Error::Command("workflow state result exceeds 1 MiB")); - } - Ok(Some(WorkflowRun { - workflow_id: workflow_id.to_vec(), - run_id, - definition_digest, - status, - state, - event_sequence, - result, - })) -} - -#[derive(Clone)] -pub(super) struct StoredRun { - pub(super) run_id: [u8; 16], - pub(super) definition_digest: [u8; 32], - pub(super) status: WorkflowStatus, - pub(super) state: Vec, - pub(super) event_sequence: u64, -} - -fn load_run(transaction: &Transaction<'_>, workflow_id: &[u8]) -> Result> { - transaction - .query_row( - "SELECT run_id, definition_digest, status, state, event_sequence FROM workflow_runs WHERE workflow_id = ?1", - [workflow_id], - decode_run, - ) - .optional() - .map_err(Into::into) -} - -pub(super) fn workflow_definition_digest( - transaction: &Transaction<'_>, - workflow_id: &[u8], -) -> Result> { - Ok(load_run(transaction, workflow_id)?.map(|run| Digest::from_bytes(run.definition_digest))) -} - -pub(super) fn workflow_definition_digest_by_run( - transaction: &Transaction<'_>, - run_id: [u8; 16], -) -> Result> { - Ok(load_run_by_id(transaction, run_id)?.map(|run| Digest::from_bytes(run.definition_digest))) -} - -pub(super) fn load_run_by_id( - transaction: &Transaction<'_>, - run_id: [u8; 16], -) -> Result> { - transaction - .query_row( - "SELECT run_id, definition_digest, status, state, event_sequence FROM workflow_runs WHERE run_id = ?1", - [run_id.as_slice()], - decode_run, - ) - .optional() - .map_err(Into::into) -} - -fn decode_run(row: &rusqlite::Row<'_>) -> rusqlite::Result { - decode_run_offset(row, 0) -} - -fn decode_run_offset(row: &rusqlite::Row<'_>, offset: usize) -> rusqlite::Result { - let run_id = row.get::<_, Vec>(offset)?; - let definition_digest = row.get::<_, Vec>(offset + 1)?; - let status = row.get::<_, i64>(offset + 2)?; - let state = row.get::<_, Vec>(offset + 3)?; - let event_sequence = row.get::<_, i64>(offset + 4)?; - let run_id = run_id - .try_into() - .map_err(|_| invalid_data("workflow run ID"))?; - let definition_digest = definition_digest - .try_into() - .map_err(|_| invalid_data("workflow definition digest"))?; - let status = WorkflowStatus::decode(status).map_err(|_| invalid_data("workflow status"))?; - if state.len() > MAX_WORKFLOW_BYTES || event_sequence < 0 { - return Err(invalid_data("workflow run")); - } - Ok(StoredRun { - run_id, - definition_digest, - status, - state, - event_sequence: event_sequence as u64, - }) -} - -fn invalid_data(field: &'static str) -> rusqlite::Error { - rusqlite::Error::FromSqlConversionFailure( - 0, - rusqlite::types::Type::Blob, - format!("invalid stored {field}").into(), - ) -} - -pub(super) fn verify_definition( - run: &StoredRun, - definition: &dyn WorkflowDefinition, -) -> Result<()> { - if run.definition_digest != *definition.digest().as_bytes() { - return Err(Error::Command("workflow definition digest is unavailable")); - } - Ok(()) -} - -pub(super) fn prepare_transition( - transaction: &Transaction<'_>, - source: &CellTarget, - run: &StoredRun, - definition: &dyn WorkflowDefinition, - event: &[u8], - now_ms: i64, -) -> Result<(u64, WorkflowDecision)> { - verify_definition(run, definition)?; - let sequence = next_sequence(transaction, run.event_sequence)?; - let context = WorkflowContext { - source: source.clone(), - run_id: run.run_id, - event_sequence: sequence, - now_ms, - }; - let decision = definition.transition(&run.state, event, context)?; - validate_decision(&decision, source, definition.effect_targets(), now_ms)?; - Ok((sequence, decision)) -} - -pub(super) fn commit_transition( - transaction: &Transaction<'_>, - effects: &mut EffectBatch, - run_id: [u8; 16], - sequence: u64, - id: [u8; 32], - event: &[u8], - now_ms: i64, - decision: WorkflowDecision, -) -> Result { - insert_event(transaction, run_id, sequence, id, event)?; - apply_decision(transaction, effects, run_id, sequence, now_ms, decision) -} - -pub(super) fn validate_decision( - decision: &WorkflowDecision, - source: &CellTarget, - effect_targets: &[NamespaceId], - now_ms: i64, -) -> Result<()> { - if decision.status == WorkflowStatus::Paused { - return Err(Error::Command( - "workflow definitions cannot return operator-only paused status", - )); - } - if decision.state.len() > MAX_WORKFLOW_BYTES - || decision - .result - .as_ref() - .is_some_and(|value| value.len() > MAX_WORKFLOW_BYTES) - || decision.actions.len() > MAX_ACTIONS - { - return Err(Error::Command("workflow decision exceeds limits")); - } - if decision.status == WorkflowStatus::Running && decision.result.is_some() { - return Err(Error::Command("running workflow cannot have a result")); - } - if decision.status != WorkflowStatus::Running - && decision - .actions - .iter() - .any(|action| !matches!(action, WorkflowAction::Effect { .. })) - { - return Err(Error::Command( - "terminal workflow cannot schedule local work", - )); - } - let mut bytes = 0_usize; - for action in &decision.actions { - match action { - WorkflowAction::Activity { - activity_type, - input, - due_at_ms, - expires_at_ms, - } => { - if activity_type.is_empty() - || activity_type.len() > 256 - || input.len() > MAX_ACTIVITY_PAYLOAD_BYTES - || *due_at_ms < now_ms - || *expires_at_ms <= now_ms - || *expires_at_ms < *due_at_ms - || *expires_at_ms > now_ms.saturating_add(MAX_ACTIVITY_LIFETIME_MS) - { - return Err(Error::Command("invalid workflow activity action")); - } - bytes = bytes - .checked_add(activity_type.len()) - .and_then(|value| value.checked_add(input.len())) - .and_then(|value| value.checked_add(32)) - .ok_or(Error::Command("workflow action byte count overflow"))?; - } - WorkflowAction::Timer { due_at_ms } => { - if *due_at_ms < now_ms { - return Err(Error::Command("workflow timer is in the past")); - } - bytes = bytes - .checked_add(24) - .ok_or(Error::Command("workflow action byte count overflow"))?; - } - WorkflowAction::Effect { intent } => { - if intent.target.tenant() != source.tenant() - || intent.target.application() != source.application() - || !effect_targets.contains(&intent.target.namespace()) - { - return Err(Error::Command("workflow effect target is not declared")); - } - validate_effect_command_intent(now_ms, intent)?; - bytes = bytes - .checked_add(intent.input.len()) - .and_then(|value| value.checked_add(intent.target.partition().len())) - .and_then(|value| value.checked_add(64)) - .ok_or(Error::Command("workflow action byte count overflow"))?; - } - } - } - if bytes > MAX_WORKFLOW_BYTES { - return Err(Error::Command("workflow actions exceed 1 MiB")); - } - Ok(()) -} - -fn apply_decision( - transaction: &Transaction<'_>, - effects: &mut EffectBatch, - run_id: [u8; 16], - sequence: u64, - now_ms: i64, - decision: WorkflowDecision, -) -> Result { - let existing: i64 = transaction.query_row( - "SELECT (SELECT count(*) FROM workflow_activities WHERE run_id = ?1 AND state IN (0, 1)) + (SELECT count(*) FROM workflow_timers WHERE run_id = ?1 AND state = 0)", - [run_id.as_slice()], - |row| row.get(0), - )?; - let local_actions = decision - .actions - .iter() - .filter(|action| !matches!(action, WorkflowAction::Effect { .. })) - .count(); - let prospective = usize::try_from(existing) - .ok() - .and_then(|value| value.checked_add(local_actions)) - .ok_or(Error::Command("invalid outstanding workflow task count"))?; - if prospective > MAX_ACTIONS { - return Err(Error::Command( - "workflow has more than 128 outstanding tasks", - )); - } - - for (ordinal, action) in decision.actions.iter().enumerate() { - let ordinal = u32::try_from(ordinal) - .map_err(|_| Error::Command("workflow action ordinal overflow"))?; - let id = action_id(run_id, sequence, ordinal); - match action { - WorkflowAction::Activity { - activity_type, - input, - due_at_ms, - expires_at_ms, - } => { - transaction.execute( - "INSERT INTO workflow_activities(run_id, activity_id, activity_type, input, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, completion_token, completion_digest, result) VALUES (?1, ?2, ?3, ?4, 0, 0, ?5, ?6, NULL, NULL, NULL, NULL, NULL)", - ( - run_id.as_slice(), - id.as_slice(), - activity_type, - input.as_slice(), - due_at_ms, - expires_at_ms, - ), - )?; - } - WorkflowAction::Timer { due_at_ms } => { - transaction.execute( - "INSERT INTO workflow_timers(run_id, timer_id, due_at_ms, state) VALUES (?1, ?2, ?3, 0)", - (run_id.as_slice(), id.as_slice(), due_at_ms), - )?; - } - WorkflowAction::Effect { intent } => { - effects.insert_command(transaction, intent)?; - } - } - } - - let completed_at_ms = (decision.status != WorkflowStatus::Running).then_some(now_ms); - if transaction.execute( - "UPDATE workflow_runs SET status = ?1, state = ?2, event_sequence = ?3, result = ?4, completed_at_ms = ?5 WHERE run_id = ?6 AND status = 0", - ( - decision.status.encode(), - decision.state.as_slice(), - sequence as i64, - decision.result.as_deref(), - completed_at_ms, - run_id.as_slice(), - ), - )? != 1 - { - return Err(Error::Command("workflow run changed during decision")); - } - if decision.status != WorkflowStatus::Running { - cancel_outstanding(transaction, run_id)?; - } - Ok(WorkflowOutcome::Applied { - run_id, - status: decision.status, - event_sequence: sequence, - }) -} - -fn cancel_outstanding(transaction: &Transaction<'_>, run_id: [u8; 16]) -> Result<()> { - transaction.execute( - "UPDATE workflow_activities SET state = 4, token = NULL, lease_until_ms = NULL WHERE run_id = ?1 AND state IN (0, 1)", - [run_id.as_slice()], - )?; - transaction.execute( - "UPDATE workflow_timers SET state = 2 WHERE run_id = ?1 AND state = 0", - [run_id.as_slice()], - )?; - Ok(()) -} - -fn next_sequence(transaction: &Transaction<'_>, current: u64) -> Result { - let total: i64 = transaction.query_row( - "SELECT event_count FROM workflow_control WHERE singleton = 1", - [], - |row| row.get(0), - )?; - if total < 0 || total as u64 >= MAX_EVENTS_PER_CELL { - return Err(Error::Capacity( - "workflow event history reached 100,000 rows", - )); - } - current - .checked_add(1) - .filter(|sequence| *sequence <= i64::MAX as u64) - .ok_or(Error::Command("workflow event sequence overflow")) -} - -/// Verifies Workflow's transactionally maintained event counter without repairing it. -pub fn verify_workflow_event_count(connection: &Connection) -> Result<()> { - let stored: i64 = connection.query_row( - "SELECT event_count FROM workflow_control WHERE singleton = 1", - [], - |row| row.get(0), - )?; - let computed: i64 = - connection.query_row("SELECT count(*) FROM workflow_events", [], |row| row.get(0))?; - if stored != computed { - return Err(Error::Command( - "workflow event counter does not match events", - )); - } - Ok(()) -} - -fn insert_event( - transaction: &Transaction<'_>, - run_id: [u8; 16], - sequence: u64, - id: [u8; 32], - event: &[u8], -) -> Result<()> { - transaction.execute( - "INSERT INTO workflow_events(run_id, sequence, event_id, event_digest, payload) VALUES (?1, ?2, ?3, ?4, ?5)", - ( - run_id.as_slice(), - sequence as i64, - id.as_slice(), - event_digest(event).as_slice(), - event, - ), - )?; - Ok(()) -} - -fn stored_event_digest( - transaction: &Transaction<'_>, - run_id: [u8; 16], - id: [u8; 32], -) -> Result> { - let digest = transaction - .query_row( - "SELECT event_digest FROM workflow_events WHERE run_id = ?1 AND event_id = ?2", - (run_id.as_slice(), id.as_slice()), - |row| row.get::<_, Vec>(0), - ) - .optional()?; - digest - .map(|value| { - value - .try_into() - .map_err(|_| Error::Command("invalid stored workflow event digest")) - }) - .transpose() -} - -fn run_id(namespace: NamespaceId, request_id: RequestId) -> [u8; 16] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.workflow-run.v1\0"); - hasher.update(namespace.as_bytes()); - hasher.update(request_id.as_bytes()); - let digest = hasher.finalize(); - let mut id = [0; 16]; - id.copy_from_slice(&digest.as_bytes()[..16]); - id -} - -fn action_id(run_id: [u8; 16], sequence: u64, ordinal: u32) -> [u8; 16] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.action.v1\0"); - hasher.update(&run_id); - hasher.update(&sequence.to_be_bytes()); - hasher.update(&ordinal.to_be_bytes()); - let digest = hasher.finalize(); - let mut id = [0; 16]; - id.copy_from_slice(&digest.as_bytes()[..16]); - id -} - -fn event_id(run_id: [u8; 16], source_id: &[u8; 16]) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(&run_id); - hasher.update(source_id); - *hasher.finalize().as_bytes() -} - -fn timer_event_id(run_id: [u8; 16], timer_id: [u8; 16]) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(&run_id); - hasher.update(&timer_id); - hasher.update(b"fired"); - *hasher.finalize().as_bytes() -} - -fn event_digest(event: &[u8]) -> [u8; 32] { - *blake3::hash(event).as_bytes() -} - -fn validate_identifier(identifier: &[u8]) -> Result<()> { - if identifier.is_empty() || identifier.len() > 1024 { - return Err(Error::Command("workflow ID must be in 1..=1024 bytes")); - } - Ok(()) -} - -fn validate_event(event: &[u8]) -> Result<()> { - if event.len() > MAX_WORKFLOW_BYTES { - return Err(Error::Command("workflow event exceeds 1 MiB")); - } - Ok(()) -} - -fn validate_now(now_ms: i64) -> Result<()> { - if now_ms < 0 { - return Err(Error::Command("workflow time must be non-negative")); - } - Ok(()) -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/primitives/workflow/activity.rs b/crates/crab-cell-runtime/src/primitives/workflow/activity.rs deleted file mode 100644 index 1c1866243..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow/activity.rs +++ /dev/null @@ -1,671 +0,0 @@ -use rand::RngCore; -use rusqlite::{Connection, OptionalExtension, Transaction, params_from_iter, types::Value}; - -use super::{ - WorkflowDefinition, WorkflowOutcome, WorkflowStatus, commit_transition, load_run_by_id, - prepare_transition, verify_definition, -}; -use crate::identity::Digest; -use crate::primitives::effects::EffectBatch; -use crate::{Error, Result}; - -pub(super) const MAX_CLAIM_ITEMS: usize = 32; -const MAX_CLAIM_BYTES: usize = 512 << 10; -const MAX_SCAN_ITEMS: usize = 128; -/// Maximum encoded input or output retained by one native activity attempt. -pub const MAX_ACTIVITY_PAYLOAD_BYTES: usize = 256 << 10; -const MAX_ACTIVITY_TYPE_BYTES: usize = 256; -pub(super) const MAX_ATTEMPTS: u32 = 20; -const MIN_LEASE_MS: u32 = 5_000; -pub(crate) const MAX_LEASE_MS: u32 = 300_000; -const DELIVERY_MARGIN_MS: i64 = 1_000; -const TERMINAL_RETENTION_MS: i64 = 30 * 24 * 60 * 60 * 1_000; - -/// Source of unpredictable activity lease tokens. -pub trait ActivityTokenSource { - /// Returns an unpredictable activity lease token. - fn next_token(&mut self) -> Result<[u8; 16]>; -} - -/// Cryptographically seeded process-local activity token source. -pub struct SystemActivityTokens; - -impl ActivityTokenSource for SystemActivityTokens { - fn next_token(&mut self) -> Result<[u8; 16]> { - let mut token = [0; 16]; - rand::rng().fill_bytes(&mut token); - Ok(token) - } -} - -/// One activity type and pinned workflow definition available on a worker. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct ActivitySupport { - /// Registered activity type. - pub activity_type: String, - /// Workflow definition the type belongs to. - pub definition_digest: Digest, -} - -/// A published activity lease that may be emitted to native Rust code. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct ActivityClaim { - /// Run the activity belongs to. - pub run_id: [u8; 16], - /// Activity identity within the run. - pub activity_id: [u8; 16], - /// Registered activity type. - pub activity_type: String, - /// Deterministic input for the handler. - pub input: Vec, - /// Workflow definition digest the claim is pinned to. - pub definition_digest: Digest, - /// Delivery attempt this lease represents. - pub attempt: u32, - /// Lease token required to extend or complete. - pub token: [u8; 16], - /// Logical time the lease expires. - pub lease_until_ms: i64, -} - -/// Idempotent completion or failure from one exact activity attempt. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct ActivityCompletion { - /// Run the activity belongs to. - pub run_id: [u8; 16], - /// Activity identity within the run. - pub activity_id: [u8; 16], - /// Attempt the completion settles. - pub attempt: u32, - /// Lease token the completion must match. - pub lease_token: [u8; 16], - /// Idempotency token for this completion. - pub completion_token: [u8; 16], - /// Result bytes to record. - pub result: Vec, - /// Whether the attempt failed. - pub failed: bool, - /// Whether a failed attempt may be retried. - pub retryable: bool, -} - -/// Business outcome of an activity lease extension. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ActivityLeaseOutcome { - /// The lease was extended to this logical time. - Extended { - /// New lease expiry. - lease_until_ms: i64, - }, - /// The token no longer matches the current lease. - LeaseLost, -} - -/// Business outcome of applying an activity completion. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum ActivityCompletionOutcome { - /// The completion advanced the workflow run. - Applied(WorkflowOutcome), - /// The failed attempt is retried. - Retrying { - /// Logical time the next attempt is due. - due_at_ms: i64, - }, - /// The same completion token was already applied. - Duplicate { - /// Result recorded for the original completion. - result: Vec, - }, - /// The completion named an activity other than the lease's. - IdentityConflict, - /// The token no longer matches the current lease. - LeaseLost, -} - -/// Reclaims expired leases and claims a bounded supported activity batch. -pub fn workflow_claim_activities( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, - lease_ms: u32, - supported: &[ActivitySupport], - tokens: &mut impl ActivityTokenSource, -) -> Result> { - validate_now(now_ms)?; - if !(1..=MAX_CLAIM_ITEMS).contains(&limit) { - return Err(Error::Command("activity claim limit must be in 1..=32")); - } - if !(MIN_LEASE_MS..=MAX_LEASE_MS).contains(&lease_ms) { - return Err(Error::Command("activity lease must be in 5..=300 seconds")); - } - validate_support(supported)?; - workflow_reclaim_expired_bounded(transaction, now_ms, MAX_SCAN_ITEMS)?; - - let mut sql = "SELECT a.run_id, a.activity_id, a.activity_type, a.input, r.definition_digest, a.attempt, a.expires_at_ms FROM workflow_activities a INDEXED BY activities_ready JOIN workflow_runs r ON r.run_id = a.run_id WHERE a.state = 0 AND a.due_at_ms <= ? AND a.expires_at_ms > ? AND a.attempt < ? AND r.status = 0 AND (a.activity_type, r.definition_digest) IN (".to_owned(); - for index in 0..supported.len() { - if index != 0 { - sql.push(','); - } - sql.push_str("(?, ?)"); - } - sql.push_str(") ORDER BY a.due_at_ms, a.run_id, a.activity_id LIMIT ?"); - let mut parameters = Vec::with_capacity(4 + supported.len() * 2); - parameters.extend([ - Value::Integer(now_ms), - Value::Integer(now_ms), - Value::Integer(i64::from(MAX_ATTEMPTS)), - ]); - for support in supported { - parameters.push(Value::Text(support.activity_type.clone())); - parameters.push(Value::Blob(support.definition_digest.as_bytes().to_vec())); - } - parameters.push(Value::Integer(MAX_SCAN_ITEMS as i64)); - let mut statement = transaction.prepare(&sql)?; - let rows = statement.query_map(params_from_iter(parameters.iter()), |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, String>(2)?, - row.get::<_, Vec>(3)?, - row.get::<_, Vec>(4)?, - row.get::<_, i64>(5)?, - row.get::<_, i64>(6)?, - )) - })?; - let mut candidates = Vec::with_capacity(limit); - let mut payload_bytes = 0_usize; - for row in rows { - if candidates.len() == limit { - break; - } - let (run_id, activity_id, activity_type, input, definition, attempt, expires_at_ms) = row?; - let run_id = exact::<16>(run_id, "workflow activity run ID")?; - let activity_id = exact::<16>(activity_id, "workflow activity ID")?; - let definition = Digest::from_bytes(exact::<32>( - definition, - "workflow activity definition digest", - )?); - if activity_type.is_empty() - || activity_type.len() > MAX_ACTIVITY_TYPE_BYTES - || input.len() > MAX_ACTIVITY_PAYLOAD_BYTES - || attempt < 0 - { - return Err(Error::Command("invalid stored workflow activity")); - } - let next_bytes = payload_bytes - .checked_add(activity_type.len()) - .and_then(|value| value.checked_add(input.len())) - .ok_or(Error::Command("activity claim byte count overflow"))?; - if next_bytes > MAX_CLAIM_BYTES { - break; - } - let attempt = u32::try_from(attempt) - .map_err(|_| Error::Command("invalid stored workflow activity attempt"))?; - payload_bytes = next_bytes; - candidates.push(( - run_id, - activity_id, - activity_type, - input, - definition, - attempt, - expires_at_ms, - )); - } - drop(statement); - - let requested_deadline = now_ms - .checked_add(i64::from(lease_ms)) - .ok_or(Error::Command("activity lease deadline overflow"))?; - let mut claimed = Vec::with_capacity(candidates.len()); - for (run_id, activity_id, activity_type, input, definition, attempt, expires_at_ms) in - candidates - { - let token = tokens.next_token()?; - let lease_until_ms = requested_deadline.min(expires_at_ms); - if transaction.execute( - "UPDATE workflow_activities SET state = 1, attempt = attempt + 1, token = ?1, lease_until_ms = ?2, completion_token = NULL, completion_digest = NULL, result = NULL WHERE run_id = ?3 AND activity_id = ?4 AND state = 0 AND due_at_ms <= ?5 AND expires_at_ms > ?5 AND attempt = ?6 AND EXISTS (SELECT 1 FROM workflow_runs r WHERE r.run_id = workflow_activities.run_id AND r.status = 0)", - ( - token.as_slice(), - lease_until_ms, - run_id.as_slice(), - activity_id.as_slice(), - now_ms, - i64::from(attempt), - ), - )? != 1 - { - return Err(Error::Command("activity claim lost selected ready row")); - } - claimed.push(ActivityClaim { - run_id, - activity_id, - activity_type, - input, - definition_digest: definition, - attempt: attempt + 1, - token, - lease_until_ms, - }); - } - Ok(claimed) -} - -/// Verifies that every claimed activity is still live after its root publishes. -pub fn workflow_validate_activity_claim( - connection: &Connection, - now_ms: i64, - claimed: &[ActivityClaim], -) -> Result { - validate_now(now_ms)?; - if claimed.len() > MAX_CLAIM_ITEMS { - return Err(Error::Command("activity claim exceeds 32 items")); - } - let minimum = now_ms - .checked_add(DELIVERY_MARGIN_MS) - .ok_or(Error::Command("activity delivery margin overflow"))?; - for activity in claimed { - if activity.input.len() > MAX_ACTIVITY_PAYLOAD_BYTES || activity.lease_until_ms < minimum { - return Ok(false); - } - let live = connection - .query_row( - "SELECT 1 FROM workflow_activities a JOIN workflow_runs r ON r.run_id = a.run_id WHERE a.run_id = ?1 AND a.activity_id = ?2 AND a.state = 1 AND a.attempt = ?3 AND a.token = ?4 AND a.lease_until_ms = ?5 AND a.lease_until_ms >= ?6 AND r.status = 0 AND r.definition_digest = ?7", - ( - activity.run_id.as_slice(), - activity.activity_id.as_slice(), - i64::from(activity.attempt), - activity.token.as_slice(), - activity.lease_until_ms, - minimum, - activity.definition_digest.as_bytes().as_slice(), - ), - |_| Ok(()), - ) - .optional()? - .is_some(); - if !live { - return Ok(false); - } - } - Ok(true) -} - -/// Extends one exact live activity lease without shortening it. -pub fn workflow_extend_activity( - transaction: &Transaction<'_>, - now_ms: i64, - claim: &ActivityClaim, - extension_ms: u32, -) -> Result { - validate_now(now_ms)?; - if !(MIN_LEASE_MS..=MAX_LEASE_MS).contains(&extension_ms) { - return Err(Error::Command( - "activity extension must be in 5..=300 seconds", - )); - } - let current = transaction - .query_row( - "SELECT a.lease_until_ms, a.expires_at_ms FROM workflow_activities a JOIN workflow_runs r ON r.run_id = a.run_id WHERE a.run_id = ?1 AND a.activity_id = ?2 AND a.state = 1 AND a.attempt = ?3 AND a.token = ?4 AND a.lease_until_ms > ?5 AND r.status = 0 AND r.definition_digest = ?6", - ( - claim.run_id.as_slice(), - claim.activity_id.as_slice(), - i64::from(claim.attempt), - claim.token.as_slice(), - now_ms, - claim.definition_digest.as_bytes().as_slice(), - ), - |row| Ok((row.get::<_, i64>(0)?, row.get::<_, i64>(1)?)), - ) - .optional()?; - let Some((current, expires_at_ms)) = current else { - return Ok(ActivityLeaseOutcome::LeaseLost); - }; - let requested = now_ms - .checked_add(i64::from(extension_ms)) - .ok_or(Error::Command("activity extension deadline overflow"))?; - let lease_until_ms = current.max(requested.min(expires_at_ms)); - if transaction.execute( - "UPDATE workflow_activities SET lease_until_ms = ?1 WHERE run_id = ?2 AND activity_id = ?3 AND state = 1 AND attempt = ?4 AND token = ?5 AND lease_until_ms = ?6", - ( - lease_until_ms, - claim.run_id.as_slice(), - claim.activity_id.as_slice(), - i64::from(claim.attempt), - claim.token.as_slice(), - current, - ), - )? != 1 - { - return Err(Error::Command("activity lease changed during serialized extension")); - } - Ok(ActivityLeaseOutcome::Extended { lease_until_ms }) -} - -/// Completes, fails or reschedules one exact activity attempt atomically. -pub fn workflow_complete_activity( - transaction: &Transaction<'_>, - source: &crate::CellTarget, - now_ms: i64, - completion: &ActivityCompletion, - definition: &dyn WorkflowDefinition, -) -> Result { - let mut effects = super::next_effect_batch(transaction, source, now_ms)?; - workflow_complete_activity_with_effects( - transaction, - &mut effects, - source, - now_ms, - completion, - definition, - ) -} - -pub(super) fn workflow_complete_activity_with_effects( - transaction: &Transaction<'_>, - effects: &mut EffectBatch, - source: &crate::CellTarget, - now_ms: i64, - completion: &ActivityCompletion, - definition: &dyn WorkflowDefinition, -) -> Result { - validate_now(now_ms)?; - if completion.result.len() > MAX_ACTIVITY_PAYLOAD_BYTES - || (!completion.failed && completion.retryable) - { - return Err(Error::Command("invalid activity completion")); - } - let Some(run) = load_run_by_id(transaction, completion.run_id)? else { - return Ok(ActivityCompletionOutcome::LeaseLost); - }; - verify_definition(&run, definition)?; - let stored = load_activity(transaction, completion.run_id, completion.activity_id)?; - let Some(stored) = stored else { - return Ok(ActivityCompletionOutcome::LeaseLost); - }; - if stored.attempt != completion.attempt { - return Ok(ActivityCompletionOutcome::LeaseLost); - } - let digest = completion_digest(completion); - if let Some(token) = stored.completion_token { - return if token == completion.completion_token && stored.completion_digest == Some(digest) { - Ok(ActivityCompletionOutcome::Duplicate { - result: stored.result, - }) - } else { - Ok(ActivityCompletionOutcome::IdentityConflict) - }; - } - if run.status != WorkflowStatus::Running - || stored.state != 1 - || stored.token != Some(completion.lease_token) - || stored - .lease_until_ms - .is_none_or(|deadline| deadline <= now_ms) - { - return Ok(ActivityCompletionOutcome::LeaseLost); - } - - if completion.failed && completion.retryable && completion.attempt < MAX_ATTEMPTS { - let due_at_ms = now_ms.saturating_add(retry_delay_ms(completion.attempt)); - if due_at_ms < stored.expires_at_ms { - update_completion(transaction, completion, &stored, 0, Some(due_at_ms), digest)?; - return Ok(ActivityCompletionOutcome::Retrying { due_at_ms }); - } - } - - let event = completion_event(completion); - let (sequence, decision) = - prepare_transition(transaction, source, &run, definition, &event, now_ms)?; - update_completion( - transaction, - completion, - &stored, - if completion.failed { 3 } else { 2 }, - None, - digest, - )?; - let outcome = commit_transition( - transaction, - effects, - completion.run_id, - sequence, - completion_event_id(completion), - &event, - now_ms, - decision, - )?; - Ok(ActivityCompletionOutcome::Applied(outcome)) -} - -/// Deletes at most 128 terminal workflow runs whose retention has elapsed. -pub fn workflow_cleanup_terminal(transaction: &Transaction<'_>, now_ms: i64) -> Result { - workflow_cleanup_terminal_bounded(transaction, now_ms, MAX_SCAN_ITEMS) -} - -pub(crate) fn workflow_cleanup_terminal_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - validate_now(now_ms)?; - validate_maintenance_limit(limit)?; - if limit == 0 { - return Ok(0); - } - let cutoff = now_ms.saturating_sub(TERMINAL_RETENTION_MS); - let run_ids = { - let mut statement = transaction.prepare( - "SELECT run_id FROM workflow_runs WHERE status BETWEEN 1 AND 3 AND completed_at_ms <= ?1 ORDER BY completed_at_ms, run_id LIMIT ?2", - )?; - statement - .query_map((cutoff, limit as i64), |row| row.get::<_, Vec>(0))? - .collect::, _>>()? - }; - for value in &run_ids { - let run_id = exact::<16>(value.clone(), "terminal workflow run ID")?; - transaction.execute( - "DELETE FROM workflow_activities WHERE run_id = ?1", - [run_id.as_slice()], - )?; - transaction.execute( - "DELETE FROM workflow_timers WHERE run_id = ?1", - [run_id.as_slice()], - )?; - transaction.execute( - "DELETE FROM workflow_events WHERE run_id = ?1", - [run_id.as_slice()], - )?; - if transaction.execute( - "DELETE FROM workflow_runs WHERE run_id = ?1 AND status BETWEEN 1 AND 3 AND completed_at_ms <= ?2", - (run_id.as_slice(), cutoff), - )? != 1 - { - return Err(Error::Command("terminal workflow changed during cleanup")); - } - } - Ok(run_ids.len()) -} - -struct StoredActivity { - state: i64, - attempt: u32, - token: Option<[u8; 16]>, - lease_until_ms: Option, - expires_at_ms: i64, - completion_token: Option<[u8; 16]>, - completion_digest: Option<[u8; 32]>, - result: Vec, -} - -fn load_activity( - transaction: &Transaction<'_>, - run_id: [u8; 16], - activity_id: [u8; 16], -) -> Result> { - let stored = transaction - .query_row( - "SELECT state, attempt, token, lease_until_ms, expires_at_ms, completion_token, completion_digest, result FROM workflow_activities WHERE run_id = ?1 AND activity_id = ?2", - (run_id.as_slice(), activity_id.as_slice()), - |row| { - Ok(( - row.get::<_, i64>(0)?, - row.get::<_, i64>(1)?, - row.get::<_, Option>>(2)?, - row.get::<_, Option>(3)?, - row.get::<_, i64>(4)?, - row.get::<_, Option>>(5)?, - row.get::<_, Option>>(6)?, - row.get::<_, Option>>(7)?, - )) - }, - ) - .optional()?; - stored - .map( - |(state, attempt, token, lease_until_ms, expires_at_ms, completion, digest, result)| { - let result = result.unwrap_or_default(); - if !(0..=4).contains(&state) - || attempt < 0 - || expires_at_ms < 0 - || result.len() > MAX_ACTIVITY_PAYLOAD_BYTES - { - return Err(Error::Command("invalid stored workflow activity")); - } - Ok(StoredActivity { - state, - attempt: u32::try_from(attempt) - .map_err(|_| Error::Command("invalid stored workflow activity attempt"))?, - token: optional_exact(token, "workflow activity token")?, - lease_until_ms, - expires_at_ms, - completion_token: optional_exact(completion, "activity completion token")?, - completion_digest: optional_exact(digest, "activity completion digest")?, - result, - }) - }, - ) - .transpose() -} - -fn update_completion( - transaction: &Transaction<'_>, - completion: &ActivityCompletion, - stored: &StoredActivity, - state: i64, - due_at_ms: Option, - digest: [u8; 32], -) -> Result<()> { - let due_at_ms = due_at_ms.unwrap_or(0); - if transaction.execute( - "UPDATE workflow_activities SET state = ?1, due_at_ms = CASE WHEN ?1 = 0 THEN ?2 ELSE due_at_ms END, token = NULL, lease_until_ms = NULL, completion_token = ?3, completion_digest = ?4, result = ?5 WHERE run_id = ?6 AND activity_id = ?7 AND state = 1 AND attempt = ?8 AND token = ?9 AND lease_until_ms = ?10", - ( - state, - due_at_ms, - completion.completion_token.as_slice(), - digest.as_slice(), - completion.result.as_slice(), - completion.run_id.as_slice(), - completion.activity_id.as_slice(), - i64::from(completion.attempt), - completion.lease_token.as_slice(), - stored.lease_until_ms, - ), - )? != 1 - { - return Err(Error::Command("activity changed during serialized completion")); - } - Ok(()) -} - -pub(crate) fn workflow_reclaim_expired_bounded( - transaction: &Transaction<'_>, - now_ms: i64, - limit: usize, -) -> Result { - validate_now(now_ms)?; - validate_maintenance_limit(limit)?; - if limit == 0 { - return Ok(0); - } - Ok(transaction.execute( - "WITH expired(run_id, activity_id) AS (SELECT run_id, activity_id FROM workflow_activities INDEXED BY activities_leases WHERE state = 1 AND lease_until_ms <= ?1 ORDER BY lease_until_ms, run_id, activity_id LIMIT ?2) UPDATE workflow_activities SET state = 0, due_at_ms = CASE WHEN attempt >= ?3 OR expires_at_ms <= ?1 THEN due_at_ms ELSE ?1 END, token = NULL, lease_until_ms = NULL WHERE (run_id, activity_id) IN (SELECT run_id, activity_id FROM expired)", - (now_ms, limit as i64, i64::from(MAX_ATTEMPTS)), - )?) -} - -fn validate_maintenance_limit(limit: usize) -> Result<()> { - if limit > MAX_SCAN_ITEMS { - return Err(Error::Command("workflow maintenance limit exceeds 128")); - } - Ok(()) -} - -fn validate_support(supported: &[ActivitySupport]) -> Result<()> { - if supported.is_empty() || supported.len() > MAX_SCAN_ITEMS { - return Err(Error::Command("activity support count must be in 1..=128")); - } - for (index, support) in supported.iter().enumerate() { - if support.activity_type.is_empty() - || support.activity_type.len() > MAX_ACTIVITY_TYPE_BYTES - || supported[..index].contains(support) - { - return Err(Error::Command("invalid or duplicate activity support")); - } - } - Ok(()) -} - -fn completion_digest(completion: &ActivityCompletion) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.activity-completion.v1\0"); - hasher.update(&[u8::from(completion.failed), u8::from(completion.retryable)]); - hasher.update(&(completion.result.len() as u32).to_be_bytes()); - hasher.update(&completion.result); - *hasher.finalize().as_bytes() -} - -fn completion_event(completion: &ActivityCompletion) -> Vec { - let mut event = Vec::with_capacity(34 + completion.result.len()); - event.extend_from_slice(b"activity\0"); - event.push(u8::from(completion.failed)); - event.extend_from_slice(&completion.activity_id); - event.extend_from_slice(&(completion.result.len() as u32).to_be_bytes()); - event.extend_from_slice(&completion.result); - event -} - -fn completion_event_id(completion: &ActivityCompletion) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(&completion.run_id); - hasher.update(&completion.activity_id); - hasher.update(&completion.completion_token); - hasher.update(if completion.failed { - b"failed" - } else { - b"completed" - }); - *hasher.finalize().as_bytes() -} - -fn retry_delay_ms(attempt: u32) -> i64 { - (100_i64 << attempt.min(10)).min(60_000) -} - -fn exact(value: Vec, field: &'static str) -> Result<[u8; N]> { - value.try_into().map_err(|_| Error::Command(field)) -} - -fn optional_exact( - value: Option>, - field: &'static str, -) -> Result> { - value.map(|value| exact(value, field)).transpose() -} - -fn validate_now(now_ms: i64) -> Result<()> { - if now_ms < 0 { - return Err(Error::Command("activity time must be non-negative")); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/primitives/workflow/activity_api.rs b/crates/crab-cell-runtime/src/primitives/workflow/activity_api.rs deleted file mode 100644 index 6e3e70fc6..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow/activity_api.rs +++ /dev/null @@ -1,522 +0,0 @@ -use std::{ - fmt, - future::Future, - marker::PhantomData, - pin::Pin, - sync::{ - Arc, - atomic::{AtomicBool, AtomicI64, Ordering}, - }, - time::{Duration, SystemTime, UNIX_EPOCH}, -}; - -use rand::RngCore; - -use crate::Error; -use crate::cell::catalog::CatalogRole; -use crate::cell::executor::{MutationIdentity, Resolution}; -use crate::client::{CellClient, Committed, InvocationError, Observed, PendingMutation, Receipt}; -use crate::identity::RequestId; -use crate::identity::{ApplicationId, CellTarget, TenantId, partition_for_shard}; -use crate::primitives::activity_pool::BlockingActivityReservation; -use crate::registry::{Command, Query, RegistryBuilder}; -use crate::registry::{CommandContext, CommandResult, QueryContext}; - -use super::{ - ActivityClaim, ActivityCompletion, ActivityCompletionOutcome, ActivityLeaseOutcome, - ActivitySupport, SystemActivityTokens, WorkflowModule, WorkflowOutcome, - activity::{MAX_ACTIVITY_PAYLOAD_BYTES, MAX_LEASE_MS}, - api::{definition, definitions}, - workflow_claim_activities, workflow_complete_activity, workflow_definition_digest_by_run, - workflow_extend_activity, workflow_validate_activity_claim, -}; - -const HANDLER_IDENTITY_LIFETIME_MS: i64 = 60_000; - -/// Compile-time operation IDs for native activity execution in one Workflow module. -pub trait WorkflowActivityModule: WorkflowModule { - /// Registered activity types the module serves. - const ACTIVITY_TYPES: &'static [&'static str]; - /// Command id that claims activity leases. - const ACTIVITY_CLAIM_COMMAND_ID: u32; - /// Command id that applies activity completions. - const ACTIVITY_COMPLETE_COMMAND_ID: u32; - /// Command id that extends activity leases. - const ACTIVITY_EXTEND_COMMAND_ID: u32; - /// Query id that revalidates claimed activity leases. - const ACTIVITY_VALIDATE_QUERY_ID: u32; -} - -/// Registers the typed claim, completion, extension and validation bindings. -pub fn register_workflow_activities( - registry: &mut RegistryBuilder, -) -> crate::Result<()> { - for definition in definitions::()? { - registry.bind_activity_inventory(M::MODULE, definition.digest(), M::ACTIVITY_TYPES)?; - } - registry.bind_activity_runner::()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_query::>() -} - -/// Registers one native handler for a module definition and activity type. -pub fn register_activity( - registry: &mut RegistryBuilder, -) -> crate::Result<()> { - for definition in definitions::()? { - registry.bind_activity::
(M::MODULE, definition.digest())?; - } - Ok(()) -} - -/// Registers one trusted blocking handler on the node-owned activity pool. -pub fn register_blocking_activity( - registry: &mut RegistryBuilder, -) -> crate::Result<()> { - for definition in definitions::()? { - registry.bind_blocking_activity::(M::MODULE, definition.digest())?; - } - Ok(()) -} - -/// Cooperative cancellation and lease state supplied to native activity code. -#[derive(Clone)] -pub struct ActivityContext { - run_id: [u8; 16], - activity_id: [u8; 16], - attempt: u32, - lease_token: [u8; 16], - cancellation: ActivityCancellation, - lease_until_ms: Arc, -} - -impl ActivityContext { - fn new(claim: &ActivityClaim) -> Self { - Self { - run_id: claim.run_id, - activity_id: claim.activity_id, - attempt: claim.attempt, - lease_token: claim.token, - cancellation: ActivityCancellation::default(), - lease_until_ms: Arc::new(AtomicI64::new(claim.lease_until_ms)), - } - } - - /// Returns the workflow run the claim belongs to. - #[must_use] - pub const fn run_id(&self) -> [u8; 16] { - self.run_id - } - - /// Returns the activity identity within the run. - #[must_use] - pub const fn activity_id(&self) -> [u8; 16] { - self.activity_id - } - - /// Returns the delivery attempt this claim represents. - #[must_use] - pub const fn attempt(&self) -> u32 { - self.attempt - } - - /// Returns the lease token required to extend or complete. - #[must_use] - pub const fn lease_token(&self) -> [u8; 16] { - self.lease_token - } - - /// Returns a stable external idempotency key shared by every retry attempt. - #[must_use] - pub fn idempotency_key(&self) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.activity-idempotency.v1\0"); - hasher.update(&self.run_id); - hasher.update(&self.activity_id); - *hasher.finalize().as_bytes() - } - - /// Returns the logical time the lease expires. - #[must_use] - pub fn lease_until_ms(&self) -> i64 { - self.lease_until_ms.load(Ordering::Acquire) - } - - /// Returns a handle the handler can poll for cancellation. - #[must_use] - pub fn cancellation(&self) -> ActivityCancellation { - self.cancellation.clone() - } -} - -/// Cloneable cancellation observation shared with one activity attempt. -#[derive(Clone, Default)] -pub struct ActivityCancellation(Arc); - -impl ActivityCancellation { - /// Reports whether the activity was cancelled. - #[must_use] - pub fn is_cancelled(&self) -> bool { - self.0.load(Ordering::Acquire) - } - - fn cancel(&self) { - self.0.store(true, Ordering::Release); - } -} - -/// Native handler result converted to one durable completion transition. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum ActivityExecution { - /// The handler completed with this result. - Completed(Vec), - /// The handler failed. - Failed { - /// Failure details recorded for the run. - details: Vec, - /// Whether the workflow may retry the activity. - retryable: bool, - }, -} - -impl ActivityExecution { - fn payload(&self) -> (&[u8], bool, bool) { - match self { - Self::Completed(result) => (result, false, false), - Self::Failed { details, retryable } => (details, true, *retryable), - } - } -} - -/// Statically linked asynchronous activity implemented by trusted Rust code. -pub trait ActivityHandler: Send + Sync + 'static { - /// Activity type this handler serves. - const TYPE: &'static str; - - /// Runs the handler for one claim. - fn execute( - context: ActivityContext, - input: Vec, - ) -> Pin + Send + 'static>>; -} - -/// Statically linked blocking activity implemented by trusted Rust code. -pub trait BlockingActivityHandler: Send + Sync + 'static { - /// Activity type this handler serves. - const TYPE: &'static str; - - /// Runs the blocking handler on the dedicated worker. - fn execute(context: ActivityContext, input: Vec) -> ActivityExecution; -} - -/// Bounded activity claim parameters supplied by the native supervisor. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowActivityClaimRequest { - /// Maximum activities to claim. - pub limit: u32, - /// Lease duration granted to each claim. - pub lease_ms: u32, -} - -/// Typed activity claim bound to one Workflow module. -pub struct WorkflowActivityClaimCommand(PhantomData M>); - -impl Command for WorkflowActivityClaimCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::ACTIVITY_CLAIM_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = WorkflowActivityClaimRequest; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let limit = usize::try_from(input.limit) - .map_err(|_| Error::Command("activity claim limit overflow"))?; - let supported = definitions::()? - .iter() - .flat_map(|definition| { - M::ACTIVITY_TYPES - .iter() - .map(|activity_type| ActivitySupport { - activity_type: (*activity_type).to_owned(), - definition_digest: definition.digest(), - }) - }) - .collect::>(); - let mut tokens = SystemActivityTokens; - Ok(CommandResult::Success(workflow_claim_activities( - context.primitive_transaction(), - context.now_ms(), - limit, - input.lease_ms, - &supported, - &mut tokens, - )?)) - } -} - -/// Typed activity completion bound to its pinned Workflow definition. -pub struct WorkflowActivityCompleteCommand(PhantomData M>); - -impl Command for WorkflowActivityCompleteCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::ACTIVITY_COMPLETE_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = ActivityCompletion; - type Output = ActivityCompletionOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let source = context.target().clone(); - let Some(digest) = - workflow_definition_digest_by_run(context.primitive_transaction(), input.run_id)? - else { - return Ok(CommandResult::Rejected( - ActivityCompletionOutcome::LeaseLost, - )); - }; - let outcome = workflow_complete_activity( - context.primitive_transaction(), - &source, - context.now_ms(), - &input, - definition::(digest)?, - )?; - Ok(match outcome { - ActivityCompletionOutcome::Applied(_) - | ActivityCompletionOutcome::Retrying { .. } - | ActivityCompletionOutcome::Duplicate { .. } => CommandResult::Success(outcome), - ActivityCompletionOutcome::IdentityConflict | ActivityCompletionOutcome::LeaseLost => { - CommandResult::Rejected(outcome) - } - }) - } -} - -/// Exact lease and requested extension for one activity attempt. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowActivityExtendRequest { - /// Claim whose lease is extended. - pub claim: ActivityClaim, - /// Additional lease time to grant. - pub extension_ms: u32, -} - -/// Typed heartbeat extension bound to one Workflow module. -pub struct WorkflowActivityExtendCommand(PhantomData M>); - -impl Command for WorkflowActivityExtendCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::ACTIVITY_EXTEND_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = WorkflowActivityExtendRequest; - type Output = ActivityLeaseOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let outcome = workflow_extend_activity( - context.primitive_transaction(), - context.now_ms(), - &input.claim, - input.extension_ms, - )?; - Ok(match outcome { - ActivityLeaseOutcome::Extended { .. } => CommandResult::Success(outcome), - ActivityLeaseOutcome::LeaseLost => CommandResult::Rejected(outcome), - }) - } -} - -/// Exact published claim set revalidated before native execution. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowActivityValidateRequest { - /// Claims to revalidate before native emission. - pub claimed: Vec, -} - -/// Typed post-publication activity validation query. -pub struct WorkflowActivityValidateQuery(PhantomData M>); - -impl Query for WorkflowActivityValidateQuery { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::ACTIVITY_VALIDATE_QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = WorkflowActivityValidateRequest; - type Output = bool; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - workflow_validate_activity_claim( - context.primitive_connection(), - context.now_ms(), - &input.claimed, - ) - } -} - -/// Per-namespace capability used only by the native activity supervisor. -pub struct WorkflowActivities { - client: CellClient, - tenant: TenantId, - application: ApplicationId, - shards: u32, - module: PhantomData M>, -} - -impl Clone for WorkflowActivities { - fn clone(&self) -> Self { - Self { - client: self.client.clone(), - tenant: self.tenant, - application: self.application, - shards: self.shards, - module: PhantomData, - } - } -} - -impl WorkflowActivities { - /// Creates a native activity capability from the immutable compiled registry. - pub fn new( - client: CellClient, - tenant: TenantId, - application: ApplicationId, - ) -> crate::Result { - let shards = client.require_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Workflow)?; - for definition in definitions::()? { - client.activity_support(M::MODULE, definition.digest())?; - } - Ok(Self { - client, - tenant, - application, - shards, - module: PhantomData, - }) - } - - async fn claim( - &self, - identity: MutationIdentity, - shard: u32, - lease_ms: u32, - ) -> std::result::Result>, InvocationError>> - { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>( - &target, - identity, - WorkflowActivityClaimRequest { limit: 1, lease_ms }, - ) - .await - } - - async fn validate( - &self, - shard: u32, - claim: ActivityClaim, - minimum: Receipt, - ) -> std::result::Result, InvocationError> { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - // Lease checks gate external work; a stale snapshot cannot prove that - // the owner has not revoked or replaced the claim. - self.client - .with_read_policy(crate::client::ReadPolicy::CurrentOwner) - .query::>( - &target, - Some(minimum), - WorkflowActivityValidateRequest { - claimed: vec![claim], - }, - ) - .await - } - - async fn extend( - &self, - identity: MutationIdentity, - shard: u32, - claim: ActivityClaim, - extension_ms: u32, - ) -> std::result::Result, InvocationError> - { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>( - &target, - identity, - WorkflowActivityExtendRequest { - claim, - extension_ms, - }, - ) - .await - } - - async fn complete( - &self, - identity: MutationIdentity, - shard: u32, - completion: ActivityCompletion, - ) -> std::result::Result< - Committed, - InvocationError, - > { - let target = self - .shard_target(shard) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>(&target, identity, completion) - .await - } - - async fn execute( - &self, - definition: crate::Digest, - activity_type: String, - input: Vec, - context: ActivityContext, - blocking: Option, - ) -> crate::Result { - self.client - .execute_activity( - M::MODULE, - definition, - &activity_type, - context, - input, - blocking, - ) - .await - } - - fn shard_target(&self, shard: u32) -> crate::Result { - if shard >= self.shards { - return Err(Error::Identity("Workflow shard outside namespace")); - } - CellTarget::new( - self.tenant, - self.application, - M::NAMESPACE, - &partition_for_shard(shard), - ) - } -} - -mod supervisor; - -pub use supervisor::{ActivityRunOutcome, ActivitySupervisor, ActivitySupervisorError}; diff --git a/crates/crab-cell-runtime/src/primitives/workflow/activity_api/supervisor.rs b/crates/crab-cell-runtime/src/primitives/workflow/activity_api/supervisor.rs deleted file mode 100644 index fbda34214..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow/activity_api/supervisor.rs +++ /dev/null @@ -1,386 +0,0 @@ -//! Activity claim/execute/completion supervision. - -use super::*; - -/// Outcome of one bounded claim, execute and completion supervisor cycle. -#[derive(Clone, Debug, PartialEq, Eq)] -pub enum ActivityRunOutcome { - /// No claimable activity remained. - Idle { - /// Receipt the observation is bound to. - receipt: Receipt, - }, - /// The source lease was lost before the completion. - LeaseLost { - /// Receipt the failed resolution is bound to. - receipt: Receipt, - }, - /// The completion named a different activity than the lease. - IdentityConflict { - /// Receipt the conflict was observed at. - receipt: Receipt, - }, - /// The failed attempt is retried. - Retrying { - /// Logical time the next attempt is due. - due_at_ms: i64, - /// Receipt the source transition published. - receipt: Receipt, - }, - /// The completion advanced the workflow run. - Completed { - /// Workflow outcome the completion produced. - workflow: WorkflowOutcome, - /// Receipt the source transition published. - receipt: Receipt, - }, - /// The same completion token was already applied. - Duplicate { - /// Result recorded for the original completion. - result: Vec, - /// Receipt the duplicate was observed at. - receipt: Receipt, - }, -} - -/// Failure that preserves unresolved mutation evidence from supervisor commands. -pub enum ActivitySupervisorError { - /// A source transition is committed but unresolved; resolve it before another cycle. - Pending(Box), - /// The published source result could not be decoded. - InvalidPublishedResult { - /// Receipt the published result was observed at. - receipt: Receipt, - /// Decoding failure that produced this error. - source: Box, - }, - /// The supervisor failed before it could resolve the source transition. - Runtime(Error), -} - -impl fmt::Debug for ActivitySupervisorError { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Pending(pending) => formatter.debug_tuple("Pending").field(pending).finish(), - Self::InvalidPublishedResult { receipt, source } => formatter - .debug_struct("InvalidPublishedResult") - .field("receipt", receipt) - .field("source", source) - .finish(), - Self::Runtime(error) => formatter.debug_tuple("Runtime").field(error).finish(), - } - } -} - -impl fmt::Display for ActivitySupervisorError { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Self::Pending(_) => { - formatter.write_str("activity supervisor mutation needs resolution") - } - Self::InvalidPublishedResult { .. } => { - formatter.write_str("activity supervisor received an invalid published result") - } - Self::Runtime(error) => write!(formatter, "activity supervisor failed: {error}"), - } - } -} - -impl std::error::Error for ActivitySupervisorError { - fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { - match self { - Self::InvalidPublishedResult { source, .. } => Some(source.as_ref()), - Self::Runtime(error) => Some(error), - Self::Pending(_) => None, - } - } -} - -/// Runs native activity attempts without holding a SQLite transaction across await. -pub struct ActivitySupervisor { - activities: WorkflowActivities, - lease_ms: u32, -} - -impl ActivitySupervisor { - /// Creates a supervisor using a 5..=300 second heartbeat lease. - pub fn new(activities: WorkflowActivities, lease_ms: u32) -> crate::Result { - if !(5_000..=300_000).contains(&lease_ms) { - return Err(Error::Command( - "activity supervisor lease must be in 5..=300 seconds", - )); - } - Ok(Self { - activities, - lease_ms, - }) - } - - /// Claims and executes at most one activity from an explicit Workflow shard. - pub async fn run_once( - &self, - shard: u32, - blocking: Option, - ) -> std::result::Result { - let claimed = self - .activities - .claim(internal_identity()?, shard, self.lease_ms) - .await - .map_err(unexpected_invocation)?; - let Some(mut claim) = claimed.output.into_iter().next() else { - return Ok(ActivityRunOutcome::Idle { - receipt: claimed.receipt, - }); - }; - let validation = self - .activities - .validate(shard, claim.clone(), claimed.receipt) - .await - .map_err(unexpected_invocation)?; - if !validation.output { - return Ok(ActivityRunOutcome::LeaseLost { - receipt: validation.receipt, - }); - } - - let context = ActivityContext::new(&claim); - let cancellation = context.cancellation(); - let _cancellation_guard = CancellationGuard(cancellation.clone()); - let lease_deadline = context.lease_until_ms.clone(); - let execution = self.activities.execute( - claim.definition_digest, - claim.activity_type.clone(), - claim.input.clone(), - context, - blocking, - ); - tokio::pin!(execution); - let heartbeat_period = Duration::from_millis(u64::from(self.lease_ms / 3)); - let heartbeat = tokio::time::sleep(heartbeat_period); - tokio::pin!(heartbeat); - let execution = loop { - tokio::select! { - result = &mut execution => break result.map_err(ActivitySupervisorError::Runtime)?, - () = &mut heartbeat => { - let extension = self.activities - .extend(internal_identity()?, shard, claim.clone(), self.lease_ms) - .await; - match extension { - Ok(committed) => match committed.output { - ActivityLeaseOutcome::Extended { lease_until_ms } => { - claim.lease_until_ms = lease_until_ms; - lease_deadline.store(lease_until_ms, Ordering::Release); - } - ActivityLeaseOutcome::LeaseLost => { - cancellation.cancel(); - return Ok(ActivityRunOutcome::LeaseLost { receipt: committed.receipt }); - } - }, - Err(InvocationError::Rejected(committed)) - if committed.output == ActivityLeaseOutcome::LeaseLost => - { - cancellation.cancel(); - return Ok(ActivityRunOutcome::LeaseLost { receipt: committed.receipt }); - } - Err(error) => { - cancellation.cancel(); - return Err(unexpected_invocation(error)); - } - } - heartbeat.as_mut().reset(tokio::time::Instant::now() + heartbeat_period); - } - } - }; - - let (result, failed, retryable) = execution.payload(); - if result.len() > MAX_ACTIVITY_PAYLOAD_BYTES { - return Err(ActivitySupervisorError::Runtime(Error::Command( - "activity handler result exceeds 256 KiB", - ))); - } - - if self.lease_ms < MAX_LEASE_MS { - // Provider-backed completion can take longer than the handler. Reserve a fresh - // bounded lease before submitting the terminal mutation, or the result can be - // rejected as expired while the owner is still durably completing it. - let extended = self - .activities - .extend(internal_identity()?, shard, claim.clone(), MAX_LEASE_MS) - .await; - let extended = match extended { - Ok(committed) => committed, - Err(InvocationError::Rejected(committed)) - if committed.output == ActivityLeaseOutcome::LeaseLost => - { - return Ok(ActivityRunOutcome::LeaseLost { - receipt: committed.receipt, - }); - } - Err(error) => return Err(unexpected_invocation(error)), - }; - match extended.output { - ActivityLeaseOutcome::Extended { lease_until_ms } => { - claim.lease_until_ms = lease_until_ms; - lease_deadline.store(lease_until_ms, Ordering::Release); - } - ActivityLeaseOutcome::LeaseLost => { - return Ok(ActivityRunOutcome::LeaseLost { - receipt: extended.receipt, - }); - } - } - } - - let completion = ActivityCompletion { - run_id: claim.run_id, - activity_id: claim.activity_id, - attempt: claim.attempt, - lease_token: claim.token, - completion_token: completion_token(&claim), - result: result.to_vec(), - failed, - retryable, - }; - let completion_identity = internal_identity()?; - let result = self - .activities - .complete(completion_identity, shard, completion.clone()) - .await; - let result = match result { - Err(InvocationError::Pending(pending)) => { - // Resolve the exact completion before returning or retrying: rerunning the - // handler could repeat an external side effect after its result committed. - let resolution = match self.activities.client.resolve(&pending).await { - Ok(resolution) => resolution, - Err(_) => return Err(ActivitySupervisorError::Pending(pending)), - }; - match resolution { - Resolution::Committed(outcome) => crate::client::decode_pending::< - ActivityCompletionOutcome, - >(&pending, outcome), - Resolution::Absent => { - self.activities - .complete(completion_identity, shard, completion) - .await - } - Resolution::Unknown | Resolution::Expired => { - return Err(ActivitySupervisorError::Pending(pending)); - } - } - } - result => result, - }; - let committed = match result { - Ok(committed) => committed, - Err(InvocationError::Rejected(committed)) => { - return match committed.output { - ActivityCompletionOutcome::LeaseLost => Ok(ActivityRunOutcome::LeaseLost { - receipt: committed.receipt, - }), - ActivityCompletionOutcome::IdentityConflict => { - Ok(ActivityRunOutcome::IdentityConflict { - receipt: committed.receipt, - }) - } - _ => Err(ActivitySupervisorError::Runtime(Error::Command( - "activity completion returned an invalid rejection", - ))), - }; - } - Err(error) => return Err(unexpected_invocation(error)), - }; - Ok(match committed.output { - ActivityCompletionOutcome::Applied(workflow) => ActivityRunOutcome::Completed { - workflow, - receipt: committed.receipt, - }, - ActivityCompletionOutcome::Retrying { due_at_ms } => ActivityRunOutcome::Retrying { - due_at_ms, - receipt: committed.receipt, - }, - ActivityCompletionOutcome::Duplicate { result } => ActivityRunOutcome::Duplicate { - result, - receipt: committed.receipt, - }, - ActivityCompletionOutcome::IdentityConflict => ActivityRunOutcome::IdentityConflict { - receipt: committed.receipt, - }, - ActivityCompletionOutcome::LeaseLost => ActivityRunOutcome::LeaseLost { - receipt: committed.receipt, - }, - }) - } -} - -struct CancellationGuard(ActivityCancellation); - -impl Drop for CancellationGuard { - fn drop(&mut self) { - self.0.cancel(); - } -} - -fn completion_token(claim: &ActivityClaim) -> [u8; 16] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.activity-completion.v1\0"); - hasher.update(&claim.run_id); - hasher.update(&claim.activity_id); - hasher.update(&claim.attempt.to_be_bytes()); - let mut token = [0; 16]; - token.copy_from_slice(&hasher.finalize().as_bytes()[..16]); - token -} - -fn internal_identity() -> std::result::Result { - let now_ms = i64::try_from( - SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_err(|_| { - ActivitySupervisorError::Runtime(Error::Command( - "system clock is before Unix epoch", - )) - })? - .as_millis(), - ) - .map_err(|_| { - ActivitySupervisorError::Runtime(Error::Command("system clock exceeds i64 milliseconds")) - })?; - let expires_at_ms = now_ms.checked_add(HANDLER_IDENTITY_LIFETIME_MS).ok_or( - ActivitySupervisorError::Runtime(Error::Command("activity mutation expiry overflow")), - )?; - let mut request_id = [0; 16]; - rand::rng().fill_bytes(&mut request_id); - Ok(MutationIdentity { - request_id: RequestId::from_bytes(request_id), - issued_at_ms: now_ms, - expires_at_ms, - }) -} - -fn unexpected_invocation(error: InvocationError) -> ActivitySupervisorError { - match error { - InvocationError::Pending(pending) => ActivitySupervisorError::Pending(pending), - InvocationError::InvalidPublishedResult { receipt, source } => { - ActivitySupervisorError::InvalidPublishedResult { receipt, source } - } - InvocationError::NotStarted(error) => ActivitySupervisorError::Runtime(error), - InvocationError::Rejected(_) => ActivitySupervisorError::Runtime(Error::Command( - "activity supervisor command was unexpectedly rejected", - )), - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn dropping_an_attempt_guard_signals_cooperative_cancellation() { - let cancellation = ActivityCancellation::default(); - { - let _guard = CancellationGuard(cancellation.clone()); - assert!(!cancellation.is_cancelled()); - } - assert!(cancellation.is_cancelled()); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/workflow/activity_codec.rs b/crates/crab-cell-runtime/src/primitives/workflow/activity_codec.rs deleted file mode 100644 index 2af89e35f..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow/activity_codec.rs +++ /dev/null @@ -1,228 +0,0 @@ -use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue, read_fixed}; - -use super::{ - ActivityClaim, ActivityCompletion, ActivityCompletionOutcome, ActivityLeaseOutcome, - WorkflowActivityClaimRequest, WorkflowActivityExtendRequest, WorkflowActivityValidateRequest, - WorkflowOutcome, - activity::{MAX_ACTIVITY_PAYLOAD_BYTES, MAX_CLAIM_ITEMS}, -}; - -const APPLIED_TAG: u8 = 0; -const RETRYING_TAG: u8 = 1; -const DUPLICATE_TAG: u8 = 2; -const IDENTITY_CONFLICT_TAG: u8 = 3; -const LEASE_LOST_TAG: u8 = 4; -const EXTENDED_TAG: u8 = 0; - -impl WireValue for WorkflowActivityClaimRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_u32(self.limit)?; - encoder.write_u32(self.lease_ms) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - limit: decoder.read_u32()?, - lease_ms: decoder.read_u32()?, - }) - } -} - -impl WireValue for ActivityClaim { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - validate_claim(self)?; - encoder.write_bytes(&self.run_id)?; - encoder.write_bytes(&self.activity_id)?; - encoder.write_text(&self.activity_type)?; - encoder.write_bytes(&self.input)?; - encoder.write_bytes(self.definition_digest.as_bytes())?; - encoder.write_u32(self.attempt)?; - encoder.write_bytes(&self.token)?; - encoder.write_i64(self.lease_until_ms) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let claim = Self { - run_id: read_fixed(decoder, "activity run ID length")?, - activity_id: read_fixed(decoder, "activity ID length")?, - activity_type: decoder.read_text()?.to_owned(), - input: decoder.read_bytes()?.to_vec(), - definition_digest: crate::Digest::from_bytes(read_fixed( - decoder, - "activity definition digest length", - )?), - attempt: decoder.read_u32()?, - token: read_fixed(decoder, "activity lease token length")?, - lease_until_ms: decoder.read_i64()?, - }; - validate_claim(&claim)?; - Ok(claim) - } -} - -impl WireValue for Vec { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if self.len() > MAX_CLAIM_ITEMS { - return Err(CodecError::Invalid("too many activity claims")); - } - encoder.write_count(self.len())?; - for claim in self { - claim.encode(encoder)?; - } - Ok(()) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let count = decoder.read_count()?; - if count > MAX_CLAIM_ITEMS { - return Err(CodecError::Invalid("too many activity claims")); - } - let mut claims = Vec::with_capacity(count); - for _ in 0..count { - claims.push(ActivityClaim::decode(decoder)?); - } - Ok(claims) - } -} - -impl WireValue for WorkflowActivityExtendRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - self.claim.encode(encoder)?; - encoder.write_u32(self.extension_ms) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - claim: ActivityClaim::decode(decoder)?, - extension_ms: decoder.read_u32()?, - }) - } -} - -impl WireValue for ActivityLeaseOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Extended { lease_until_ms } => { - encoder.write_u8(EXTENDED_TAG)?; - encoder.write_i64(*lease_until_ms) - } - Self::LeaseLost => encoder.write_u8(LEASE_LOST_TAG), - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - EXTENDED_TAG => Ok(Self::Extended { - lease_until_ms: decoder.read_i64()?, - }), - LEASE_LOST_TAG => Ok(Self::LeaseLost), - _ => Err(CodecError::Invalid("invalid activity lease outcome tag")), - } - } -} - -impl WireValue for ActivityCompletion { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - if self.result.len() > MAX_ACTIVITY_PAYLOAD_BYTES || (!self.failed && self.retryable) { - return Err(CodecError::Invalid("invalid activity completion")); - } - encoder.write_bytes(&self.run_id)?; - encoder.write_bytes(&self.activity_id)?; - encoder.write_u32(self.attempt)?; - encoder.write_bytes(&self.lease_token)?; - encoder.write_bytes(&self.completion_token)?; - encoder.write_bytes(&self.result)?; - encoder.write_bool(self.failed)?; - encoder.write_bool(self.retryable) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let completion = Self { - run_id: read_fixed(decoder, "activity completion run ID length")?, - activity_id: read_fixed(decoder, "activity completion ID length")?, - attempt: decoder.read_u32()?, - lease_token: read_fixed(decoder, "activity completion lease token length")?, - completion_token: read_fixed(decoder, "activity completion token length")?, - result: decoder.read_bytes()?.to_vec(), - failed: decoder.read_bool()?, - retryable: decoder.read_bool()?, - }; - if completion.result.len() > MAX_ACTIVITY_PAYLOAD_BYTES - || (!completion.failed && completion.retryable) - { - return Err(CodecError::Invalid("invalid activity completion")); - } - Ok(completion) - } -} - -impl WireValue for ActivityCompletionOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Applied(workflow) => { - encoder.write_u8(APPLIED_TAG)?; - workflow.encode(encoder) - } - Self::Retrying { due_at_ms } => { - encoder.write_u8(RETRYING_TAG)?; - encoder.write_i64(*due_at_ms) - } - Self::Duplicate { result } => { - encoder.write_u8(DUPLICATE_TAG)?; - encoder.write_bytes(result) - } - Self::IdentityConflict => encoder.write_u8(IDENTITY_CONFLICT_TAG), - Self::LeaseLost => encoder.write_u8(LEASE_LOST_TAG), - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - APPLIED_TAG => Ok(Self::Applied(WorkflowOutcome::decode(decoder)?)), - RETRYING_TAG => Ok(Self::Retrying { - due_at_ms: decoder.read_i64()?, - }), - DUPLICATE_TAG => { - let result = decoder.read_bytes()?.to_vec(); - if result.len() > MAX_ACTIVITY_PAYLOAD_BYTES { - return Err(CodecError::Invalid("activity result exceeds 256 KiB")); - } - Ok(Self::Duplicate { result }) - } - IDENTITY_CONFLICT_TAG => Ok(Self::IdentityConflict), - LEASE_LOST_TAG => Ok(Self::LeaseLost), - _ => Err(CodecError::Invalid( - "invalid activity completion outcome tag", - )), - } - } -} - -impl WireValue for WorkflowActivityValidateRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - self.claimed.encode(encoder) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - claimed: Vec::::decode(decoder)?, - }) - } -} - -fn validate_claim(claim: &ActivityClaim) -> Result<(), CodecError> { - if claim.activity_type.is_empty() - || claim.activity_type.len() > 256 - || claim.input.len() > MAX_ACTIVITY_PAYLOAD_BYTES - || claim - .definition_digest - .as_bytes() - .iter() - .all(|byte| *byte == 0) - || claim.attempt == 0 - || claim.lease_until_ms < 0 - { - return Err(CodecError::Invalid("invalid activity claim")); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/primitives/workflow/api.rs b/crates/crab-cell-runtime/src/primitives/workflow/api.rs deleted file mode 100644 index fbc86f16a..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow/api.rs +++ /dev/null @@ -1,403 +0,0 @@ -use std::marker::PhantomData; - -use crate::cell::catalog::CatalogRole; -use crate::client::{CellClient, Committed, InvocationError, Observed, Receipt}; -use crate::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue, read_fixed}; -use crate::identity::RequestId; -use crate::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, TenantId, partition_for_shard, shard_for_scope, -}; -use crate::registry::{Command, Query, RegistryBuilder}; -use crate::registry::{CommandContext, CommandResult, QueryContext}; - -mod codec; - -use super::{ - MAX_WORKFLOW_BYTES, WorkflowControl, WorkflowControlAction, WorkflowDefinition, - WorkflowOutcome, WorkflowRun, WorkflowSignal, WorkflowStart, WorkflowStatus, workflow_cancel, - workflow_control, workflow_signal, workflow_start, workflow_state, -}; - -/// Compile-time namespace, definition and operation IDs for one Workflow module. -pub trait WorkflowModule: Send + Sync + 'static { - /// Module the Workflow surfaces register under. - const MODULE: &'static str; - /// Namespace that owns this Workflow module. - const NAMESPACE: NamespaceId; - /// Definition the module currently serves. - const CURRENT_DEFINITION: &'static dyn WorkflowDefinition; - /// Definitions registered in this compiled image. - const DEFINITIONS: &'static [&'static dyn WorkflowDefinition]; - /// Codec version of the Workflow surfaces. - const CODEC_VERSION: u32 = 1; - /// Command id that starts a run. - const START_COMMAND_ID: u32; - /// Command id that delivers a signal. - const SIGNAL_COMMAND_ID: u32; - /// Command id that cancels a run. - const CANCEL_COMMAND_ID: u32; - /// Command id that pauses, resumes, or restarts a run. - const CONTROL_COMMAND_ID: u32; - /// Query id that reads a run. - const GET_QUERY_ID: u32; -} - -/// Registers every executable definition and the module's typed Workflow bindings. -pub fn register_workflow(registry: &mut RegistryBuilder) -> crate::Result<()> { - for definition in definitions::()? { - registry.bind_workflow_definition(M::MODULE, *definition)?; - } - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_query::>() -} - -/// Typed Workflow operator control bound to immutable module operation IDs. -pub struct WorkflowControlCommand(PhantomData M>); - -impl Command for WorkflowControlCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::CONTROL_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = WorkflowControl; - type Output = WorkflowOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let source = context.target().clone(); - classify(workflow_control( - context.primitive_transaction(), - &source, - context.now_ms(), - &input, - M::CURRENT_DEFINITION, - )?) - } -} - -/// Typed Workflow start bound to immutable module operation IDs. -pub struct WorkflowStartCommand(PhantomData M>); - -impl Command for WorkflowStartCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::START_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = WorkflowStart; - type Output = WorkflowOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let source = context.target().clone(); - classify(workflow_start( - context.primitive_transaction(), - &source, - context.now_ms(), - &input, - M::CURRENT_DEFINITION, - )?) - } -} - -/// Typed Workflow signal bound to immutable module operation IDs. -pub struct WorkflowSignalCommand(PhantomData M>); - -impl Command for WorkflowSignalCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::SIGNAL_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = WorkflowSignal; - type Output = WorkflowOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - let source = context.target().clone(); - let Some(definition) = - definition_for_workflow::(context.primitive_transaction(), &input.workflow_id)? - else { - return classify(WorkflowOutcome::RunMismatch); - }; - classify(workflow_signal( - context.primitive_transaction(), - &source, - context.now_ms(), - &input, - definition, - )?) - } -} - -pub(super) fn definitions() --> crate::Result<&'static [&'static dyn WorkflowDefinition]> { - if M::DEFINITIONS.is_empty() - || !M::DEFINITIONS - .iter() - .any(|definition| definition.digest() == M::CURRENT_DEFINITION.digest()) - { - return Err(crate::Error::Registry( - "current workflow definition is absent from inventory", - )); - } - Ok(M::DEFINITIONS) -} - -pub(super) fn definition( - digest: Digest, -) -> crate::Result<&'static dyn WorkflowDefinition> { - definitions::()? - .iter() - .copied() - .find(|definition| definition.digest() == digest) - .ok_or(crate::Error::Command( - "workflow definition digest is unavailable", - )) -} - -fn definition_for_workflow( - transaction: &crab_ltx::rusqlite::Transaction<'_>, - workflow_id: &[u8], -) -> crate::Result> { - super::workflow_definition_digest(transaction, workflow_id)? - .map(definition::) - .transpose() -} - -/// Typed Workflow cancellation bound to immutable module operation IDs. -pub struct WorkflowCancelCommand(PhantomData M>); - -impl Command for WorkflowCancelCommand { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::CANCEL_COMMAND_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = WorkflowSignal; - type Output = WorkflowOutcome; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crate::Result> { - classify(workflow_cancel( - context.primitive_transaction(), - context.now_ms(), - &input, - )?) - } -} - -/// Bounded current-state query for one workflow identity. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct WorkflowGetRequest { - /// Workflow identity to read. - pub workflow_id: Vec, -} - -/// Typed Workflow state query bound to immutable module operation IDs. -pub struct WorkflowGetQuery(PhantomData M>); - -impl Query for WorkflowGetQuery { - const MODULE: &'static str = M::MODULE; - const ID: u32 = M::GET_QUERY_ID; - const CODEC_VERSION: u32 = M::CODEC_VERSION; - type Input = WorkflowGetRequest; - type Output = Option; - - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> crate::Result { - workflow_state(context.primitive_connection(), &input.workflow_id) - } -} - -/// Authorized Workflow capability with deterministic workflow-ID sharding. -pub struct WorkflowNamespace { - client: CellClient, - tenant: TenantId, - application: ApplicationId, - shards: u32, - module: PhantomData M>, -} - -impl Clone for WorkflowNamespace { - fn clone(&self) -> Self { - Self { - client: self.client.clone(), - tenant: self.tenant, - application: self.application, - shards: self.shards, - module: PhantomData, - } - } -} - -impl WorkflowNamespace { - /// Creates a Workflow capability from its compiled namespace topology. - pub fn new( - client: CellClient, - tenant: TenantId, - application: ApplicationId, - ) -> crate::Result { - let shards = client.require_namespace(M::NAMESPACE, M::MODULE, CatalogRole::Workflow)?; - Ok(Self { - client, - tenant, - application, - shards, - module: PhantomData, - }) - } - - /// Starts one workflow with the runtime mutation identity as its run identity. - pub async fn start( - &self, - identity: crate::cell::executor::MutationIdentity, - workflow_id: Vec, - event: Vec, - ) -> std::result::Result, InvocationError> { - let target = self - .target(&workflow_id) - .map_err(InvocationError::NotStarted)?; - let input = WorkflowStart { - workflow_id, - request_id: identity.request_id, - event, - }; - self.client - .command::>(&target, identity, input) - .await - } - - /// Delivers one idempotent external signal to its workflow-ID shard. - pub async fn signal( - &self, - identity: crate::cell::executor::MutationIdentity, - signal: WorkflowSignal, - ) -> std::result::Result, InvocationError> { - let target = self - .target(&signal.workflow_id) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>(&target, identity, signal) - .await - } - - /// Cancels one running workflow and its outstanding local work. - pub async fn cancel( - &self, - identity: crate::cell::executor::MutationIdentity, - signal: WorkflowSignal, - ) -> std::result::Result, InvocationError> { - let target = self - .target(&signal.workflow_id) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>(&target, identity, signal) - .await - } - - /// Pauses a quiescent run so timers and new activity claims stop. - pub async fn pause( - &self, - identity: crate::cell::executor::MutationIdentity, - workflow_id: Vec, - run_id: [u8; 16], - ) -> std::result::Result, InvocationError> { - self.control(identity, workflow_id, run_id, WorkflowControlAction::Pause) - .await - } - - /// Resumes a paused run without changing its deterministic history. - pub async fn resume( - &self, - identity: crate::cell::executor::MutationIdentity, - workflow_id: Vec, - run_id: [u8; 16], - ) -> std::result::Result, InvocationError> { - self.control(identity, workflow_id, run_id, WorkflowControlAction::Resume) - .await - } - - /// Replaces a terminal run with a new run of the current definition. - pub async fn restart( - &self, - identity: crate::cell::executor::MutationIdentity, - workflow_id: Vec, - run_id: [u8; 16], - event: Vec, - ) -> std::result::Result, InvocationError> { - let request_id = identity.request_id; - self.control( - identity, - workflow_id, - run_id, - WorkflowControlAction::Restart { request_id, event }, - ) - .await - } - - async fn control( - &self, - identity: crate::cell::executor::MutationIdentity, - workflow_id: Vec, - run_id: [u8; 16], - action: WorkflowControlAction, - ) -> std::result::Result, InvocationError> { - let target = self - .target(&workflow_id) - .map_err(InvocationError::NotStarted)?; - self.client - .command::>( - &target, - identity, - WorkflowControl { - workflow_id, - run_id, - action, - }, - ) - .await - } - - /// Reads one workflow at an optional minimum publication receipt. - pub async fn state( - &self, - workflow_id: Vec, - minimum: Option, - ) -> std::result::Result>, InvocationError>> - { - let target = self - .target(&workflow_id) - .map_err(InvocationError::NotStarted)?; - self.client - .query::>(&target, minimum, WorkflowGetRequest { workflow_id }) - .await - } - - fn target(&self, workflow_id: &[u8]) -> crate::Result { - let shard = shard_for_scope(M::NAMESPACE, workflow_id, self.shards)?; - CellTarget::new( - self.tenant, - self.application, - M::NAMESPACE, - &partition_for_shard(shard), - ) - } -} - -fn classify(outcome: WorkflowOutcome) -> crate::Result> { - Ok(match outcome { - WorkflowOutcome::Applied { .. } | WorkflowOutcome::Duplicate { .. } => { - CommandResult::Success(outcome) - } - WorkflowOutcome::AlreadyExists - | WorkflowOutcome::IdentityConflict - | WorkflowOutcome::RunMismatch - | WorkflowOutcome::NotRunning - | WorkflowOutcome::NotDue - | WorkflowOutcome::Busy => CommandResult::Rejected(outcome), - }) -} diff --git a/crates/crab-cell-runtime/src/primitives/workflow/api/codec.rs b/crates/crab-cell-runtime/src/primitives/workflow/api/codec.rs deleted file mode 100644 index 69a2393a1..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow/api/codec.rs +++ /dev/null @@ -1,449 +0,0 @@ -//! Workflow wire codecs for start, signal, control, outcome, and run payloads. - -use super::*; - -const APPLIED_TAG: u8 = 0; -const DUPLICATE_TAG: u8 = 1; -const ALREADY_EXISTS_TAG: u8 = 2; -const IDENTITY_CONFLICT_TAG: u8 = 3; -const RUN_MISMATCH_TAG: u8 = 4; -const NOT_RUNNING_TAG: u8 = 5; -const NOT_DUE_TAG: u8 = 6; -const BUSY_TAG: u8 = 7; -const CONTROL_PAUSE_TAG: u8 = 0; -const CONTROL_RESUME_TAG: u8 = 1; -const CONTROL_RESTART_TAG: u8 = 2; - -impl WireValue for WorkflowStart { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.workflow_id)?; - encoder.write_bytes(self.request_id.as_bytes())?; - encoder.write_bytes(&self.event) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - workflow_id: decoder.read_bytes()?.to_vec(), - request_id: RequestId::from_bytes(read_fixed(decoder, "workflow request ID length")?), - event: decoder.read_bytes()?.to_vec(), - }) - } -} - -impl WireValue for WorkflowSignal { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.workflow_id)?; - encoder.write_bytes(&self.run_id)?; - encoder.write_bytes(&self.signal_id)?; - encoder.write_bytes(&self.event) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - workflow_id: decoder.read_bytes()?.to_vec(), - run_id: read_fixed(decoder, "workflow run ID length")?, - signal_id: read_fixed(decoder, "workflow signal ID length")?, - event: decoder.read_bytes()?.to_vec(), - }) - } -} - -impl WireValue for WorkflowOutcome { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - match self { - Self::Applied { - run_id, - status, - event_sequence, - } => encode_outcome(APPLIED_TAG, *run_id, *status, *event_sequence, encoder), - Self::Duplicate { - run_id, - status, - event_sequence, - } => encode_outcome(DUPLICATE_TAG, *run_id, *status, *event_sequence, encoder), - Self::AlreadyExists => encoder.write_u8(ALREADY_EXISTS_TAG), - Self::IdentityConflict => encoder.write_u8(IDENTITY_CONFLICT_TAG), - Self::RunMismatch => encoder.write_u8(RUN_MISMATCH_TAG), - Self::NotRunning => encoder.write_u8(NOT_RUNNING_TAG), - Self::NotDue => encoder.write_u8(NOT_DUE_TAG), - Self::Busy => encoder.write_u8(BUSY_TAG), - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - APPLIED_TAG => { - decode_outcome(decoder, |run_id, status, event_sequence| Self::Applied { - run_id, - status, - event_sequence, - }) - } - DUPLICATE_TAG => { - decode_outcome(decoder, |run_id, status, event_sequence| Self::Duplicate { - run_id, - status, - event_sequence, - }) - } - ALREADY_EXISTS_TAG => Ok(Self::AlreadyExists), - IDENTITY_CONFLICT_TAG => Ok(Self::IdentityConflict), - RUN_MISMATCH_TAG => Ok(Self::RunMismatch), - NOT_RUNNING_TAG => Ok(Self::NotRunning), - NOT_DUE_TAG => Ok(Self::NotDue), - BUSY_TAG => Ok(Self::Busy), - _ => Err(CodecError::Invalid("invalid workflow outcome tag")), - } - } -} - -impl WireValue for WorkflowControl { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.workflow_id)?; - encoder.write_bytes(&self.run_id)?; - match &self.action { - WorkflowControlAction::Pause => encoder.write_u8(CONTROL_PAUSE_TAG), - WorkflowControlAction::Resume => encoder.write_u8(CONTROL_RESUME_TAG), - WorkflowControlAction::Restart { request_id, event } => { - encoder.write_u8(CONTROL_RESTART_TAG)?; - encoder.write_bytes(request_id.as_bytes())?; - encoder.write_bytes(event) - } - } - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let workflow_id = decoder.read_bytes()?.to_vec(); - let run_id = read_fixed(decoder, "workflow run ID length")?; - let action = match decoder.read_u8()? { - CONTROL_PAUSE_TAG => WorkflowControlAction::Pause, - CONTROL_RESUME_TAG => WorkflowControlAction::Resume, - CONTROL_RESTART_TAG => WorkflowControlAction::Restart { - request_id: RequestId::from_bytes(read_fixed( - decoder, - "workflow restart request ID length", - )?), - event: decoder.read_bytes()?.to_vec(), - }, - _ => return Err(CodecError::Invalid("invalid workflow control action tag")), - }; - Ok(Self { - workflow_id, - run_id, - action, - }) - } -} - -impl WireValue for WorkflowGetRequest { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_bytes(&self.workflow_id) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - Ok(Self { - workflow_id: decoder.read_bytes()?.to_vec(), - }) - } -} - -impl WireValue for WorkflowRun { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - validate_run(self)?; - encoder.write_bytes(&self.workflow_id)?; - encoder.write_bytes(&self.run_id)?; - encoder.write_bytes(self.definition_digest.as_bytes())?; - encode_status(self.status, encoder)?; - encoder.write_bytes(&self.state)?; - encoder.write_u64(self.event_sequence)?; - self.result.encode(encoder) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - let run = Self { - workflow_id: decoder.read_bytes()?.to_vec(), - run_id: read_fixed(decoder, "workflow run ID length")?, - definition_digest: Digest::from_bytes(read_fixed( - decoder, - "workflow definition digest length", - )?), - status: decode_status(decoder)?, - state: decoder.read_bytes()?.to_vec(), - event_sequence: decoder.read_u64()?, - result: Option::>::decode(decoder)?, - }; - validate_run(&run)?; - Ok(run) - } -} - -fn encode_outcome( - tag: u8, - run_id: [u8; 16], - status: WorkflowStatus, - event_sequence: u64, - encoder: &mut BoundedEncoder, -) -> Result<(), CodecError> { - if event_sequence == 0 { - return Err(CodecError::Invalid("invalid workflow event sequence")); - } - encoder.write_u8(tag)?; - encoder.write_bytes(&run_id)?; - encode_status(status, encoder)?; - encoder.write_u64(event_sequence) -} - -fn decode_outcome( - decoder: &mut BoundedDecoder<'_>, - outcome: impl FnOnce([u8; 16], WorkflowStatus, u64) -> WorkflowOutcome, -) -> Result { - let run_id = read_fixed(decoder, "workflow run ID length")?; - let status = decode_status(decoder)?; - let event_sequence = decoder.read_u64()?; - if event_sequence == 0 { - return Err(CodecError::Invalid("invalid workflow event sequence")); - } - Ok(outcome(run_id, status, event_sequence)) -} - -fn encode_status(status: WorkflowStatus, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_u8(match status { - WorkflowStatus::Running => 0, - WorkflowStatus::Completed => 1, - WorkflowStatus::Failed => 2, - WorkflowStatus::Cancelled => 3, - WorkflowStatus::Paused => 4, - }) -} - -fn decode_status(decoder: &mut BoundedDecoder<'_>) -> Result { - match decoder.read_u8()? { - 0 => Ok(WorkflowStatus::Running), - 1 => Ok(WorkflowStatus::Completed), - 2 => Ok(WorkflowStatus::Failed), - 3 => Ok(WorkflowStatus::Cancelled), - 4 => Ok(WorkflowStatus::Paused), - _ => Err(CodecError::Invalid("invalid workflow status tag")), - } -} - -fn validate_run(run: &WorkflowRun) -> Result<(), CodecError> { - let result_bytes = run.result.as_ref().map_or(0, Vec::len); - if run.workflow_id.is_empty() - || run.workflow_id.len() > 1024 - || run - .definition_digest - .as_bytes() - .iter() - .all(|byte| *byte == 0) - || run.event_sequence == 0 - || run.state.len() > MAX_WORKFLOW_BYTES - || result_bytes > MAX_WORKFLOW_BYTES - || run.state.len().saturating_add(result_bytes) > MAX_WORKFLOW_BYTES - || (matches!(run.status, WorkflowStatus::Running | WorkflowStatus::Paused) - && run.result.is_some()) - { - return Err(CodecError::Invalid("invalid workflow run")); - } - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::codec::roundtrip; - use crab_ltx::rusqlite::Connection; - - struct TestDefinition { - digest: Digest, - prefix: &'static [u8], - } - - impl WorkflowDefinition for TestDefinition { - fn digest(&self) -> Digest { - self.digest - } - - fn transition( - &self, - _state: &[u8], - event: &[u8], - _context: crate::primitives::workflow::WorkflowContext, - ) -> crate::Result { - let mut state = self.prefix.to_vec(); - state.extend_from_slice(event); - Ok(crate::primitives::workflow::WorkflowDecision { - status: WorkflowStatus::Running, - state, - result: None, - actions: Vec::new(), - }) - } - } - - static OLD_DEFINITION: TestDefinition = TestDefinition { - digest: Digest::from_bytes([11; 32]), - prefix: b"old:", - }; - static NEW_DEFINITION: TestDefinition = TestDefinition { - digest: Digest::from_bytes([12; 32]), - prefix: b"new:", - }; - static TEST_DEFINITIONS: [&dyn WorkflowDefinition; 2] = [&OLD_DEFINITION, &NEW_DEFINITION]; - - struct RolloverWorkflow; - - impl WorkflowModule for RolloverWorkflow { - const MODULE: &'static str = "rollover"; - const NAMESPACE: NamespaceId = NamespaceId::from_bytes([13; 16]); - const CURRENT_DEFINITION: &'static dyn WorkflowDefinition = &NEW_DEFINITION; - const DEFINITIONS: &'static [&'static dyn WorkflowDefinition] = &TEST_DEFINITIONS; - const START_COMMAND_ID: u32 = 1; - const SIGNAL_COMMAND_ID: u32 = 2; - const CANCEL_COMMAND_ID: u32 = 3; - const CONTROL_COMMAND_ID: u32 = 4; - const GET_QUERY_ID: u32 = 1; - } - - #[test] - fn workflow_codecs_roundtrip_commands_outcomes_and_state() { - roundtrip(WorkflowStart { - workflow_id: b"build-42".to_vec(), - request_id: RequestId::from_bytes([1; 16]), - event: b"start".to_vec(), - }); - roundtrip(WorkflowSignal { - workflow_id: b"build-42".to_vec(), - run_id: [2; 16], - signal_id: [3; 16], - event: b"finish".to_vec(), - }); - roundtrip(WorkflowOutcome::Applied { - run_id: [2; 16], - status: WorkflowStatus::Running, - event_sequence: 1, - }); - for outcome in [ - WorkflowOutcome::AlreadyExists, - WorkflowOutcome::IdentityConflict, - WorkflowOutcome::RunMismatch, - WorkflowOutcome::NotRunning, - WorkflowOutcome::NotDue, - WorkflowOutcome::Busy, - ] { - roundtrip(outcome); - } - roundtrip(WorkflowGetRequest { - workflow_id: b"build-42".to_vec(), - }); - roundtrip(WorkflowControl { - workflow_id: b"build-42".to_vec(), - run_id: [2; 16], - action: WorkflowControlAction::Restart { - request_id: RequestId::from_bytes([8; 16]), - event: b"again".to_vec(), - }, - }); - roundtrip(WorkflowOutcome::Applied { - run_id: [2; 16], - status: WorkflowStatus::Paused, - event_sequence: 1, - }); - roundtrip(Some(WorkflowRun { - workflow_id: b"build-42".to_vec(), - run_id: [2; 16], - definition_digest: Digest::from_bytes([4; 32]), - status: WorkflowStatus::Completed, - state: b"done".to_vec(), - event_sequence: 2, - result: Some(b"ok".to_vec()), - })); - } - - #[test] - fn workflow_decoder_rejects_invalid_ids_sequences_and_state_shape() { - let mut invalid = BoundedEncoder::new(32).unwrap(); - invalid.write_u8(APPLIED_TAG).unwrap(); - invalid.write_bytes(&[1; 15]).unwrap(); - let bytes = invalid.finish(); - let mut decoder = BoundedDecoder::new(&bytes, 32).unwrap(); - assert!(matches!( - WorkflowOutcome::decode(&mut decoder), - Err(CodecError::Invalid("workflow run ID length")) - )); - - let run = WorkflowRun { - workflow_id: b"build-42".to_vec(), - run_id: [2; 16], - definition_digest: Digest::from_bytes([4; 32]), - status: WorkflowStatus::Running, - state: Vec::new(), - event_sequence: 1, - result: Some(b"not allowed".to_vec()), - }; - let mut encoder = BoundedEncoder::new(128).unwrap(); - assert!(matches!( - run.encode(&mut encoder), - Err(CodecError::Invalid("invalid workflow run")) - )); - } - - #[test] - fn persisted_definition_digest_dispatches_to_retained_old_code() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - let source = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - RolloverWorkflow::NAMESPACE, - b"rollover", - ) - .unwrap(); - crate::cell::schema::install_runtime_schema_in( - &transaction, - source.cell_id(), - crate::identity::IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - crate::primitives::workflow::install_workflow_schema(&transaction).unwrap(); - let request = WorkflowStart { - workflow_id: b"old-run".to_vec(), - request_id: RequestId::from_bytes([14; 16]), - event: b"start".to_vec(), - }; - let started = - super::super::workflow_start(&transaction, &source, 10, &request, &OLD_DEFINITION) - .unwrap(); - let WorkflowOutcome::Applied { run_id, .. } = started else { - panic!("old workflow did not start"); - }; - - let retained = - definition_for_workflow::(&transaction, &request.workflow_id) - .unwrap() - .unwrap(); - assert_eq!(retained.digest(), OLD_DEFINITION.digest()); - super::super::workflow_signal( - &transaction, - &source, - 11, - &WorkflowSignal { - workflow_id: request.workflow_id.clone(), - run_id, - signal_id: [15; 16], - event: b"continue".to_vec(), - }, - retained, - ) - .unwrap(); - let run = super::super::workflow_state(&transaction, &request.workflow_id) - .unwrap() - .unwrap(); - assert_eq!(run.state, b"old:continue"); - - let current = - definition::(RolloverWorkflow::CURRENT_DEFINITION.digest()).unwrap(); - assert_eq!(current.digest(), NEW_DEFINITION.digest()); - } -} diff --git a/crates/crab-cell-runtime/src/primitives/workflow/commands.rs b/crates/crab-cell-runtime/src/primitives/workflow/commands.rs deleted file mode 100644 index a8e44568a..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow/commands.rs +++ /dev/null @@ -1,306 +0,0 @@ -//! The Workflow command surface: install, start, signal, cancel, control. -//! -//! Each entry point prepares one transition, applies its decision effects, and -//! commits the event and run row in the caller's transaction, so a caller -//! never observes a half-applied decision. - -use super::*; - -/// Installs the exact version-one Workflow schema during bootstrap or migration. -pub fn install_workflow_schema(transaction: &Transaction<'_>) -> Result<()> { - transaction.execute_batch(WORKFLOW_SCHEMA)?; - Ok(()) -} - -/// Starts one workflow and applies its first deterministic transition atomically. -pub fn workflow_start( - transaction: &Transaction<'_>, - source: &CellTarget, - now_ms: i64, - request: &WorkflowStart, - definition: &dyn WorkflowDefinition, -) -> Result { - let mut effects = next_effect_batch(transaction, source, now_ms)?; - workflow_start_with_effects( - transaction, - &mut effects, - source, - now_ms, - request, - definition, - ) -} - -fn workflow_start_with_effects( - transaction: &Transaction<'_>, - effects: &mut EffectBatch, - source: &CellTarget, - now_ms: i64, - request: &WorkflowStart, - definition: &dyn WorkflowDefinition, -) -> Result { - validate_now(now_ms)?; - validate_identifier(&request.workflow_id)?; - validate_event(&request.event)?; - if transaction - .query_row( - "SELECT 1 FROM workflow_runs WHERE workflow_id = ?1", - [request.workflow_id.as_slice()], - |_| Ok(()), - ) - .optional()? - .is_some() - { - return Ok(WorkflowOutcome::AlreadyExists); - } - - let run_id = run_id(source.namespace(), request.request_id); - let sequence = next_sequence(transaction, 0)?; - let decision = definition.transition( - &[], - &request.event, - WorkflowContext { - source: source.clone(), - run_id, - event_sequence: sequence, - now_ms, - }, - )?; - validate_decision(&decision, source, definition.effect_targets(), now_ms)?; - transaction.execute( - "INSERT INTO workflow_runs(workflow_id, run_id, definition_digest, status, state, event_sequence, result, completed_at_ms) VALUES (?1, ?2, ?3, 0, X'', 0, NULL, NULL)", - ( - request.workflow_id.as_slice(), - run_id.as_slice(), - definition.digest().as_bytes().as_slice(), - ), - )?; - insert_event( - transaction, - run_id, - sequence, - event_id(run_id, request.request_id.as_bytes()), - &request.event, - )?; - apply_decision(transaction, effects, run_id, sequence, now_ms, decision) -} - -/// Applies one idempotent signal to a running workflow. -pub fn workflow_signal( - transaction: &Transaction<'_>, - source: &CellTarget, - now_ms: i64, - signal: &WorkflowSignal, - definition: &dyn WorkflowDefinition, -) -> Result { - let mut effects = next_effect_batch(transaction, source, now_ms)?; - workflow_signal_with_effects( - transaction, - &mut effects, - source, - now_ms, - signal, - definition, - ) -} - -fn workflow_signal_with_effects( - transaction: &Transaction<'_>, - effects: &mut EffectBatch, - source: &CellTarget, - now_ms: i64, - signal: &WorkflowSignal, - definition: &dyn WorkflowDefinition, -) -> Result { - validate_now(now_ms)?; - validate_identifier(&signal.workflow_id)?; - validate_event(&signal.event)?; - let run = load_run(transaction, &signal.workflow_id)?; - let Some(run) = run else { - return Ok(WorkflowOutcome::RunMismatch); - }; - if run.run_id != signal.run_id { - return Ok(WorkflowOutcome::RunMismatch); - } - verify_definition(&run, definition)?; - let id = event_id(run.run_id, &signal.signal_id); - if let Some(digest) = stored_event_digest(transaction, run.run_id, id)? { - return if digest == event_digest(&signal.event) { - Ok(WorkflowOutcome::Duplicate { - run_id: run.run_id, - status: run.status, - event_sequence: run.event_sequence, - }) - } else { - Ok(WorkflowOutcome::IdentityConflict) - }; - } - if run.status != WorkflowStatus::Running { - return Ok(WorkflowOutcome::NotRunning); - } - let (sequence, decision) = - prepare_transition(transaction, source, &run, definition, &signal.event, now_ms)?; - commit_transition( - transaction, - effects, - run.run_id, - sequence, - id, - &signal.event, - now_ms, - decision, - ) -} - -/// Cancels a running or paused workflow and its outstanding local work. -pub fn workflow_cancel( - transaction: &Transaction<'_>, - now_ms: i64, - signal: &WorkflowSignal, -) -> Result { - validate_now(now_ms)?; - validate_identifier(&signal.workflow_id)?; - validate_event(&signal.event)?; - let Some(run) = load_run(transaction, &signal.workflow_id)? else { - return Ok(WorkflowOutcome::RunMismatch); - }; - if run.run_id != signal.run_id { - return Ok(WorkflowOutcome::RunMismatch); - } - let id = event_id(run.run_id, &signal.signal_id); - if let Some(digest) = stored_event_digest(transaction, run.run_id, id)? { - return if digest == event_digest(&signal.event) { - Ok(WorkflowOutcome::Duplicate { - run_id: run.run_id, - status: run.status, - event_sequence: run.event_sequence, - }) - } else { - Ok(WorkflowOutcome::IdentityConflict) - }; - } - if !matches!(run.status, WorkflowStatus::Running | WorkflowStatus::Paused) { - return Ok(WorkflowOutcome::NotRunning); - } - let sequence = next_sequence(transaction, run.event_sequence)?; - insert_event(transaction, run.run_id, sequence, id, &signal.event)?; - if transaction.execute( - "UPDATE workflow_runs SET status = 3, event_sequence = ?1, completed_at_ms = ?2 WHERE run_id = ?3 AND status IN (0, 4)", - (sequence as i64, now_ms, run.run_id.as_slice()), - )? != 1 - { - return Err(Error::Command("workflow run changed during cancellation")); - } - cancel_outstanding(transaction, run.run_id)?; - Ok(WorkflowOutcome::Applied { - run_id: run.run_id, - status: WorkflowStatus::Cancelled, - event_sequence: sequence, - }) -} - -/// Pauses, resumes, or restarts one exact workflow run. -pub fn workflow_control( - transaction: &Transaction<'_>, - source: &CellTarget, - now_ms: i64, - request: &WorkflowControl, - definition: &dyn WorkflowDefinition, -) -> Result { - validate_now(now_ms)?; - validate_identifier(&request.workflow_id)?; - let Some(run) = load_run(transaction, &request.workflow_id)? else { - return Ok(WorkflowOutcome::RunMismatch); - }; - if run.run_id != request.run_id { - return Ok(WorkflowOutcome::RunMismatch); - } - match &request.action { - WorkflowControlAction::Pause => { - if run.status != WorkflowStatus::Running { - return Ok(WorkflowOutcome::NotRunning); - } - let leased = transaction.query_row( - "SELECT count(*) FROM workflow_activities WHERE run_id = ?1 AND state = 1", - [run.run_id.as_slice()], - |row| row.get::<_, i64>(0), - )?; - if leased != 0 { - return Ok(WorkflowOutcome::Busy); - } - if transaction.execute( - "UPDATE workflow_runs SET status = 4 WHERE run_id = ?1 AND status = 0", - [run.run_id.as_slice()], - )? != 1 - { - return Err(Error::Command("workflow changed during pause")); - } - Ok(WorkflowOutcome::Applied { - run_id: run.run_id, - status: WorkflowStatus::Paused, - event_sequence: run.event_sequence, - }) - } - WorkflowControlAction::Resume => { - if run.status != WorkflowStatus::Paused { - return Ok(WorkflowOutcome::NotRunning); - } - if transaction.execute( - "UPDATE workflow_runs SET status = 0 WHERE run_id = ?1 AND status = 4", - [run.run_id.as_slice()], - )? != 1 - { - return Err(Error::Command("workflow changed during resume")); - } - Ok(WorkflowOutcome::Applied { - run_id: run.run_id, - status: WorkflowStatus::Running, - event_sequence: run.event_sequence, - }) - } - WorkflowControlAction::Restart { request_id, event } => { - if matches!(run.status, WorkflowStatus::Running | WorkflowStatus::Paused) { - return Ok(WorkflowOutcome::Busy); - } - if run_id(source.namespace(), *request_id) == run.run_id { - return Ok(WorkflowOutcome::IdentityConflict); - } - validate_event(event)?; - delete_workflow_run(transaction, run.run_id)?; - workflow_start( - transaction, - source, - now_ms, - &WorkflowStart { - workflow_id: request.workflow_id.clone(), - request_id: *request_id, - event: event.clone(), - }, - definition, - ) - } - } -} - -fn delete_workflow_run(transaction: &Transaction<'_>, run_id: [u8; 16]) -> Result<()> { - transaction.execute( - "DELETE FROM workflow_activities WHERE run_id = ?1", - [run_id.as_slice()], - )?; - transaction.execute( - "DELETE FROM workflow_timers WHERE run_id = ?1", - [run_id.as_slice()], - )?; - transaction.execute( - "DELETE FROM workflow_events WHERE run_id = ?1", - [run_id.as_slice()], - )?; - if transaction.execute( - "DELETE FROM workflow_runs WHERE run_id = ?1 AND status BETWEEN 1 AND 3", - [run_id.as_slice()], - )? != 1 - { - return Err(Error::Command("workflow changed during restart")); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/primitives/workflow/maintenance.rs b/crates/crab-cell-runtime/src/primitives/workflow/maintenance.rs deleted file mode 100644 index 4246281f5..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow/maintenance.rs +++ /dev/null @@ -1,244 +0,0 @@ -//! Timer firing and activity-expiry maintenance. -//! -//! Both entry points apply their event, decision effects, and terminal -//! activity bookkeeping inside the caller's transaction, so a Sweeper or -//! Cron Tick can never publish a half-applied timer. - -use super::*; - -/// Fires one due timer and applies its event in the same transaction. -pub fn workflow_fire_timer( - transaction: &Transaction<'_>, - source: &CellTarget, - now_ms: i64, - run_id: [u8; 16], - timer_id: [u8; 16], - definition: &dyn WorkflowDefinition, -) -> Result { - let mut effects = next_effect_batch(transaction, source, now_ms)?; - workflow_fire_timer_with_effects( - transaction, - &mut effects, - source, - now_ms, - run_id, - timer_id, - definition, - ) -} - -fn workflow_fire_timer_with_effects( - transaction: &Transaction<'_>, - effects: &mut EffectBatch, - source: &CellTarget, - now_ms: i64, - run_id: [u8; 16], - timer_id: [u8; 16], - definition: &dyn WorkflowDefinition, -) -> Result { - validate_now(now_ms)?; - let Some(run) = load_run_by_id(transaction, run_id)? else { - return Ok(WorkflowOutcome::RunMismatch); - }; - verify_definition(&run, definition)?; - let timer = transaction - .query_row( - "SELECT due_at_ms, state FROM workflow_timers WHERE run_id = ?1 AND timer_id = ?2", - (run_id.as_slice(), timer_id.as_slice()), - |row| Ok((row.get::<_, i64>(0)?, row.get::<_, i64>(1)?)), - ) - .optional()?; - let Some((due_at_ms, state)) = timer else { - return Ok(WorkflowOutcome::RunMismatch); - }; - if state == 1 { - return Ok(WorkflowOutcome::Duplicate { - run_id, - status: run.status, - event_sequence: run.event_sequence, - }); - } - if state != 0 || run.status != WorkflowStatus::Running { - return Ok(WorkflowOutcome::NotRunning); - } - if due_at_ms > now_ms { - return Ok(WorkflowOutcome::NotDue); - } - let mut event = b"timer\0".to_vec(); - event.extend_from_slice(&timer_id); - let (sequence, decision) = - prepare_transition(transaction, source, &run, definition, &event, now_ms)?; - if transaction.execute( - "UPDATE workflow_timers SET state = 1 WHERE run_id = ?1 AND timer_id = ?2 AND state = 0 AND due_at_ms <= ?3", - (run_id.as_slice(), timer_id.as_slice(), now_ms), - )? != 1 - { - return Err(Error::Command("workflow timer changed during serialized fire")); - } - commit_transition( - transaction, - effects, - run_id, - sequence, - timer_event_id(run_id, timer_id), - &event, - now_ms, - decision, - ) -} - -pub(crate) fn workflow_fail_one_expired_activity( - transaction: &Transaction<'_>, - effects: &mut EffectBatch, - source: &CellTarget, - now_ms: i64, - definitions: &[&'static dyn WorkflowDefinition], -) -> Result { - validate_now(now_ms)?; - let candidate = transaction - .query_row( - "SELECT a.run_id, a.activity_id, a.attempt, a.expires_at_ms, r.definition_digest FROM workflow_activities a INDEXED BY activities_due JOIN workflow_runs r ON r.run_id = a.run_id WHERE ((a.state = 0 AND (a.expires_at_ms <= ?1 OR a.attempt >= ?2)) OR (a.state = 1 AND a.lease_until_ms <= ?1 AND (a.expires_at_ms <= ?1 OR a.attempt >= ?2))) AND r.status = 0 ORDER BY a.due_at_ms, a.run_id, a.activity_id LIMIT 1", - (now_ms, i64::from(activity::MAX_ATTEMPTS)), - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, i64>(2)?, - row.get::<_, i64>(3)?, - row.get::<_, Vec>(4)?, - )) - }, - ) - .optional()?; - let Some((run_id, activity_id, attempt, expires_at_ms, digest)) = candidate else { - return Ok(false); - }; - let run_id = exact_array(run_id, "invalid stored workflow run ID")?; - let activity_id = exact_array(activity_id, "invalid stored workflow activity ID")?; - let digest = Digest::from_bytes(exact_array( - digest, - "invalid stored workflow definition digest", - )?); - let definition = retained_definition(definitions, digest)?; - let run = load_run_by_id(transaction, run_id)? - .ok_or(Error::Command("workflow activity references a missing run"))?; - let details = if expires_at_ms <= now_ms { - b"expired".as_slice() - } else if attempt >= i64::from(activity::MAX_ATTEMPTS) { - b"attempts-exhausted".as_slice() - } else { - return Err(Error::Command("invalid due workflow activity")); - }; - let event = activity_failure_event(activity_id, details); - let (sequence, decision) = - prepare_transition(transaction, source, &run, definition, &event, now_ms)?; - if transaction.execute( - "UPDATE workflow_activities SET state = 3, token = NULL, lease_until_ms = NULL WHERE run_id = ?1 AND activity_id = ?2 AND ((state = 0 AND (expires_at_ms <= ?3 OR attempt >= ?4)) OR (state = 1 AND lease_until_ms <= ?3 AND (expires_at_ms <= ?3 OR attempt >= ?4)))", - ( - run_id.as_slice(), - activity_id.as_slice(), - now_ms, - i64::from(activity::MAX_ATTEMPTS), - ), - )? != 1 - { - return Err(Error::Command( - "workflow activity changed during terminal failure", - )); - } - commit_transition( - transaction, - effects, - run_id, - sequence, - activity_failure_event_id(run_id, activity_id), - &event, - now_ms, - decision, - )?; - Ok(true) -} - -pub(crate) fn workflow_fire_one_due_timer( - transaction: &Transaction<'_>, - effects: &mut EffectBatch, - source: &CellTarget, - now_ms: i64, - definitions: &[&'static dyn WorkflowDefinition], -) -> Result { - validate_now(now_ms)?; - let candidate = transaction - .query_row( - "SELECT t.run_id, t.timer_id, r.definition_digest FROM workflow_timers t INDEXED BY timers_due JOIN workflow_runs r ON r.run_id = t.run_id WHERE t.state = 0 AND t.due_at_ms <= ?1 AND r.status = 0 ORDER BY t.due_at_ms, t.run_id, t.timer_id LIMIT 1", - [now_ms], - |row| { - Ok(( - row.get::<_, Vec>(0)?, - row.get::<_, Vec>(1)?, - row.get::<_, Vec>(2)?, - )) - }, - ) - .optional()?; - let Some((run_id, timer_id, digest)) = candidate else { - return Ok(false); - }; - let run_id = exact_array(run_id, "invalid stored workflow run ID")?; - let timer_id = exact_array(timer_id, "invalid stored workflow timer ID")?; - let digest = Digest::from_bytes(exact_array( - digest, - "invalid stored workflow definition digest", - )?); - workflow_fire_timer_with_effects( - transaction, - effects, - source, - now_ms, - run_id, - timer_id, - retained_definition(definitions, digest)?, - )?; - Ok(true) -} - -pub(super) fn next_effect_batch( - transaction: &Transaction<'_>, - source: &CellTarget, - now_ms: i64, -) -> Result { - let sequence = transaction.query_row( - "SELECT commit_sequence + 1 FROM sys_meta WHERE singleton = 1 AND commit_sequence < 9223372036854775807", - [], - |row| row.get::<_, u64>(0), - )?; - EffectBatch::new(transaction, source, sequence, now_ms) -} - -fn retained_definition( - definitions: &[&'static dyn WorkflowDefinition], - digest: Digest, -) -> Result<&'static dyn WorkflowDefinition> { - definitions - .iter() - .copied() - .find(|definition| definition.digest() == digest) - .ok_or(Error::Command("workflow definition digest is unavailable")) -} - -fn activity_failure_event(activity_id: [u8; 16], details: &[u8]) -> Vec { - let mut event = Vec::with_capacity(34 + details.len()); - event.extend_from_slice(b"activity\0"); - event.push(1); - event.extend_from_slice(&activity_id); - event.extend_from_slice(&(details.len() as u32).to_be_bytes()); - event.extend_from_slice(details); - event -} - -fn activity_failure_event_id(run_id: [u8; 16], activity_id: [u8; 16]) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.activity-terminal.v1\0"); - hasher.update(&run_id); - hasher.update(&activity_id); - *hasher.finalize().as_bytes() -} diff --git a/crates/crab-cell-runtime/src/primitives/workflow/tests.rs b/crates/crab-cell-runtime/src/primitives/workflow/tests.rs deleted file mode 100644 index 34dcb311c..000000000 --- a/crates/crab-cell-runtime/src/primitives/workflow/tests.rs +++ /dev/null @@ -1,372 +0,0 @@ -use super::*; -use crate::identity::IncarnationId; -use crate::identity::{ApplicationId, TenantId}; -use crab_ltx::rusqlite::Connection; - -struct RunningDefinition; - -impl WorkflowDefinition for RunningDefinition { - fn digest(&self) -> Digest { - Digest::from_bytes([8; 32]) - } - - fn transition(&self, _: &[u8], event: &[u8], _: WorkflowContext) -> Result { - Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state: event.to_vec(), - result: None, - actions: Vec::new(), - }) - } -} - -#[test] -fn embedded_workflow_migration_matches_normative_contract() { - assert_eq!( - WORKFLOW_SCHEMA, - include_str!("../../../docs/contracts/workflow.sql") - ); -} - -#[test] -fn status_codec_rejects_unknown_values() { - assert_eq!(WorkflowStatus::decode(0).unwrap(), WorkflowStatus::Running); - assert_eq!(WorkflowStatus::decode(4).unwrap(), WorkflowStatus::Paused); - assert!(WorkflowStatus::decode(5).is_err()); -} - -#[test] -fn pause_resume_and_terminal_restart_preserve_run_identity_rules() { - let mut connection = Connection::open_in_memory().unwrap(); - let transaction = connection.transaction().unwrap(); - let source = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NamespaceId::from_bytes([3; 16]), - b"workflow", - ) - .unwrap(); - crate::cell::schema::install_runtime_schema_in( - &transaction, - source.cell_id(), - IncarnationId::from_bytes([4; 16]), - 1, - ) - .unwrap(); - install_workflow_schema(&transaction).unwrap(); - let start = WorkflowStart { - workflow_id: b"build-42".to_vec(), - request_id: RequestId::from_bytes([5; 16]), - event: b"start".to_vec(), - }; - let WorkflowOutcome::Applied { run_id, .. } = - workflow_start(&transaction, &source, 10, &start, &RunningDefinition).unwrap() - else { - panic!("workflow did not start"); - }; - transaction - .execute( - "INSERT INTO workflow_activities(run_id, activity_id, activity_type, input, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, completion_token, completion_digest, result) VALUES (?1, ?2, 'test', X'', 1, 1, 10, 100, ?3, 50, NULL, NULL, NULL)", - ( - run_id.as_slice(), - [10_u8; 16].as_slice(), - [11_u8; 16].as_slice(), - ), - ) - .unwrap(); - assert_eq!( - workflow_control( - &transaction, - &source, - 11, - &WorkflowControl { - workflow_id: start.workflow_id.clone(), - run_id, - action: WorkflowControlAction::Pause, - }, - &RunningDefinition, - ) - .unwrap(), - WorkflowOutcome::Busy - ); - transaction - .execute( - "DELETE FROM workflow_activities WHERE run_id = ?1", - [run_id.as_slice()], - ) - .unwrap(); - let paused = workflow_control( - &transaction, - &source, - 11, - &WorkflowControl { - workflow_id: start.workflow_id.clone(), - run_id, - action: WorkflowControlAction::Pause, - }, - &RunningDefinition, - ) - .unwrap(); - assert!(matches!( - paused, - WorkflowOutcome::Applied { - status: WorkflowStatus::Paused, - .. - } - )); - assert_eq!( - workflow_signal( - &transaction, - &source, - 12, - &WorkflowSignal { - workflow_id: start.workflow_id.clone(), - run_id, - signal_id: [6; 16], - event: b"blocked".to_vec(), - }, - &RunningDefinition, - ) - .unwrap(), - WorkflowOutcome::NotRunning - ); - workflow_control( - &transaction, - &source, - 13, - &WorkflowControl { - workflow_id: start.workflow_id.clone(), - run_id, - action: WorkflowControlAction::Resume, - }, - &RunningDefinition, - ) - .unwrap(); - workflow_cancel( - &transaction, - 14, - &WorkflowSignal { - workflow_id: start.workflow_id.clone(), - run_id, - signal_id: [7; 16], - event: b"cancel".to_vec(), - }, - ) - .unwrap(); - let restarted = workflow_control( - &transaction, - &source, - 15, - &WorkflowControl { - workflow_id: start.workflow_id, - run_id, - action: WorkflowControlAction::Restart { - request_id: RequestId::from_bytes([9; 16]), - event: b"restart".to_vec(), - }, - }, - &RunningDefinition, - ) - .unwrap(); - let WorkflowOutcome::Applied { - run_id: restarted_run, - .. - } = restarted - else { - panic!("workflow did not restart"); - }; - assert_ne!(run_id, restarted_run); -} - -struct SizedStateDefinition(usize); - -impl WorkflowDefinition for SizedStateDefinition { - fn digest(&self) -> Digest { - Digest::from_bytes([9; 32]) - } - - fn transition(&self, _: &[u8], _: &[u8], _: WorkflowContext) -> Result { - Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state: vec![b's'; self.0], - result: None, - actions: Vec::new(), - }) - } -} - -struct ActivityDefinition { - input_bytes: usize, - lifetime_ms: i64, -} - -impl WorkflowDefinition for ActivityDefinition { - fn digest(&self) -> Digest { - Digest::from_bytes([12; 32]) - } - - fn transition(&self, _: &[u8], _: &[u8], _: WorkflowContext) -> Result { - Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state: b"running".to_vec(), - result: None, - actions: vec![WorkflowAction::Activity { - activity_type: "probe".to_string(), - input: vec![b'i'; self.input_bytes], - due_at_ms: 10, - expires_at_ms: 10 + self.lifetime_ms, - }], - }) - } -} - -struct Tokens(u8); - -impl ActivityTokenSource for Tokens { - fn next_token(&mut self) -> Result<[u8; 16]> { - self.0 = self - .0 - .checked_add(1) - .ok_or(Error::Command("test activity token overflow"))?; - Ok([self.0; 16]) - } -} - -fn workflow_connection() -> (Connection, CellTarget) { - let connection = Connection::open_in_memory().unwrap(); - let source = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NamespaceId::from_bytes([3; 16]), - b"workflow", - ) - .unwrap(); - let transaction = connection.unchecked_transaction().unwrap(); - crate::cell::schema::install_runtime_schema_in( - &transaction, - source.cell_id(), - IncarnationId::from_bytes([4; 16]), - 1, - ) - .unwrap(); - install_workflow_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - (connection, source) -} - -fn start_request(event: &[u8]) -> WorkflowStart { - WorkflowStart { - workflow_id: b"build-42".to_vec(), - request_id: RequestId::from_bytes([7; 16]), - event: event.to_vec(), - } -} - -#[test] -fn workflow_state_at_the_one_mib_boundary_is_accepted() { - let (mut connection, source) = workflow_connection(); - let transaction = connection.transaction().unwrap(); - let outcome = workflow_start( - &transaction, - &source, - 10, - &start_request(b"start"), - &SizedStateDefinition(MAX_WORKFLOW_BYTES), - ) - .unwrap(); - assert!(matches!(outcome, WorkflowOutcome::Applied { .. })); -} - -#[test] -fn workflow_state_over_one_mib_is_rejected() { - let (mut connection, source) = workflow_connection(); - let transaction = connection.transaction().unwrap(); - let outcome = workflow_start( - &transaction, - &source, - 10, - &start_request(b"start"), - &SizedStateDefinition(MAX_WORKFLOW_BYTES + 1), - ); - assert!(matches!( - outcome, - Err(Error::Command("workflow decision exceeds limits")) - )); -} - -#[test] -fn activity_input_over_256_kib_is_rejected() { - let (mut connection, source) = workflow_connection(); - let transaction = connection.transaction().unwrap(); - let outcome = workflow_start( - &transaction, - &source, - 10, - &start_request(b"start"), - &ActivityDefinition { - input_bytes: MAX_ACTIVITY_PAYLOAD_BYTES + 1, - lifetime_ms: 1_000, - }, - ); - assert!(matches!( - outcome, - Err(Error::Command("invalid workflow activity action")) - )); -} - -#[test] -fn activity_lifetime_past_seven_days_is_rejected() { - let (mut connection, source) = workflow_connection(); - let transaction = connection.transaction().unwrap(); - let outcome = workflow_start( - &transaction, - &source, - 10, - &start_request(b"start"), - &ActivityDefinition { - input_bytes: 0, - lifetime_ms: MAX_ACTIVITY_LIFETIME_MS + 1, - }, - ); - assert!(matches!( - outcome, - Err(Error::Command("invalid workflow activity action")) - )); -} - -#[test] -fn activity_attempts_stop_at_twenty() { - let (mut connection, source) = workflow_connection(); - let transaction = connection.transaction().unwrap(); - let definition = RunningDefinition; - let WorkflowOutcome::Applied { run_id, .. } = workflow_start( - &transaction, - &source, - 10, - &start_request(b"start"), - &definition, - ) - .unwrap() else { - panic!("workflow did not start"); - }; - transaction - .execute( - "INSERT INTO workflow_activities(run_id, activity_id, activity_type, input, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, completion_token, completion_digest, result) VALUES (?1, ?2, 'probe', X'', 0, ?3, 10, 100, NULL, NULL, NULL, NULL, NULL)", - ( - run_id.as_slice(), - [13_u8; 16].as_slice(), - i64::from(crate::primitives::workflow::activity::MAX_ATTEMPTS), - ), - ) - .unwrap(); - let supported = [ActivitySupport { - activity_type: "probe".to_string(), - definition_digest: definition.digest(), - }]; - let mut tokens = Tokens(0); - assert!( - workflow_claim_activities(&transaction, 10, 1, 5_000, &supported, &mut tokens) - .unwrap() - .is_empty() - ); -} diff --git a/crates/crab-cell-runtime/src/publication.rs b/crates/crab-cell-runtime/src/publication.rs deleted file mode 100644 index e1fd5c775..000000000 --- a/crates/crab-cell-runtime/src/publication.rs +++ /dev/null @@ -1,871 +0,0 @@ -//! Immutable root preparation, authority CAS, and result release. -use crate::cell::executor::{CellExecutor, StoredOutcome}; -use crate::control::Transition; -use crate::control::authority::{CellAuthority, VersionedControl}; -use crate::fleet::telemetry::DurabilitySubmissionOutcome; -use crate::identity::{ApplicationId, encode_hex}; -use crate::node::durability::NodeDurability; -use crate::node::log::CommitTicket; -use crate::node::log_shipper::NodeLogSubmission; -use crate::retry::{Backoff, retry_hint, retryable_storage_error}; -use crate::{Error, Result}; - -const COMPACTION_CHECK_INTERVAL: u8 = 8; -const COMPACTION_DEBT_SEGMENTS: usize = 32; -const MAX_COMPACTION_CASCADE: usize = 9; -const RENEW_INTERVAL: std::time::Duration = std::time::Duration::from_secs(3); -const SELF_FENCE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(10); - -pub(crate) type NodeDurabilityBinding = (ApplicationId, std::sync::Arc); -pub(crate) type NodeDurabilitySlot = - std::sync::Arc>>; - -#[derive(Clone)] -pub(crate) struct CellDurabilitySubmitter { - cell: crate::CellId, - incarnation: crate::identity::IncarnationId, - epoch: u64, - node_lease: Option, - node_durability: Option, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, -} - -/// Coordinates immutable preparation, authority CAS and result release. -/// -/// A failed CAS response is reconciled against origin before returning. Exact -/// prepared-root equality proves success; a pure renewal can be retried without -/// rerunning SQL. Any ownership or root divergence fences local admission. -pub struct CellPublisher { - replica: crab_ltx::CellReplica, - authority: CellAuthority, - observed: VersionedControl, - scratch_directory: std::path::PathBuf, - segment_count: Option, - appends_since_compaction_check: u8, - renew_at: std::time::Instant, - node_lease: Option, - node_durability: Option, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, -} - -impl CellPublisher { - /// Creates an owner-bound publisher with a private compaction scratch directory. - /// - /// The directory must exist and remain private to this Cell activation. - #[must_use] - pub fn new( - replica: crab_ltx::CellReplica, - authority: CellAuthority, - observed: VersionedControl, - scratch_directory: std::path::PathBuf, - ) -> Self { - Self { - replica, - authority, - observed, - scratch_directory, - segment_count: None, - appends_since_compaction_check: 0, - renew_at: std::time::Instant::now() + RENEW_INTERVAL, - node_lease: None, - node_durability: None, - telemetry: crate::fleet::telemetry::CellTelemetryHandle::default(), - } - } - - pub(crate) fn with_node_lease(mut self, node_lease: crate::NodeLeaseGuard) -> Self { - self.node_lease = Some(node_lease); - self - } - - pub(crate) fn with_node_durability_slot(mut self, durability: NodeDurabilitySlot) -> Self { - self.node_durability = Some(durability); - self - } - - pub(crate) fn with_telemetry( - mut self, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, - ) -> Self { - self.telemetry = telemetry; - self - } - - pub(crate) async fn submit_migration_durability( - &self, - pending: &crate::cell::executor::PendingMigration, - ) -> Result> { - self.durability_submitter() - .submit(pending.commit_sequence(), pending.cuts()) - .await - } - - pub(crate) fn durability_submitter(&self) -> CellDurabilitySubmitter { - let control = self.observed.value(); - CellDurabilitySubmitter { - cell: control.cell, - incarnation: control.incarnation, - epoch: control.epoch, - node_lease: self.node_lease.clone(), - node_durability: self.node_durability.clone(), - telemetry: self.telemetry.clone(), - } - } - - pub(crate) fn record_object_proof(&self, waited: std::time::Duration) { - self.telemetry - .durability_proof(crate::node::log::DurabilitySource::Object, waited); - } - - /// Returns the control version the publisher last observed. - #[must_use] - pub fn control(&self) -> &VersionedControl { - &self.observed - } - - /// Returns the Cell storage layout this publisher writes through. - #[must_use] - pub(crate) fn layout(&self) -> &crab_ltx::CellStorageLayout { - self.authority.layout() - } - - pub(crate) fn renewal_due(&self, now: std::time::Instant) -> bool { - now >= self.renew_at - } - - pub(crate) fn renewal_at(&self) -> std::time::Instant { - self.renew_at - } - - pub(crate) fn compaction_due(&self) -> bool { - self.observed.value().ltx_root().is_some() - && self.appends_since_compaction_check >= COMPACTION_CHECK_INTERVAL - } - - /// Runs at most one promotion while the actor owns the publisher token. - /// Retryable preparation failures leave the debt for a later quiet period. - pub(crate) async fn compact_one_quiet(&mut self) -> Result> { - self.check_node_lease()?; - let Some(base) = self.observed.value().ltx_root() else { - self.appends_since_compaction_check = 0; - return Ok(Some(false)); - }; - let replica = self.replica.clone(); - let scratch_directory = self.scratch_directory.clone(); - let attempt = replica.prepare_scheduled_compaction(&base, &scratch_directory); - tokio::pin!(attempt); - let prepared = loop { - tokio::select! { - result = &mut attempt => break result, - _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => { - self.renew().await?; - } - } - }; - let prepared = match prepared { - Ok(prepared) => prepared, - Err(error) if retryable_ltx_error(&error) => return Ok(None), - Err(error) => return Err(error.into()), - }; - let Some(prepared) = prepared else { - self.appends_since_compaction_check = 0; - return Ok(Some(false)); - }; - let next_due_ms = self.observed.value().next_due_ms; - self.publish_prepared(&prepared, next_due_ms).await?; - self.segment_count = Some(prepared.verified().segment_count()); - self.appends_since_compaction_check = 0; - Ok(Some(true)) - } - - /// Advances owner progress or fences when the renewal cannot be proven in time. - pub(crate) async fn renew(&mut self) -> Result<()> { - self.check_node_lease()?; - let deadline = std::time::Instant::now() + SELF_FENCE_TIMEOUT; - let deadline_at = tokio::time::Instant::from_std(deadline); - let mut backoff = Backoff::default(); - loop { - self.check_node_lease()?; - let successor = self.observed.value().renew()?; - let transition = tokio::time::timeout_at( - deadline_at, - self.authority - .transition(&self.observed, successor.clone(), Transition::Renew), - ) - .await - .map_err(|_| Error::Fenced)?; - match transition { - Ok(renewed) => { - self.check_node_lease()?; - if std::time::Instant::now() >= deadline { - return Err(Error::Fenced); - } - self.observed = renewed; - self.renew_at = std::time::Instant::now() + RENEW_INTERVAL; - return Ok(()); - } - Err(error) => { - let current = tokio::time::timeout_at( - deadline_at, - self.authority.load(self.observed.value().cell), - ) - .await - .map_err(|_| Error::Fenced)?? - .ok_or(Error::Fenced)?; - if current.value().is_same_or_pure_renewal_of(&successor) { - self.observed = current; - self.renew_at = std::time::Instant::now() + RENEW_INTERVAL; - return Ok(()); - } - let still_owned = current - .value() - .is_same_or_pure_renewal_of(self.observed.value()); - if still_owned && retryable_publication_error(&error) { - self.observed = current; - backoff - .wait_until(runtime_retry_hint(&error), deadline) - .await?; - continue; - } - return Err(if still_owned { error } else { Error::Fenced }); - } - } - } - } - - // Reconcile a lost activation CAS before exposing the restored handle. - pub(crate) async fn activate(&mut self) -> Result<()> { - self.check_node_lease()?; - let mut backoff = Backoff::default(); - loop { - self.check_node_lease()?; - let successor = self.observed.value().activate()?; - match self - .authority - .transition(&self.observed, successor.clone(), Transition::Activate) - .await - { - Ok(activated) => { - self.check_node_lease()?; - self.observed = activated; - self.renew_at = std::time::Instant::now() + RENEW_INTERVAL; - return Ok(()); - } - Err(error) => { - let current = loop { - match self.authority.load(self.observed.value().cell).await { - Ok(Some(current)) => break current, - Ok(None) => return Err(Error::Fenced), - Err(load_error) if retryable_publication_error(&load_error) => { - backoff.wait(runtime_retry_hint(&load_error)).await; - } - Err(load_error) => return Err(load_error), - } - }; - if current.value().is_same_or_pure_renewal_of(&successor) { - self.observed = current; - self.renew_at = std::time::Instant::now() + RENEW_INTERVAL; - return Ok(()); - } - let still_owned = current - .value() - .is_same_or_pure_renewal_of(self.observed.value()); - if still_owned && retryable_publication_error(&error) { - self.observed = current; - backoff.wait(runtime_retry_hint(&error)).await; - continue; - } - return Err(if still_owned { error } else { Error::Fenced }); - } - } - } - } - - pub(crate) async fn prepare( - &mut self, - pending: &crate::cell::executor::PendingCommit, - ) -> Result { - let result = self - .prepare_append( - pending.cuts(), - pending.outcome().commit_sequence(), - self.observed.value().schema, - ) - .await; - self.record_publication_cost(); - result - } - - pub(crate) async fn prepare_initial( - &mut self, - cuts: &crab_ltx::CaptureBatch, - ) -> Result { - if self.observed.value().root.is_some() { - return Err(Error::Control("bootstrap control already has a root")); - } - let result = self - .prepare_cuts(None, cuts, 0, self.observed.value().schema) - .await; - self.record_publication_cost(); - let prepared = result?; - self.note_append(&prepared); - Ok(prepared) - } - - /// Prepares the captured cut under its registry-selected target schema. - pub async fn prepare_migration( - &mut self, - pending: &crate::cell::executor::PendingMigration, - ) -> Result { - if self.observed.value().schema != pending.from_schema() { - return Err(Error::Fenced); - } - let result = self - .prepare_append( - pending.cuts(), - pending.commit_sequence(), - pending.to_schema(), - ) - .await; - self.record_publication_cost(); - result - } - - /// Reports the immutable cost of the last preparation attempt. - /// - /// The ledger is drained on every attempt, including failed ones, so a - /// provider retry loop that uploads objects before failing is visible as - /// cost instead of silently disappearing. - fn record_publication_cost(&self) { - let cost = self.replica.take_publication_cost(); - if cost.objects > 0 { - self.telemetry.publication_cost(cost.objects, cost.bytes); - } - } - - async fn prepare_append( - &mut self, - cuts: &crab_ltx::CaptureBatch, - commit_sequence: u64, - schema: u32, - ) -> Result { - let base = self.compact_before_append(cuts.segments.len()).await?; - let prepared = match self - .prepare_cuts(base.as_ref(), cuts, commit_sequence, schema) - .await - { - Ok(prepared) => prepared, - Err(Error::Ltx(error)) if error.is_cell_graph_limit() => { - let Some(root) = base else { - return Err(error.into()); - }; - let Some(compacted) = self.force_full_compaction(&root).await? else { - return Err(error.into()); - }; - self.prepare_cuts(Some(&compacted), cuts, commit_sequence, schema) - .await? - } - Err(error) => return Err(error), - }; - self.note_append(&prepared); - Ok(prepared) - } - - fn note_append(&mut self, prepared: &crab_ltx::PreparedRoot) { - self.segment_count = Some(prepared.verified().segment_count()); - self.appends_since_compaction_check = self.appends_since_compaction_check.saturating_add(1); - tracing::debug!( - segments = prepared.verified().segment_count(), - compaction_debt = self.appends_since_compaction_check, - "Cell LTX append prepared" - ); - } - - async fn compact_before_append( - &mut self, - incoming_segments: usize, - ) -> Result> { - let Some(mut base) = self.observed.value().ltx_root() else { - return Ok(None); - }; - let segment_count = match self.segment_count { - Some(count) => count, - None => { - let count = self.replica.open_root(&base).await?.segment_count(); - self.segment_count = Some(count); - count - } - }; - let segment_limit = self.replica.limits().max_segments.min(4_096); - let projected = segment_count.saturating_add(incoming_segments); - let debt_limit = COMPACTION_DEBT_SEGMENTS.min(segment_limit); - let under_pressure = projected >= debt_limit; - if !under_pressure { - return Ok(Some(base)); - } - tracing::debug!( - segments = segment_count, - incoming_segments, - "Cell LTX compaction pressure" - ); - - for _ in 0..MAX_COMPACTION_CASCADE { - let Some(prepared) = self.prepare_scheduled_compaction(&base).await? else { - if self.segment_count.is_some_and(|count| { - count > 1 && count.saturating_add(incoming_segments) >= debt_limit - }) && let Some(compacted) = self.force_full_compaction(&base).await? - { - base = compacted; - } - self.appends_since_compaction_check = 0; - return Ok(Some(base)); - }; - let next_due_ms = self.observed.value().next_due_ms; - base = self.publish_prepared(&prepared, next_due_ms).await?; - let compacted_segments = prepared.verified().segment_count(); - self.segment_count = Some(compacted_segments); - if compacted_segments.saturating_add(incoming_segments) < debt_limit { - self.appends_since_compaction_check = 0; - return Ok(Some(base)); - } - } - Err(Error::Control( - "Cell compaction cascade exceeded level limit", - )) - } - - async fn prepare_scheduled_compaction( - &mut self, - base: &crab_ltx::RootRef, - ) -> Result> { - let mut backoff = Backoff::default(); - loop { - let replica = self.replica.clone(); - let scratch_directory = self.scratch_directory.clone(); - let attempt = replica.prepare_scheduled_compaction(base, &scratch_directory); - tokio::pin!(attempt); - let result = loop { - tokio::select! { - result = &mut attempt => break result, - _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => { - self.renew().await?; - } - } - }; - self.record_publication_cost(); - match result { - Ok(prepared) => return Ok(prepared), - Err(error) if retryable_ltx_error(&error) => { - backoff.wait(ltx_retry_hint(&error)).await; - } - Err(error) => return Err(error.into()), - } - } - } - - async fn force_full_compaction( - &mut self, - base: &crab_ltx::RootRef, - ) -> Result> { - let segment_count = self - .segment_count - .ok_or(Error::Control("Cell segment count is unavailable"))?; - if segment_count <= 1 { - return Ok(None); - } - tracing::debug!(segments = segment_count, "Cell LTX full compaction forced"); - let mut backoff = Backoff::default(); - let prepared = loop { - let replica = self.replica.clone(); - let scratch_directory = self.scratch_directory.clone(); - let attempt = replica.prepare_compaction(base, 0..segment_count, 9, &scratch_directory); - tokio::pin!(attempt); - let result = loop { - tokio::select! { - result = &mut attempt => break result, - _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => { - self.renew().await?; - } - } - }; - self.record_publication_cost(); - match result { - Ok(prepared) => break prepared, - Err(error) if retryable_ltx_error(&error) => { - backoff.wait(ltx_retry_hint(&error)).await; - } - Err(error) => return Err(error.into()), - } - }; - let next_due_ms = self.observed.value().next_due_ms; - let root = self.publish_prepared(&prepared, next_due_ms).await?; - self.segment_count = Some(prepared.verified().segment_count()); - Ok(Some(root)) - } - - async fn prepare_cuts( - &mut self, - base: Option<&crab_ltx::RootRef>, - cuts: &crab_ltx::CaptureBatch, - commit_sequence: u64, - schema: u32, - ) -> Result { - let mut backoff = Backoff::default(); - loop { - let replica = self.replica.clone(); - let attempt = replica.prepare(base, cuts, commit_sequence, schema); - tokio::pin!(attempt); - let result = loop { - tokio::select! { - result = &mut attempt => break result, - _ = tokio::time::sleep_until(tokio::time::Instant::from_std(self.renew_at)) => { - self.renew().await?; - } - } - }; - match result { - Ok(prepared) => return Ok(prepared), - Err(error) if retryable_ltx_error(&error) => { - backoff.wait(ltx_retry_hint(&error)).await; - } - Err(error) => return Err(error.into()), - } - } - } - - pub(crate) async fn publish_prepared( - &mut self, - prepared: &crab_ltx::PreparedRoot, - next_due_ms: Option, - ) -> Result { - self.publish_proposal(prepared, next_due_ms, None).await - } - - /// Publishes one root together with its new executable code/schema pair. - pub async fn publish_migration( - &mut self, - prepared: &crab_ltx::PreparedRoot, - next_due_ms: Option, - code: crate::Digest, - schema: u32, - ) -> Result { - self.publish_proposal(prepared, next_due_ms, Some((code, schema))) - .await - } - - async fn publish_proposal( - &mut self, - prepared: &crab_ltx::PreparedRoot, - next_due_ms: Option, - migration: Option<(crate::Digest, u32)>, - ) -> Result { - self.check_node_lease()?; - let mut backoff = Backoff::default(); - loop { - self.check_node_lease()?; - let (successor, transition) = match migration { - Some((code, schema)) => ( - self.observed - .value() - .migrate_prepared(prepared, next_due_ms, code, schema)?, - Transition::Migrate, - ), - None => ( - self.observed - .value() - .publish_prepared(prepared, next_due_ms)?, - Transition::Publish, - ), - }; - match self - .authority - .transition(&self.observed, successor.clone(), transition) - .await - { - Ok(published) => { - self.check_node_lease()?; - self.observed = published; - self.renew_at = std::time::Instant::now() + RENEW_INTERVAL; - return Ok(prepared.root()); - } - Err(error) => { - let current = loop { - match self.authority.load(self.observed.value().cell).await { - Ok(Some(current)) => break current, - Ok(None) => return Err(Error::Fenced), - Err(load_error) if retryable_publication_error(&load_error) => { - backoff.wait(runtime_retry_hint(&load_error)).await; - } - Err(load_error) => return Err(load_error), - } - }; - if current.value().ltx_root() == Some(prepared.root()) { - self.observed = current; - return if self.observed.value().is_same_or_pure_renewal_of(&successor) { - self.renew_at = std::time::Instant::now() + RENEW_INTERVAL; - Ok(prepared.root()) - } else { - Err(Error::Fenced) - }; - } - if current - .value() - .is_same_or_pure_renewal_of(self.observed.value()) - && retryable_publication_error(&error) - { - self.observed = current; - backoff.wait(runtime_retry_hint(&error)).await; - continue; - } - return Err( - if current - .value() - .is_same_or_pure_renewal_of(self.observed.value()) - { - error - } else { - Error::Fenced - }, - ); - } - } - } - } - - /// Publishes the executor's retained commit or reconciles an ambiguous CAS. - pub async fn publish_pending(&mut self, executor: &mut CellExecutor) -> Result { - let pending = executor.pending().ok_or(Error::PendingPublication)?; - let next_due_ms = pending.next_due_ms(); - let prepared = self.prepare(pending).await?; - executor.bind_prepared(&prepared)?; - let root = match self.publish_prepared(&prepared, next_due_ms).await { - Ok(root) => root, - Err(error @ Error::Fenced) => { - executor.fence(); - return Err(error); - } - Err(error) => return Err(error), - }; - executor.confirm_published(&root) - } - - /// Releases ownership after the SQL worker has closed the drained Cell. - pub(crate) async fn release(&mut self) -> Result<()> { - self.check_node_lease()?; - let mut backoff = Backoff::default(); - loop { - self.check_node_lease()?; - let successor = self.observed.value().release()?; - match self - .authority - .transition(&self.observed, successor.clone(), Transition::Release) - .await - { - Ok(released) => { - self.check_node_lease()?; - self.observed = released; - return Ok(()); - } - Err(error) => { - let current = loop { - match self.authority.load(self.observed.value().cell).await { - Ok(Some(current)) => break current, - Ok(None) => return Err(Error::Fenced), - Err(load_error) if retryable_publication_error(&load_error) => { - backoff.wait(runtime_retry_hint(&load_error)).await; - } - Err(load_error) => return Err(load_error), - } - }; - if current.value() == &successor { - self.observed = current; - return Ok(()); - } - let still_owned = current - .value() - .is_same_or_pure_renewal_of(self.observed.value()); - if still_owned && retryable_publication_error(&error) { - self.observed = current; - backoff.wait(runtime_retry_hint(&error)).await; - continue; - } - return Err(if still_owned { error } else { Error::Fenced }); - } - } - } - } - - /// Releases the newest control still owned by this fenced executor. - /// - /// Reloading first adopts an ambiguous publication or pure renewal. A new - /// epoch or owner proves that recovery already belongs to another executor. - pub(crate) async fn release_after_fence(&mut self) -> Result<()> { - let expected = self.observed.value().clone(); - let expected_owner = expected.owner.clone().ok_or(Error::Fenced)?; - let expected_cell = expected.cell; - let expected_incarnation = expected.incarnation; - let expected_epoch = expected.epoch; - let mut backoff = Backoff::default(); - let current = loop { - match self.authority.load(expected_cell).await { - Ok(Some(current)) => break current, - Ok(None) => return Err(Error::Fenced), - Err(error) if retryable_publication_error(&error) => { - backoff.wait(runtime_retry_hint(&error)).await; - } - Err(error) => return Err(error), - } - }; - let value = current.value(); - if value.epoch != expected_epoch || value.owner.as_ref() != Some(&expected_owner) { - return Ok(()); - } - if value.incarnation != expected_incarnation - || value.code != expected.code - || value.schema != expected.schema - { - return Err(Error::Control("fenced control changed runtime identity")); - } - if value.root.is_none() - || !matches!( - value.state, - crate::control::ControlState::Recovering | crate::control::ControlState::Serving - ) - { - return Err(Error::Control( - "fenced active control lost its published root", - )); - } - self.observed = current; - self.release().await - } - - fn check_node_lease(&self) -> Result<()> { - self.node_lease - .as_ref() - .map_or(Ok(()), crate::NodeLeaseGuard::check) - } -} - -#[derive(Clone)] -pub(crate) struct PendingDurability { - durability: std::sync::Arc, - ticket: CommitTicket, - submitted_at: std::time::Instant, - telemetry: crate::fleet::telemetry::CellTelemetryHandle, -} - -impl PendingDurability { - pub(crate) async fn prove(&self) -> Result { - let proof = self.durability.prove(self.ticket).await?; - if proof.source() == crate::node::log::DurabilitySource::Fleet { - self.telemetry - .durability_proof(proof.source(), self.submitted_at.elapsed()); - } - Ok(proof.source()) - } - - pub(crate) async fn prove_fleet(&self) -> Result<()> { - let proof = self.durability.prove_fleet(self.ticket).await?; - self.telemetry - .durability_proof(proof.source(), self.submitted_at.elapsed()); - Ok(()) - } - - pub(crate) async fn prove_object(&self) -> Result<()> { - let proof = self.durability.prove_object(self.ticket).await?; - self.telemetry - .durability_proof(proof.source(), self.submitted_at.elapsed()); - Ok(()) - } -} - -impl CellDurabilitySubmitter { - pub(crate) async fn submit( - &self, - commit_sequence: u64, - cuts: &crab_ltx::CaptureBatch, - ) -> Result> { - let Some(slot) = self.node_durability.as_ref() else { - self.telemetry - .durability_submission(DurabilitySubmissionOutcome::Unsupported); - return Ok(None); - }; - let Some((application, durability)) = slot - .read() - .map_err(|_| Error::Control("Cell runtime node durability lock poisoned"))? - .clone() - else { - self.telemetry - .durability_submission(DurabilitySubmissionOutcome::Unavailable); - return Ok(None); - }; - self.check_node_lease()?; - let submission = NodeLogSubmission::new( - application, - self.cell, - self.incarnation, - self.epoch, - commit_sequence, - cuts, - )?; - let ticket = match durability.submit(submission).await { - Ok(ticket) => ticket, - Err(error) => { - self.check_node_lease()?; - // The commit still succeeds through object coverage, so this - // event and its counter are the only way to observe that an - // enrolled lane refused the captured commit. - tracing::warn!( - cell = %encode_hex(self.cell.as_bytes()), - error = %error, - "node-log submission rejected; using object coverage" - ); - self.telemetry - .durability_submission(DurabilitySubmissionOutcome::Rejected); - return Ok(None); - } - }; - self.telemetry - .durability_submission(DurabilitySubmissionOutcome::Fleet); - Ok(Some(PendingDurability { - durability: std::sync::Arc::clone(&durability), - ticket, - submitted_at: std::time::Instant::now(), - telemetry: self.telemetry.clone(), - })) - } - - fn check_node_lease(&self) -> Result<()> { - self.node_lease - .as_ref() - .map_or(Ok(()), crate::NodeLeaseGuard::check) - } -} - -fn retryable_publication_error(error: &Error) -> bool { - let Error::Storage(error) = error else { - return false; - }; - retryable_storage_error(error) -} - -fn retryable_ltx_error(error: &crab_ltx::CrabError) -> bool { - // The failure class is the retry contract; no sender decides from the - // error's shape or message. - error.classify().is_retryable() -} - -fn runtime_retry_hint(error: &Error) -> Option { - let Error::Storage(error) = error else { - return None; - }; - retry_hint(error) -} - -fn ltx_retry_hint(error: &crab_ltx::CrabError) -> Option { - error.classify().retry_after() -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/publication/tests.rs b/crates/crab-cell-runtime/src/publication/tests.rs deleted file mode 100644 index 86b8353cc..000000000 --- a/crates/crab-cell-runtime/src/publication/tests.rs +++ /dev/null @@ -1,323 +0,0 @@ -use std::sync::Arc; - -use bytes::Bytes; -use crab_ltx::{CaptureBatch, CellReplica, CellStorageLayout, Db, Limits}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use super::{ - CellDurabilitySubmitter, CellPublisher, DurabilitySubmissionOutcome, NodeDurabilitySlot, -}; -use crate::Error; -use crate::control::authority::CellAuthority; -use crate::control::{Control, Owner}; -use crate::identity::IncarnationId; -use crate::identity::{CellId, Digest, SessionId}; - -#[tokio::test] -async fn quiet_compaction_publishes_exact_root_after_eight_appends() { - let directory = tempfile::tempdir().unwrap(); - let mut database = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - let cell = CellId::from_bytes([41; 32]); - let incarnation = IncarnationId::from_bytes([42; 16]); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("quiet-compaction"), - [43; 16], - ); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let control = Control::initial( - cell, - incarnation, - Owner { - session: SessionId::from_bytes([44; 16]), - endpoint: "https://node.internal:8081".into(), - }, - Digest::from_bytes([45; 32]), - 1, - ) - .unwrap(); - layout - .store() - .create_strict( - &layout.control_path(cell.as_bytes()), - Bytes::from(control.encode().unwrap()), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout); - let observed = authority.load(cell).await.unwrap().unwrap(); - let mut publisher = CellPublisher::new( - replica.clone(), - authority, - observed, - directory.path().to_owned(), - ); - for sequence in 1..=8_u64 { - database - .transaction(|transaction| { - if sequence == 1 { - transaction - .execute_batch("CREATE TABLE events(sequence INTEGER PRIMARY KEY)")?; - } - transaction.execute("INSERT INTO events VALUES (?1)", [sequence])?; - Ok(()) - }) - .unwrap(); - let cuts = database.capture_deferred().unwrap(); - let prepared = publisher.prepare_append(&cuts, sequence, 1).await.unwrap(); - publisher.publish_prepared(&prepared, None).await.unwrap(); - } - assert!(publisher.compaction_due()); - let before = publisher.control().value().ltx_root().unwrap(); - assert_eq!(replica.open_root(&before).await.unwrap().segment_count(), 8); - assert_eq!(publisher.compact_one_quiet().await.unwrap(), Some(true)); - let after = publisher.control().value().ltx_root().unwrap(); - assert_eq!(after.position, before.position); - assert_eq!(after.commit_sequence, before.commit_sequence); - assert_eq!(replica.open_root(&after).await.unwrap().segment_count(), 1); - assert_eq!(publisher.compact_one_quiet().await.unwrap(), Some(false)); - assert!(!publisher.compaction_due()); - - let mut segments = Vec::new(); - let mut position = after.position; - for sequence in 9..=10_u64 { - database - .transaction(|transaction| { - transaction.execute("INSERT INTO events VALUES (?1)", [sequence])?; - Ok(()) - }) - .unwrap(); - let captured = database.capture_deferred().unwrap(); - segments.extend(captured.segments); - position = captured.position; - } - let cuts = CaptureBatch { - segments, - position, - timing: Default::default(), - }; - assert_eq!(cuts.segments.len(), 2); - let prepared = publisher.prepare_append(&cuts, 9, 1).await.unwrap(); - publisher.publish_prepared(&prepared, None).await.unwrap(); - let extended = publisher.control().value().ltx_root().unwrap(); - assert_eq!(extended.position, position); - assert_eq!( - replica.open_root(&extended).await.unwrap().segment_count(), - 3 - ); - - for sequence in 10..=37_u64 { - database - .transaction(|transaction| { - transaction.execute("INSERT INTO events VALUES (?1)", [sequence + 1])?; - Ok(()) - }) - .unwrap(); - let cuts = database.capture_deferred().unwrap(); - let prepared = publisher.prepare_append(&cuts, sequence, 1).await.unwrap(); - publisher.publish_prepared(&prepared, None).await.unwrap(); - } - let at_ceiling = publisher.control().value().ltx_root().unwrap(); - assert_eq!( - replica - .open_root(&at_ceiling) - .await - .unwrap() - .segment_count(), - 31 - ); - database - .transaction(|transaction| { - transaction.execute("INSERT INTO events VALUES (39)", [])?; - Ok(()) - }) - .unwrap(); - let cuts = database.capture_deferred().unwrap(); - let prepared = publisher.prepare_append(&cuts, 38, 1).await.unwrap(); - publisher.publish_prepared(&prepared, None).await.unwrap(); - let forced = publisher.control().value().ltx_root().unwrap(); - assert!(replica.open_root(&forced).await.unwrap().segment_count() < 32); - database.close().unwrap(); -} - -#[derive(Default)] -struct RecordingSubmissions { - outcomes: std::sync::Mutex>, -} - -impl crate::fleet::telemetry::CellTelemetry for RecordingSubmissions { - fn durability_submission(&self, outcome: DurabilitySubmissionOutcome) { - self.outcomes.lock().unwrap().push(outcome); - } -} - -#[tokio::test] -async fn commits_report_when_no_enrolled_lane_can_carry_them() { - let telemetry = crate::fleet::telemetry::CellTelemetryHandle::default(); - let recording = Arc::new(RecordingSubmissions::default()); - telemetry.install(recording.clone()).unwrap(); - let cuts = CaptureBatch { - segments: Vec::new(), - position: Default::default(), - timing: Default::default(), - }; - let submitter = CellDurabilitySubmitter { - cell: CellId::from_bytes([71; 32]), - incarnation: IncarnationId::from_bytes([72; 16]), - epoch: 1, - node_lease: None, - node_durability: None, - telemetry: telemetry.clone(), - }; - assert!(submitter.submit(1, &cuts).await.unwrap().is_none()); - - let lane: NodeDurabilitySlot = Arc::new(std::sync::RwLock::new(None)); - let submitter = CellDurabilitySubmitter { - node_durability: Some(lane), - telemetry, - ..submitter - }; - assert!(submitter.submit(1, &cuts).await.unwrap().is_none()); - - assert_eq!( - *recording.outcomes.lock().unwrap(), - vec![ - DurabilitySubmissionOutcome::Unsupported, - DurabilitySubmissionOutcome::Unavailable, - ] - ); -} - -/// Transport that refuses every follower request; the gate is fenced first, -/// so no frame reaches it in this test. -struct RefusingTransport; - -impl crate::node::log_transport::NodeLogTransport for RefusingTransport { - fn append<'a>( - &'a self, - _member: crate::identity::NodeId, - _request: crate::node::log_transport::AppendRequest, - ) -> futures_util::future::BoxFuture<'a, crate::Result> { - Box::pin(async { Err(Error::Node("test transport refuses appends")) }) - } - - fn seal<'a>( - &'a self, - _member: crate::identity::NodeId, - _request: crate::node::log_transport::SealRequest, - ) -> futures_util::future::BoxFuture<'a, crate::Result> { - Box::pin(async { Err(Error::Node("test transport refuses seals")) }) - } - - fn retire<'a>( - &'a self, - _member: crate::identity::NodeId, - _request: crate::node::log_transport::RetireRequest, - ) -> futures_util::future::BoxFuture<'a, crate::Result> { - Box::pin(async { Err(Error::Node("test transport refuses retirements")) }) - } - - fn tail<'a>( - &'a self, - _member: crate::identity::NodeId, - _request: crate::node::log_transport::TailRequest, - ) -> futures_util::future::BoxFuture<'a, crate::Result>> { - Box::pin(async { Err(Error::Node("test transport refuses tails")) }) - } -} - -/// Authority that refuses activation; the fenced gate never asks it anything. -struct RefusingAuthority; - -impl crate::node::durability::NodeLogAuthority for RefusingAuthority { - fn activate<'a>( - &'a self, - _log_epoch: u64, - ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { - Box::pin(async { Err(Error::Node("test authority refuses activation")) }) - } - - fn advance_coverage<'a>( - &'a self, - _log_epoch: u64, - _tiered_through: u64, - ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { - Box::pin(async { Err(Error::Node("test authority refuses coverage")) }) - } - - fn close<'a>( - &'a self, - _barrier: &'a crate::node::log::NodeLogRotationBarrier, - ) -> futures_util::future::BoxFuture<'a, crate::Result<()>> { - Box::pin(async { Err(Error::Node("test authority refuses closing")) }) - } -} - -#[tokio::test] -async fn commits_report_a_fenced_lane_instead_of_failing() { - let telemetry = crate::fleet::telemetry::CellTelemetryHandle::default(); - let recording = Arc::new(RecordingSubmissions::default()); - telemetry.install(recording.clone()).unwrap(); - - let gate = crate::node::log::DurabilityGate::new( - SessionId::from_bytes([81; 16]), - crate::identity::NodeId::from_bytes([82; 16]), - 9, - [crate::identity::NodeId::from_bytes([83; 16])], - ) - .unwrap(); - let transport: Arc = - Arc::new(RefusingTransport); - let shipper = crate::node::log_shipper::NodeLogShipper::new_with_telemetry( - gate.clone(), - Arc::clone(&transport), - Limits::default(), - telemetry.clone(), - ) - .unwrap(); - let lease = crate::node::lease::NodeLeaseGuard::new(0, 60_000).unwrap(); - let durability = Arc::new(crate::node::durability::NodeDurability::new( - gate.clone(), - shipper, - Arc::new(RefusingAuthority), - transport, - lease, - )); - gate.stop_shipping(); - - let directory = tempfile::tempdir().unwrap(); - let mut database = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch("CREATE TABLE events(sequence INTEGER PRIMARY KEY)")?; - transaction.execute("INSERT INTO events VALUES (1)", [])?; - Ok(()) - }) - .unwrap(); - let cuts = database.capture_deferred().unwrap(); - - let submitter = CellDurabilitySubmitter { - cell: CellId::from_bytes([84; 32]), - incarnation: IncarnationId::from_bytes([85; 16]), - epoch: 1, - node_lease: None, - node_durability: Some(Arc::new(std::sync::RwLock::new(Some(( - crate::identity::ApplicationId::from_bytes([86; 16]), - durability, - ))))), - telemetry, - }; - assert!(submitter.submit(1, &cuts).await.unwrap().is_none()); - assert_eq!( - *recording.outcomes.lock().unwrap(), - vec![DurabilitySubmissionOutcome::Rejected] - ); - database.close().unwrap(); -} diff --git a/crates/crab-cell-runtime/src/qualification.rs b/crates/crab-cell-runtime/src/qualification.rs deleted file mode 100644 index d10babd30..000000000 --- a/crates/crab-cell-runtime/src/qualification.rs +++ /dev/null @@ -1,41 +0,0 @@ -//! Qualification profiles, receipts, matrix manifests, and cluster validation. -pub mod cluster; - -pub use profile::{ - QUALIFICATION_CASE_COVERAGE_BYTES, QUALIFICATION_CASE_COVERAGE_OPERATIONS, QUALIFICATION_CASES, - QUALIFICATION_MATRIX_ROWS, QUALIFICATION_MATRIX_SCHEMA_VERSION, QUALIFICATION_PRIMITIVES, - QUALIFICATION_PROFILE_SCHEMA_VERSION, QUALIFICATION_PROTECTED_EVIDENCE_MAX_AGE_MS, - QUALIFICATION_PROTECTED_EVIDENCE_MAX_CLOCK_SKEW_MS, QUALIFICATION_RESOURCE_METRICS, - QUALIFICATION_SCHEMA_VERSION, QualificationCase, QualificationProfile, -}; -pub use receipt::{ - QUALIFICATION_PROVIDER_EVIDENCE_SCHEMA_VERSION, QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION, - QualificationExecutionEvidence, QualificationMatrixEntry, QualificationMatrixManifest, - QualificationMetric, QualificationOwnership, QualificationProviderEvidence, - QualificationReceipt, QualificationRunArtifact, QualificationRunner, -}; -pub use workload::{ - QualificationExecution, QualificationLatencyHistogram, QualificationOperation, - QualificationOperationExecutor, QualificationOperationIter, QualificationOutcome, - QualificationPrimitiveCounts, QualificationRunSummary, QualificationWorkload, -}; -mod profile; -mod receipt; -mod workload; - -use std::{ - collections::BTreeSet, - future::Future, - path::Path, - time::{Duration, Instant}, -}; - -use ed25519_dalek::{Signature, Signer, SigningKey, Verifier, VerifyingKey}; -use futures_util::{StreamExt, stream::FuturesUnordered}; -use serde::{Deserialize, Serialize}; - -use crate::identity::Digest; -use crate::{Error, Result}; - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/qualification/cluster.rs b/crates/crab-cell-runtime/src/qualification/cluster.rs deleted file mode 100644 index 68e553e87..000000000 --- a/crates/crab-cell-runtime/src/qualification/cluster.rs +++ /dev/null @@ -1,1074 +0,0 @@ -//! Cluster qualification receipt validation. -use serde::Deserialize; -use serde_json::{Map, Value}; - -use crate::identity::Digest; -use crate::identity::encode_hex; -use crate::{Error, Result}; - -const MAX_CLUSTER_RECEIPT_BYTES: usize = 8 << 20; - -mod selection; - -use selection::*; -const CLUSTER_RECEIPT_SCHEMA_VERSION: u64 = 6; -const MAX_CLUSTER_WORK_VALUE: u64 = 1 << 40; - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct ClusterQualificationReceipt { - version: u64, - source_revision: String, - image: ClusterImageBinding, - project: String, - owner_loss: Value, - fleet_only_commit: Value, - follower_replacement: Value, - second_owner_loss: Value, - fallback_owner_loss: Value, - capacity: Value, - measured_disk: Value, - placement: Value, - metrics: Value, - capacity_metric_parity: Value, - selection: ClusterSelectionEvidence, - work: ClusterWorkEvidence, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct ClusterImageBinding { - reference: String, - digest: String, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct ClusterSelectionEvidence { - owner_loss: SelectionCycle, - second_owner_loss: SelectionCycle, - fallback: SelectionCycle, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct SelectionCycle { - failed_session: String, - successor_session: String, - failed_node: String, - successor_node: String, - failed_log_members: Vec, - selected_original_follower: bool, - terminal_result: String, -} - -#[derive(Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct ClusterWorkEvidence { - owner_loss: WorkCycle, - second_owner_loss: WorkCycle, - fallback: WorkCycle, -} - -#[derive(Clone, Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct WorkCycle { - candidate_count: u64, - affected_cells: u64, - catalog_shards: u64, - catalog_pages: u64, - control_reads: u64, - follower_pages: u64, - follower_frames: u64, - follower_bytes: u64, - peer_requests: u64, - bundle_bytes: u64, - object_reads: u64, - object_writes: u64, - phases: RecoveryPhaseEvidence, -} - -#[derive(Clone, Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct RecoveryPhaseEvidence { - claim: RecoveryPhaseTiming, - witness: RecoveryPhaseTiming, - scope_validation: RecoveryPhaseTiming, - pin_attach: RecoveryPhaseTiming, - seal: RecoveryPhaseTiming, -} - -#[derive(Clone, Copy, Debug, Deserialize)] -#[serde(deny_unknown_fields)] -struct RecoveryPhaseTiming { - count: u64, - duration_ms: u64, -} - -/// Validates one raw four-process failover receipt against its release identity. -/// -/// The receipt is intentionally parsed as a strict top-level record while -/// opaque API responses and Prometheus text remain values. This keeps the -/// evidence contract narrow without duplicating every product response type. -pub fn validate_cluster_receipt( - bytes: &[u8], - source_revision: &str, - image: Digest, - require_published_image: bool, -) -> Result<()> { - if bytes.is_empty() || bytes.len() > MAX_CLUSTER_RECEIPT_BYTES { - return Err(Error::Control("cluster qualification receipt size")); - } - let receipt: ClusterQualificationReceipt = serde_json::from_slice(bytes)?; - if receipt.version != CLUSTER_RECEIPT_SCHEMA_VERSION { - return Err(Error::Control("cluster qualification receipt version")); - } - validate_revision(source_revision)?; - if receipt.source_revision != source_revision { - return Err(Error::Control("cluster qualification source revision")); - } - let expected_digest = format!("sha256:{}", encode_hex(image.as_bytes())); - if receipt.image.digest != expected_digest { - return Err(Error::Control("cluster qualification image digest")); - } - if require_published_image { - if !receipt.image.reference.starts_with("ghcr.io/") - || !receipt - .image - .reference - .ends_with(&format!("@{expected_digest}")) - || receipt.image.reference.len() <= "ghcr.io/@".len() + expected_digest.len() - { - return Err(Error::Control("cluster qualification image reference")); - } - } else if receipt.image.reference != "source-only" { - return Err(Error::Control("cluster qualification source image")); - } - let project_prefix = "crab-http-cluster-qualification-"; - if !receipt.project.starts_with(project_prefix) || receipt.project.len() == project_prefix.len() - { - return Err(Error::Control("cluster qualification project")); - } - - validate_owner_loss(&receipt.owner_loss)?; - validate_fleet_only_commit(&receipt.fleet_only_commit)?; - validate_follower_replacement(&receipt.follower_replacement)?; - validate_second_owner_loss(&receipt.second_owner_loss)?; - validate_owner_chain( - &receipt.owner_loss, - &receipt.second_owner_loss, - &receipt.fleet_only_commit, - )?; - validate_fallback_owner_loss(&receipt.fallback_owner_loss)?; - validate_capacity(&receipt.capacity)?; - validate_measured_disk(&receipt.measured_disk)?; - validate_placement(&receipt.placement)?; - validate_follower_affinity( - &receipt.owner_loss, - &receipt.fleet_only_commit, - &receipt.placement, - )?; - validate_metrics(&receipt.metrics)?; - validate_metric_parity(&receipt.capacity_metric_parity)?; - validate_selection( - &receipt.selection, - &receipt.owner_loss, - &receipt.second_owner_loss, - &receipt.fallback_owner_loss, - &receipt.fleet_only_commit, - &receipt.follower_replacement, - &receipt.placement, - )?; - validate_work(&receipt.work, &receipt.fallback_owner_loss) -} - -fn member_nodes(object: &Map) -> Result> { - object_value(object, "member_nodes")? - .as_array() - .ok_or(Error::Control("cluster receipt array"))? - .iter() - .map(|member| { - let value = member - .as_str() - .ok_or(Error::Control("cluster receipt node"))?; - validate_node_value(value)?; - Ok(value) - }) - .collect() -} - -fn validate_work(work: &ClusterWorkEvidence, fallback_owner_loss: &Value) -> Result<()> { - validate_work_cycle(&work.owner_loss, false)?; - validate_work_cycle(&work.second_owner_loss, false)?; - validate_fallback_work(&work.fallback, fallback_owner_loss) -} - -fn validate_fallback_work(work: &WorkCycle, fallback_owner_loss: &Value) -> Result<()> { - let fallback = as_object(fallback_owner_loss, "fallback owner loss")?; - let fallback_log = as_object( - object_value(fallback, "node_log_before")?, - "fallback node log", - )?; - if !boolean(fallback_log, "active")? && work.is_empty() { - // An inactive log never enabled fleet proof, so request-path takeover - // may finish without scheduling a node-log recovery job. - return Ok(()); - } - validate_work_cycle(work, true) -} - -fn validate_work_cycle(work: &WorkCycle, allow_object_only: bool) -> Result<()> { - for value in [ - work.candidate_count, - work.affected_cells, - work.catalog_shards, - work.catalog_pages, - work.control_reads, - work.bundle_bytes, - work.object_reads, - work.object_writes, - ] { - if value == 0 || value > MAX_CLUSTER_WORK_VALUE { - return Err(Error::Control("cluster recovery work counter")); - } - } - let follower_work = [ - work.follower_pages, - work.follower_frames, - work.follower_bytes, - work.peer_requests, - ]; - for value in follower_work { - if value > MAX_CLUSTER_WORK_VALUE || (!allow_object_only && value == 0) { - return Err(Error::Control("cluster recovery follower work counter")); - } - } - let follower_work_is_empty = follower_work.iter().all(|value| *value == 0); - if allow_object_only && !follower_work_is_empty && follower_work.contains(&0) { - return Err(Error::Control("cluster recovery follower work counter")); - } - let phases = [ - &work.phases.claim, - &work.phases.witness, - &work.phases.scope_validation, - &work.phases.pin_attach, - &work.phases.seal, - ]; - for phase in phases { - if phase.count == 0 || phase.duration_ms > MAX_CLUSTER_WORK_VALUE { - return Err(Error::Control("cluster recovery phase evidence")); - } - } - Ok(()) -} - -impl WorkCycle { - fn is_empty(&self) -> bool { - [ - self.candidate_count, - self.affected_cells, - self.catalog_shards, - self.catalog_pages, - self.control_reads, - self.follower_pages, - self.follower_frames, - self.follower_bytes, - self.peer_requests, - self.bundle_bytes, - self.object_reads, - self.object_writes, - self.phases.claim.count, - self.phases.claim.duration_ms, - self.phases.witness.count, - self.phases.witness.duration_ms, - self.phases.scope_validation.count, - self.phases.scope_validation.duration_ms, - self.phases.pin_attach.count, - self.phases.pin_attach.duration_ms, - self.phases.seal.count, - self.phases.seal.duration_ms, - ] - .into_iter() - .all(|value| value == 0) - } -} - -fn validate_owner_chain( - owner_loss: &Value, - second_owner_loss: &Value, - fleet_only_commit: &Value, -) -> Result<()> { - let first = as_object(owner_loss, "owner loss")?; - let second = as_object(second_owner_loss, "second owner loss")?; - let fleet = as_object(fleet_only_commit, "fleet-only commit")?; - let before = as_object( - object_value(fleet, "control_before_owner_loss")?, - "control before owner loss", - )?; - let owner = as_object(object_value(before, "owner")?, "owner")?; - - // The two fault rows must describe one authority history, not unrelated - // passing takeovers spliced into a single receipt. - if session(owner, "session")? != session(first, "session_before")? - || number(before, "epoch")? != number(first, "epoch_before")? - || session(first, "session_after")? != session(second, "session_before")? - || number(first, "epoch_after")? != number(second, "epoch_before")? - || session(first, "session_before")? == session(second, "session_after")? - { - return Err(Error::Control("cluster qualification owner chain")); - } - - require_same_root( - root(object_value(before, "root")?)?, - root(object_value(first, "root_before")?)?, - )?; - require_root_nonregression( - root(object_value(first, "root_continued")?)?, - root(object_value(second, "root_before")?)?, - ) -} - -fn require_same_root(left: RootEvidence<'_>, right: RootEvidence<'_>) -> Result<()> { - if left.digest != right.digest - || left.checksum != right.checksum - || left.txid != right.txid - || left.commit_sequence != right.commit_sequence - { - return Err(Error::Control("cluster qualification root chain")); - } - Ok(()) -} - -fn require_root_nonregression(before: RootEvidence<'_>, after: RootEvidence<'_>) -> Result<()> { - if after.txid < before.txid || after.commit_sequence < before.commit_sequence { - return Err(Error::Control("cluster qualification root chain")); - } - if after.txid == before.txid || after.commit_sequence == before.commit_sequence { - return require_same_root(before, after); - } - if after.digest == before.digest { - return Err(Error::Control("cluster qualification root chain")); - } - Ok(()) -} - -fn validate_owner_loss(value: &Value) -> Result<()> { - let object = as_object(value, "owner loss")?; - let before_session = session(object, "session_before")?; - let after_session = session(object, "session_after")?; - if before_session == after_session { - return Err(Error::Control("owner loss did not change session")); - } - let before_epoch = number(object, "epoch_before")?; - let after_epoch = number(object, "epoch_after")?; - if after_epoch <= before_epoch { - return Err(Error::Control("owner loss epoch did not advance")); - } - validate_timing(object)?; - let before_root = root(object_value(object, "root_before")?)?; - let restored_root = root(object_value(object, "root_after_restore")?)?; - let continued_root = root(object_value(object, "root_continued")?)?; - require_root_advance(before_root, restored_root)?; - require_root_advance(restored_root, continued_root) -} - -fn validate_fleet_only_commit(value: &Value) -> Result<()> { - let object = as_object(value, "fleet-only commit")?; - let log = as_object(object_value(object, "node_log_after")?, "node log after")?; - if string(log, "state")? != "open" - || number(log, "epoch")? == 0 - || !boolean(log, "active")? - || array_len(log, "member_nodes")? == 0 - { - return Err(Error::Control("fleet-only active log")); - } - if number(object, "owner_uncovered_bytes")? == 0 - || number(object, "follower_retained_bytes")? == 0 - || !boolean(object, "immutable_object_put_rejected")? - || !boolean(object, "owner_disk_removed_before_policy_restore")? - { - return Err(Error::Control("fleet-only durability evidence")); - } - let control = as_object( - object_value(object, "control_before_owner_loss")?, - "control", - )?; - if string(control, "state")? != "serving" { - return Err(Error::Control("fleet-only control state")); - } - let lease = as_object(object_value(control, "owner_lease")?, "owner lease")?; - if string(lease, "state")? != "live" - || number(lease, "expires_at_ms")? <= number(lease, "observed_at_ms")? - { - return Err(Error::Control("fleet-only owner lease")); - } - if number( - as_object(object_value(object, "response")?, "response")?, - "id", - )? != 1 - || array_len( - as_object(object_value(object, "restored_labels")?, "restored labels")?, - "items", - )? != 1 - { - return Err(Error::Control("fleet-only response evidence")); - } - Ok(()) -} - -fn validate_follower_replacement(value: &Value) -> Result<()> { - let object = as_object(value, "follower replacement")?; - let before = as_object(object_value(object, "node_log_before")?, "node log before")?; - let after = as_object(object_value(object, "node_log_after")?, "node log after")?; - if number(after, "epoch")? <= number(before, "epoch")? || array_len(after, "member_nodes")? != 1 - { - return Err(Error::Control("follower replacement log")); - } - if number( - as_object( - object_value(object, "object_covered_response")?, - "object response", - )?, - "number", - )? != 3 - || number( - as_object( - object_value(object, "fleet_only_response")?, - "fleet response", - )?, - "id", - )? != 2 - { - return Err(Error::Control("follower replacement response")); - } - Ok(()) -} - -fn validate_second_owner_loss(value: &Value) -> Result<()> { - let object = as_object(value, "second owner loss")?; - if session(object, "session_before")? == session(object, "session_after")? { - return Err(Error::Control("second owner loss did not change session")); - } - if number(object, "epoch_after")? <= number(object, "epoch_before")? { - return Err(Error::Control("second owner loss epoch did not advance")); - } - validate_timing(object)?; - let before = root(object_value(object, "root_before")?)?; - let after = root(object_value(object, "root_after")?)?; - require_root_advance(before, after)?; - if array_len( - as_object(object_value(object, "restored_labels")?, "restored labels")?, - "items", - )? != 2 - { - return Err(Error::Control("second owner loss labels")); - } - Ok(()) -} - -fn validate_fallback_owner_loss(value: &Value) -> Result<()> { - let object = as_object(value, "fallback owner loss")?; - if session(object, "session_before")? == session(object, "session_after")? { - return Err(Error::Control("fallback owner loss did not change session")); - } - if number(object, "epoch_after")? <= number(object, "epoch_before")? { - return Err(Error::Control("fallback owner loss epoch did not advance")); - } - validate_timing(object)?; - let before = root(object_value(object, "root_before")?)?; - let after = root(object_value(object, "root_after")?)?; - require_root_advance(before, after)?; - let log = as_object( - object_value(object, "node_log_before")?, - "fallback node log", - )?; - let members = member_nodes(log)?; - if members.is_empty() - || number( - as_object(object_value(object, "response")?, "fallback response")?, - "id", - )? != 3 - || array_len( - as_object(object_value(object, "restored_labels")?, "fallback labels")?, - "items", - )? != 3 - { - return Err(Error::Control("fallback owner-loss evidence")); - } - Ok(()) -} - -fn validate_timing(object: &Map) -> Result<()> { - let timing = as_object(object_value(object, "timing")?, "timing")?; - let owner_killed = number(timing, "owner_killed_ms")?; - let advertisement_expired = number(timing, "advertisement_expired_ms")?; - let recovery_sealed = number(timing, "recovery_sealed_ms")?; - let first_served = number(timing, "first_served_ms")?; - if owner_killed == 0 - || owner_killed > advertisement_expired - || advertisement_expired > recovery_sealed - || recovery_sealed > first_served - { - return Err(Error::Control("cluster qualification timing")); - } - Ok(()) -} - -fn validate_capacity(value: &Value) -> Result<()> { - let object = as_object(value, "capacity")?; - for node in ["node_a", "node_b", "node_c", "node_d"] { - let node = as_object(object_value(object, node)?, "capacity node")?; - if number(node, "version")? != 1 { - return Err(Error::Control("capacity schema")); - } - let resources = as_object(object_value(node, "resources")?, "capacity resources")?; - if number(resources, "memory_bytes")? == 0 || number(resources, "free_disk_bytes")? == 0 { - return Err(Error::Control("capacity resources")); - } - let admission = as_object(object_value(node, "admission")?, "capacity admission")?; - if number(admission, "active_cells")? == 0 { - return Err(Error::Control("capacity admission")); - } - } - Ok(()) -} - -fn validate_measured_disk(value: &Value) -> Result<()> { - let object = as_object(value, "measured disk")?; - for node in [ - "node_a_bytes", - "node_b_bytes", - "node_c_bytes", - "node_d_bytes", - "tolerance_bytes", - ] { - number(object, node)?; - } - Ok(()) -} - -fn validate_placement(value: &Value) -> Result<()> { - let object = as_object(value, "placement")?; - for node in ["node_a", "node_b", "node_c", "node_d"] { - let node = as_object(object_value(object, node)?, "placement node")?; - if !boolean(node, "live")? { - return Err(Error::Control("placement node is not live")); - } - let advertisement = as_object( - object_value(node, "advertisement")?, - "placement advertisement", - )?; - as_object( - object_value(advertisement, "placement")?, - "placement capacity", - )?; - as_object(object_value(advertisement, "log")?, "placement log")?; - } - Ok(()) -} - -fn validate_follower_affinity( - owner_loss: &Value, - fleet_only_commit: &Value, - placement: &Value, -) -> Result<()> { - let owner_loss = as_object(owner_loss, "owner loss")?; - let failed_session = session(owner_loss, "session_before")?; - let successor_session = session(owner_loss, "session_after")?; - let fleet_only_commit = as_object(fleet_only_commit, "fleet-only commit")?; - let placement = as_object(placement, "placement")?; - let mut failed_node = None; - let mut successor_node = None; - for (label, session_key) in [ - ("node_a", "node_a_session"), - ("node_b", "node_b_session"), - ("node_c", "node_c_session"), - ] { - let placement_session = session(placement, session_key)?; - let node = as_object(object_value(placement, label)?, "placement node")?; - if session(node, "session")? != placement_session { - return Err(Error::Control("placement session identity")); - } - let advertisement = as_object( - object_value(node, "advertisement")?, - "placement advertisement", - )?; - let node_id = node_id(advertisement, "node")?; - if placement_session == failed_session { - failed_node = Some(node_id); - } - if placement_session == successor_session { - successor_node = Some(node_id); - } - } - let failed_node = failed_node.ok_or(Error::Control("failed owner placement"))?; - let successor_node = successor_node.ok_or(Error::Control("successor placement"))?; - if failed_node == successor_node { - return Err(Error::Control("successor reused failed owner node")); - } - - // A fully covered log may be replaced before the follower-only write. - // Affinity belongs to the active log that protected that acknowledgement. - let node_log = as_object( - object_value(fleet_only_commit, "node_log_after")?, - "node log", - )?; - let members = object_value(node_log, "member_nodes")? - .as_array() - .ok_or(Error::Control("cluster receipt array"))?; - if !members - .iter() - .any(|member| member.as_str() == Some(successor_node)) - { - return Err(Error::Control("successor was not an original follower")); - } - Ok(()) -} - -fn validate_metrics(value: &Value) -> Result<()> { - let object = as_object(value, "metrics")?; - for node in ["node_a", "node_b", "node_c", "node_d"] { - let metrics = object_value(object, node)? - .as_str() - .ok_or(Error::Control("metrics payload"))?; - for forbidden in [ - "node=\"", - "session=\"", - "cell=\"", - "application=\"", - "incarnation=\"", - "owner=\"", - "repository=\"", - "tenant=\"", - "bucket=\"", - "node_id=\"", - ] { - if metrics.contains(forbidden) { - return Err(Error::Control("metrics payload contains an identifier")); - } - } - if metrics.is_empty() { - return Err(Error::Control("metrics payload")); - } - } - Ok(()) -} - -fn validate_metric_parity(value: &Value) -> Result<()> { - let object = as_object(value, "capacity metric parity")?; - for field in [ - "local_disk", - "active_cells", - "measured_local_disk", - "signed_placement", - ] { - if !boolean(object, field)? { - return Err(Error::Control("capacity metric parity")); - } - } - Ok(()) -} - -fn require_root_advance(before: RootEvidence<'_>, after: RootEvidence<'_>) -> Result<()> { - if before.digest == after.digest - || after.txid <= before.txid - || after.commit_sequence <= before.commit_sequence - { - return Err(Error::Control("recovered root did not advance")); - } - Ok(()) -} - -#[derive(Clone, Copy)] -struct RootEvidence<'a> { - digest: &'a str, - checksum: u64, - txid: u64, - commit_sequence: u64, -} - -fn root(value: &Value) -> Result> { - let object = as_object(value, "root")?; - let digest = string(object, "digest")?; - if digest.len() != 64 || !digest.bytes().all(is_lower_hex) { - return Err(Error::Control("root digest")); - } - let txid = number(object, "txid")?; - let checksum = number(object, "checksum")?; - let commit_sequence = number(object, "commit_sequence")?; - if txid == 0 || commit_sequence == 0 { - return Err(Error::Control("root watermark")); - } - Ok(RootEvidence { - digest, - checksum, - txid, - commit_sequence, - }) -} - -fn validate_revision(value: &str) -> Result<()> { - if value.len() != 40 || !value.bytes().all(is_lower_hex) { - return Err(Error::Control("cluster qualification source revision")); - } - Ok(()) -} - -fn as_object<'a>(value: &'a Value, _name: &'static str) -> Result<&'a Map> { - value - .as_object() - .ok_or(Error::Control("cluster receipt object")) -} - -fn object_value<'a>(object: &'a Map, key: &str) -> Result<&'a Value> { - object - .get(key) - .ok_or(Error::Control("cluster receipt field")) -} - -fn string<'a>(object: &'a Map, key: &str) -> Result<&'a str> { - object_value(object, key)? - .as_str() - .ok_or(Error::Control("cluster receipt string")) -} - -fn session<'a>(object: &'a Map, key: &str) -> Result<&'a str> { - let value = string(object, key)?; - validate_session_value(value)?; - Ok(value) -} - -fn node_id<'a>(object: &'a Map, key: &str) -> Result<&'a str> { - let value = string(object, key)?; - validate_node_value(value)?; - Ok(value) -} - -fn validate_session_value(value: &str) -> Result<()> { - if value.len() != 32 || !value.bytes().all(is_lower_hex) { - return Err(Error::Control("cluster receipt session")); - } - Ok(()) -} - -fn validate_node_value(value: &str) -> Result<()> { - if value.len() != 32 || !value.bytes().all(is_lower_hex) { - return Err(Error::Control("cluster receipt node")); - } - Ok(()) -} - -fn number(object: &Map, key: &str) -> Result { - object_value(object, key)? - .as_u64() - .ok_or(Error::Control("cluster receipt number")) -} - -fn boolean(object: &Map, key: &str) -> Result { - object_value(object, key)? - .as_bool() - .ok_or(Error::Control("cluster receipt boolean")) -} - -fn array_len(object: &Map, key: &str) -> Result { - object_value(object, key)? - .as_array() - .map(Vec::len) - .ok_or(Error::Control("cluster receipt array")) -} - -fn is_lower_hex(byte: u8) -> bool { - byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte) -} - -#[cfg(test)] -mod tests { - use serde_json::{Value, json}; - - use super::{ - ClusterWorkEvidence, RecoveryPhaseEvidence, RecoveryPhaseTiming, WorkCycle, - validate_fallback_work, validate_follower_affinity, validate_metrics, validate_owner_chain, - validate_revision, validate_timing, validate_work_cycle, - }; - - fn chained_owner_losses() -> (Value, Value, Value) { - let first_root = - json!({"digest": "a".repeat(64), "checksum": 10, "txid": 1, "commit_sequence": 1}); - let continued_root = - json!({"digest": "b".repeat(64), "checksum": 20, "txid": 2, "commit_sequence": 2}); - let owner_loss = json!({ - "session_before": "1".repeat(32), - "session_after": "2".repeat(32), - "epoch_before": 1, - "epoch_after": 2, - "root_before": first_root, - "root_continued": continued_root - }); - let second_owner_loss = json!({ - "session_before": "2".repeat(32), - "session_after": "3".repeat(32), - "epoch_before": 2, - "epoch_after": 3, - "root_before": continued_root - }); - let fleet_only_commit = json!({ - "control_before_owner_loss": { - "owner": {"session": "1".repeat(32)}, - "epoch": 1, - "root": first_root - } - }); - (owner_loss, second_owner_loss, fleet_only_commit) - } - - #[test] - fn successive_owner_losses_bind_one_monotonic_history() { - let (owner_loss, second_owner_loss, fleet_only_commit) = chained_owner_losses(); - assert!(validate_owner_chain(&owner_loss, &second_owner_loss, &fleet_only_commit).is_ok()); - - let mut spliced = second_owner_loss.clone(); - spliced["session_before"] = json!("4".repeat(32)); - assert!(validate_owner_chain(&owner_loss, &spliced, &fleet_only_commit).is_err()); - let mut regressed = second_owner_loss.clone(); - regressed["root_before"]["commit_sequence"] = json!(1); - assert!(validate_owner_chain(&owner_loss, ®ressed, &fleet_only_commit).is_err()); - let mut digest_mismatch = fleet_only_commit.clone(); - digest_mismatch["control_before_owner_loss"]["root"]["digest"] = json!("d".repeat(64)); - assert!(validate_owner_chain(&owner_loss, &second_owner_loss, &digest_mismatch).is_err()); - let mut checksum_mismatch = fleet_only_commit.clone(); - checksum_mismatch["control_before_owner_loss"]["root"]["checksum"] = json!(11); - assert!(validate_owner_chain(&owner_loss, &second_owner_loss, &checksum_mismatch).is_err()); - } - - #[test] - fn source_revision_requires_a_lowercase_commit_shape() { - assert!(validate_revision(&"a".repeat(40)).is_ok()); - assert!(validate_revision(&"A".repeat(40)).is_err()); - assert!(validate_revision("not-a-commit".repeat(4).as_str()).is_err()); - } - - #[test] - fn timing_requires_monotonic_failover_boundaries() { - let valid = json!({ - "timing": { - "owner_killed_ms": 10, - "advertisement_expired_ms": 20, - "recovery_sealed_ms": 30, - "first_served_ms": 40 - } - }); - assert!(validate_timing(valid.as_object().unwrap()).is_ok()); - - let invalid = json!({ - "timing": { - "owner_killed_ms": 20, - "advertisement_expired_ms": 10, - "recovery_sealed_ms": 30, - "first_served_ms": 40 - } - }); - assert!(validate_timing(invalid.as_object().unwrap()).is_err()); - } - - #[test] - fn follower_affinity_requires_a_member_of_the_failed_log() { - let failed_session = "a".repeat(32); - let successor_session = "b".repeat(32); - let failed_node = "c".repeat(32); - let successor_node = "d".repeat(32); - let owner_loss = json!({ - "session_before": failed_session, - "session_after": successor_session, - }); - let fleet_only_commit = json!({ - "node_log_after": {"member_nodes": [successor_node]} - }); - let placement = json!({ - "node_a_session": "e".repeat(32), - "node_b_session": "a".repeat(32), - "node_c_session": "b".repeat(32), - "node_a": {"session": "e".repeat(32), "advertisement": {"node": "f".repeat(32)}}, - "node_b": {"session": "a".repeat(32), "advertisement": {"node": "c".repeat(32)}}, - "node_c": {"session": "b".repeat(32), "advertisement": {"node": "d".repeat(32)}}, - "node_d_session": "1".repeat(32), - "node_d": {"session": "1".repeat(32), "advertisement": {"node": "2".repeat(32)}} - }); - assert!(validate_follower_affinity(&owner_loss, &fleet_only_commit, &placement).is_ok()); - - let mut invalid_fleet = fleet_only_commit.clone(); - invalid_fleet["node_log_after"]["member_nodes"] = json!([failed_node]); - assert!(validate_follower_affinity(&owner_loss, &invalid_fleet, &placement).is_err()); - - let mut mismatched_placement = placement.clone(); - mismatched_placement["node_c"]["session"] = json!("f".repeat(32)); - assert!( - validate_follower_affinity(&owner_loss, &fleet_only_commit, &mismatched_placement) - .is_err() - ); - - let mut reused_owner = placement; - reused_owner["node_c"]["advertisement"]["node"] = json!(failed_node); - assert!( - validate_follower_affinity(&owner_loss, &fleet_only_commit, &reused_owner).is_err() - ); - } - - #[test] - fn work_evidence_requires_a_bounded_observation_per_phase() { - let phase = RecoveryPhaseTiming { - count: 1, - duration_ms: 0, - }; - let valid = WorkCycle { - candidate_count: 1, - affected_cells: 1, - catalog_shards: 1, - catalog_pages: 1, - control_reads: 1, - follower_pages: 1, - follower_frames: 1, - follower_bytes: 1, - peer_requests: 1, - bundle_bytes: 1, - object_reads: 1, - object_writes: 1, - phases: RecoveryPhaseEvidence { - claim: phase, - witness: phase, - scope_validation: phase, - pin_attach: phase, - seal: phase, - }, - }; - assert!(validate_work_cycle(&valid, false).is_ok()); - let object_only = WorkCycle { - follower_pages: 0, - follower_frames: 0, - follower_bytes: 0, - peer_requests: 0, - ..valid.clone() - }; - assert!(validate_work_cycle(&object_only, true).is_ok()); - let partial_follower_work = WorkCycle { - follower_pages: 1, - follower_frames: 0, - ..object_only - }; - assert!(validate_work_cycle(&partial_follower_work, true).is_err()); - - let invalid = WorkCycle { - phases: RecoveryPhaseEvidence { - claim: RecoveryPhaseTiming { - count: 0, - duration_ms: 0, - }, - ..valid.phases - }, - ..valid - }; - assert!(validate_work_cycle(&invalid, false).is_err()); - } - - #[test] - fn fallback_work_allows_direct_takeover_only_for_inactive_logs() { - let zero = WorkCycle { - candidate_count: 0, - affected_cells: 0, - catalog_shards: 0, - catalog_pages: 0, - control_reads: 0, - follower_pages: 0, - follower_frames: 0, - follower_bytes: 0, - peer_requests: 0, - bundle_bytes: 0, - object_reads: 0, - object_writes: 0, - phases: RecoveryPhaseEvidence { - claim: RecoveryPhaseTiming { - count: 0, - duration_ms: 0, - }, - witness: RecoveryPhaseTiming { - count: 0, - duration_ms: 0, - }, - scope_validation: RecoveryPhaseTiming { - count: 0, - duration_ms: 0, - }, - pin_attach: RecoveryPhaseTiming { - count: 0, - duration_ms: 0, - }, - seal: RecoveryPhaseTiming { - count: 0, - duration_ms: 0, - }, - }, - }; - let inactive = json!({"node_log_before": {"active": false}}); - assert!(validate_fallback_work(&zero, &inactive).is_ok()); - - let active = json!({"node_log_before": {"active": true}}); - assert!(validate_fallback_work(&zero, &active).is_err()); - } - - #[test] - fn metrics_reject_identifier_bearing_labels() { - let valid = json!({ - "node_a": "crab_cell_node_log_recovery_work_total{kind=\"candidate_count\"} 1", - "node_b": "crab_cell_node_log_recovery_work_total{kind=\"candidate_count\"} 1", - "node_c": "crab_cell_node_log_recovery_work_total{kind=\"candidate_count\"} 1", - "node_d": "crab_cell_node_log_recovery_work_total{kind=\"candidate_count\"} 1" - }); - assert!(validate_metrics(&valid).is_ok()); - let invalid = json!({ - "node_a": "crab_cell_node_log_recovery_work_total{node=\"deadbeef\"} 1", - "node_b": "ok", - "node_c": "ok", - "node_d": "ok" - }); - assert!(validate_metrics(&invalid).is_err()); - } - - #[test] - fn work_receipt_shape_rejects_missing_or_unknown_phase_fields() { - let phase = json!({"count": 1, "duration_ms": 0}); - let cycle = json!({ - "candidate_count": 1, - "affected_cells": 1, - "catalog_shards": 1, - "catalog_pages": 1, - "control_reads": 1, - "follower_pages": 1, - "follower_frames": 1, - "follower_bytes": 1, - "peer_requests": 1, - "bundle_bytes": 1, - "object_reads": 1, - "object_writes": 1, - "phases": { - "claim": phase, - "witness": phase, - "scope_validation": phase, - "pin_attach": phase, - "seal": phase - } - }); - let valid = json!({ - "owner_loss": cycle.clone(), - "second_owner_loss": cycle.clone(), - "fallback": cycle.clone() - }); - assert!(serde_json::from_value::(valid).is_ok()); - - let mut missing = cycle.clone(); - missing["phases"].as_object_mut().unwrap().remove("seal"); - assert!(serde_json::from_value::(missing).is_err()); - - let mut unknown = cycle; - unknown["phases"]["seal"]["extra"] = json!(true); - assert!(serde_json::from_value::(unknown).is_err()); - } -} diff --git a/crates/crab-cell-runtime/src/qualification/cluster/selection.rs b/crates/crab-cell-runtime/src/qualification/cluster/selection.rs deleted file mode 100644 index f01c6d1f3..000000000 --- a/crates/crab-cell-runtime/src/qualification/cluster/selection.rs +++ /dev/null @@ -1,175 +0,0 @@ -//! Selection and placement evidence inside a cluster receipt. -//! -//! The receipt binds one signed owner session to the placement it was -//! elected from, so these checks re-derive which node the placement named -//! before any later phase is allowed to trust the selection. - -use super::*; - -pub(super) fn validate_selection( - selection: &ClusterSelectionEvidence, - owner_loss: &Value, - second_owner_loss: &Value, - fallback_owner_loss: &Value, - fleet_only_commit: &Value, - follower_replacement: &Value, - placement: &Value, -) -> Result<()> { - let owner_loss = as_object(owner_loss, "owner loss")?; - let second_owner_loss = as_object(second_owner_loss, "second owner loss")?; - let fallback_owner_loss = as_object(fallback_owner_loss, "fallback owner loss")?; - validate_selection_cycle( - &selection.owner_loss, - session(owner_loss, "session_before")?, - session(owner_loss, "session_after")?, - true, - )?; - validate_selection_cycle( - &selection.second_owner_loss, - session(second_owner_loss, "session_before")?, - session(second_owner_loss, "session_after")?, - true, - )?; - validate_selection_cycle( - &selection.fallback, - session(fallback_owner_loss, "session_before")?, - session(fallback_owner_loss, "session_after")?, - false, - )?; - - let placement = as_object(placement, "placement")?; - let expected_failed = - placement_node_for_session(placement, &selection.owner_loss.failed_session)?; - let expected_successor = - placement_node_for_session(placement, &selection.owner_loss.successor_session)?; - if selection.owner_loss.failed_node != expected_failed - || selection.owner_loss.successor_node != expected_successor - { - return Err(Error::Control("owner-loss selection identity")); - } - let first_members = member_nodes(as_object( - object_value( - as_object(fleet_only_commit, "fleet-only commit")?, - "node_log_after", - )?, - "node log", - )?)?; - if !first_members - .iter() - .any(|member| *member == selection.owner_loss.successor_node) - || first_members - != selection - .owner_loss - .failed_log_members - .iter() - .map(String::as_str) - .collect::>() - { - return Err(Error::Control("owner-loss successor is not a follower")); - } - - let second_members = member_nodes(as_object( - object_value( - as_object(follower_replacement, "follower replacement")?, - "node_log_after", - )?, - "node log", - )?)?; - if selection.second_owner_loss.failed_node != selection.owner_loss.successor_node - || selection.second_owner_loss.failed_node == selection.second_owner_loss.successor_node - || !second_members - .iter() - .any(|member| *member == selection.second_owner_loss.successor_node) - || second_members - != selection - .second_owner_loss - .failed_log_members - .iter() - .map(String::as_str) - .collect::>() - { - return Err(Error::Control("second owner-loss selection identity")); - } - let fallback_log = as_object( - object_value(fallback_owner_loss, "node_log_before")?, - "fallback node log", - )?; - let fallback_candidate = as_object( - object_value(fallback_owner_loss, "candidate_record")?, - "fallback candidate", - )?; - let fallback_candidate_advertisement = as_object( - object_value(fallback_candidate, "advertisement")?, - "fallback candidate advertisement", - )?; - let fallback_members = member_nodes(fallback_log)?; - if selection.fallback.failed_session != selection.second_owner_loss.successor_session - || selection.fallback.failed_node != selection.second_owner_loss.successor_node - || !boolean(fallback_candidate, "live")? - || session(fallback_candidate, "session")? != selection.fallback.successor_session - || node_id(fallback_candidate_advertisement, "node")? != selection.fallback.successor_node - || fallback_members - != selection - .fallback - .failed_log_members - .iter() - .map(String::as_str) - .collect::>() - || fallback_members - .iter() - .any(|member| *member == selection.fallback.successor_node) - { - return Err(Error::Control("fallback selection identity")); - } - Ok(()) -} - -fn validate_selection_cycle( - cycle: &SelectionCycle, - expected_failed_session: &str, - expected_successor_session: &str, - selected_original_follower: bool, -) -> Result<()> { - validate_session_value(&cycle.failed_session)?; - validate_session_value(&cycle.successor_session)?; - validate_node_value(&cycle.failed_node)?; - validate_node_value(&cycle.successor_node)?; - if cycle.failed_session != expected_failed_session - || cycle.successor_session != expected_successor_session - || cycle.failed_session == cycle.successor_session - || cycle.failed_node == cycle.successor_node - || cycle.selected_original_follower != selected_original_follower - || cycle.terminal_result != "succeeded" - || cycle.failed_log_members.is_empty() - || cycle - .failed_log_members - .iter() - .any(|member| validate_node_value(member).is_err()) - || cycle - .failed_log_members - .windows(2) - .any(|members| members[0] >= members[1]) - { - return Err(Error::Control("cluster selection evidence")); - } - Ok(()) -} - -fn placement_node_for_session<'a>( - placement: &'a Map, - expected_session: &str, -) -> Result<&'a str> { - for label in ["node_a", "node_b", "node_c", "node_d"] { - let node = as_object(object_value(placement, label)?, "placement node")?; - if session(node, "session")? == expected_session { - return node_id( - as_object( - object_value(node, "advertisement")?, - "placement advertisement", - )?, - "node", - ); - } - } - Err(Error::Control("selection session is not in placement")) -} diff --git a/crates/crab-cell-runtime/src/qualification/profile.rs b/crates/crab-cell-runtime/src/qualification/profile.rs deleted file mode 100644 index 39e6bf267..000000000 --- a/crates/crab-cell-runtime/src/qualification/profile.rs +++ /dev/null @@ -1,571 +0,0 @@ -//! Qualification profiles, cases, and primitive coverage. - -use super::*; - -pub(super) const MAX_LABEL_BYTES: usize = 256; -pub(super) const MAX_METRICS: usize = 64; -pub(super) const MAX_RECEIPT_BYTES: usize = 1 << 20; -pub(super) const MAX_QUALIFICATION_CELLS: u64 = 1_000_000; -pub(super) const MAX_QUALIFICATION_OPERATIONS: u64 = 100_000_000; -pub(super) const MAX_QUALIFICATION_DURATION_SECS: u64 = 7 * 24 * 60 * 60; -pub(super) const MAX_QUALIFICATION_CONCURRENCY: usize = 1_024; - -/// Maximum age of protected qualification evidence accepted by a release gate. -pub const QUALIFICATION_PROTECTED_EVIDENCE_MAX_AGE_MS: u64 = 7 * 24 * 60 * 60 * 1_000; -/// Clock skew tolerated when a protected receipt is checked by a release gate. -pub const QUALIFICATION_PROTECTED_EVIDENCE_MAX_CLOCK_SKEW_MS: u64 = 5 * 60 * 1_000; - -/// Current wire schema for qualification evidence. -pub const QUALIFICATION_SCHEMA_VERSION: u32 = 5; -/// Schema for a manifest that binds one receipt to every qualification row. -pub const QUALIFICATION_MATRIX_SCHEMA_VERSION: u32 = 2; -/// Schema for a versioned workload threshold profile. -pub const QUALIFICATION_PROFILE_SCHEMA_VERSION: u32 = 2; -/// Required workload rows for a complete release qualification matrix. -pub const QUALIFICATION_MATRIX_ROWS: &[&str] = &[ - "protocol", - "storage", - "publication", - "warm-path", - "churn", - "fleet", - "failover", - "primitives", - "accounting", - "compatibility", -]; - -/// Versioned thresholds used to decide whether a qualification run is admissible. -/// -/// The profile is intentionally separate from measured receipts: thresholds must -/// be selected before a candidate run and its digest is the identity bound by -/// the harness and release gate. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationProfile { - pub(super) schema_version: u32, - pub(super) name: String, - pub(super) minimum_cells: u64, - pub(super) minimum_operations: u64, - pub(super) minimum_duration_secs: u64, - pub(super) maximum_p99_latency_ms: u64, - pub(super) minimum_throughput_ops_per_sec: u64, - pub(super) maximum_peak_rss_bytes: u64, - pub(super) maximum_local_disk_bytes: u64, - pub(super) maximum_file_descriptors: u64, - pub(super) maximum_bucket_calls: u64, - pub(super) provider: String, - pub(super) topology: String, -} - -impl QualificationProfile { - /// Creates one immutable threshold profile. - pub fn new( - name: String, - minimum_cells: u64, - minimum_operations: u64, - minimum_duration_secs: u64, - maximum_p99_latency_ms: u64, - ) -> Result { - validate_label(&name, "qualification profile name")?; - if minimum_cells == 0 - || minimum_operations == 0 - || minimum_duration_secs == 0 - || maximum_p99_latency_ms == 0 - { - return Err(Error::Control("qualification profile threshold is zero")); - } - let profile = Self { - schema_version: QUALIFICATION_PROFILE_SCHEMA_VERSION, - name, - minimum_cells, - minimum_operations, - minimum_duration_secs, - maximum_p99_latency_ms, - minimum_throughput_ops_per_sec: 0, - maximum_peak_rss_bytes: 0, - maximum_local_disk_bytes: 0, - maximum_file_descriptors: 0, - maximum_bucket_calls: 0, - provider: String::new(), - topology: String::new(), - }; - profile.validate()?; - Ok(profile) - } - - /// Returns the deterministic local correctness profile. - pub fn pr_contract() -> Self { - Self::built_in("pr-contract-v1", 1, 1, 1, 5_000, "", "", 0, 0, 0, 0, 0) - } - - /// Returns the three-process provider iteration profile. - pub fn local_provider() -> Self { - Self::built_in( - "local-provider-v1", - 256, - 1_000_000, - 60, - 1_000, - "rustfs", - "three-process", - 1, - 8 * 1024 * 1024 * 1024, - 20 * 1024 * 1024 * 1024, - 10_000, - 10_000_000, - ) - } - - /// Returns the dedicated scale profile. - pub fn scale() -> Self { - Self::built_in( - "scale-v1", - 10_000, - 10_000_000, - 3_600, - 500, - "rustfs", - "dedicated-hosts", - 2_777, - 32 * 1024 * 1024 * 1024, - 200 * 1024 * 1024 * 1024, - 100_000, - 100_000_000, - ) - } - - /// Returns the protected Kubernetes fault profile. - pub fn fault() -> Self { - Self::built_in( - "fault-v1", - 256, - 1_000_000, - 60, - 1_000, - "rustfs", - "kubernetes", - 1, - 8 * 1024 * 1024 * 1024, - 20 * 1024 * 1024 * 1024, - 10_000, - 10_000_000, - ) - } - - /// Returns the protected Kubernetes fault profile for an S3 deployment. - pub fn fault_s3() -> Self { - Self::fault_for("fault-s3-v1", "s3") - } - - /// Returns the protected Kubernetes fault profile for a GCS deployment. - pub fn fault_gcs() -> Self { - Self::fault_for("fault-gcs-v1", "gcs") - } - - /// Returns the protected Kubernetes fault profile for an Azure deployment. - pub fn fault_azure() -> Self { - Self::fault_for("fault-azure-v1", "azure") - } - - /// Returns the provider-specific correctness profile. - pub fn provider() -> Self { - Self::built_in( - "provider-v1", - 256, - 1_000_000, - 60, - 1_000, - "provider-matrix", - "three-process", - 1, - 8 * 1024 * 1024 * 1024, - 20 * 1024 * 1024 * 1024, - 10_000, - 10_000_000, - ) - } - - /// Returns the protected provider correctness profile for S3. - pub fn provider_s3() -> Self { - Self::provider_for("provider-s3-v1", "s3") - } - - /// Returns the protected provider correctness profile for GCS. - pub fn provider_gcs() -> Self { - Self::provider_for("provider-gcs-v1", "gcs") - } - - /// Returns the protected provider correctness profile for Azure. - pub fn provider_azure() -> Self { - Self::provider_for("provider-azure-v1", "azure") - } - - /// Returns the rolling-release compatibility profile. - pub fn compatibility() -> Self { - Self::built_in( - "compatibility-v1", - 256, - 1_000_000, - 60, - 1_000, - "rustfs", - "rolling", - 1, - 8 * 1024 * 1024 * 1024, - 20 * 1024 * 1024 * 1024, - 10_000, - 10_000_000, - ) - } - - /// Returns the profile name used in signed receipts. - #[must_use] - pub fn name(&self) -> &str { - &self.name - } - - /// Returns the minimum Cell cardinality. - #[must_use] - pub const fn minimum_cells(&self) -> u64 { - self.minimum_cells - } - - /// Returns the minimum operation count. - #[must_use] - pub const fn minimum_operations(&self) -> u64 { - self.minimum_operations - } - - /// Returns the minimum steady-state duration. - #[must_use] - pub const fn minimum_duration_secs(&self) -> u64 { - self.minimum_duration_secs - } - - /// Returns the maximum permitted p99 latency. - #[must_use] - pub const fn maximum_p99_latency_ms(&self) -> u64 { - self.maximum_p99_latency_ms - } - - /// Returns the minimum measured throughput, or zero when the profile does - /// not impose a throughput threshold. - #[must_use] - pub const fn minimum_throughput_ops_per_sec(&self) -> u64 { - self.minimum_throughput_ops_per_sec - } - - /// Returns the maximum permitted measured peak RSS, or zero when omitted. - #[must_use] - pub const fn maximum_peak_rss_bytes(&self) -> u64 { - self.maximum_peak_rss_bytes - } - - /// Returns the maximum permitted measured local-disk usage, or zero when omitted. - #[must_use] - pub const fn maximum_local_disk_bytes(&self) -> u64 { - self.maximum_local_disk_bytes - } - - /// Returns the maximum permitted measured file-descriptor count, or zero when omitted. - #[must_use] - pub const fn maximum_file_descriptors(&self) -> u64 { - self.maximum_file_descriptors - } - - /// Returns the maximum permitted object-store call count, or zero when omitted. - #[must_use] - pub const fn maximum_bucket_calls(&self) -> u64 { - self.maximum_bucket_calls - } - - /// Returns the required provider label, or an empty value for a generic profile. - #[must_use] - pub fn required_provider(&self) -> &str { - &self.provider - } - - /// Returns the required topology label, or an empty value for a generic profile. - #[must_use] - pub fn required_topology(&self) -> &str { - &self.topology - } - - /// Returns whether this profile represents release evidence rather than a - /// local contract check. - #[must_use] - pub fn requires_protected_evidence(&self) -> bool { - self != &Self::pr_contract() - } - - /// Returns whether the protected profile names a provider whose - /// conditional, range, and multipart semantics require raw evidence. - #[must_use] - pub fn requires_provider_evidence(&self) -> bool { - self.requires_protected_evidence() && !self.provider.is_empty() - } - - /// Returns whether this profile requires an injected fault schedule and - /// ownership transition evidence. - #[must_use] - pub fn requires_fault_injection(&self) -> bool { - self.name == "fault-v1" || self.name.starts_with("fault-") - } - - /// Returns whether this named deployment profile requires every primitive - /// lifecycle case to be observed by the typed executor. - #[must_use] - pub fn requires_lifecycle_case_coverage(&self) -> bool { - self.requires_protected_evidence() && !self.provider.is_empty() && !self.topology.is_empty() - } - - pub(super) fn requires_resource_measurements(&self) -> bool { - self.maximum_peak_rss_bytes != 0 - || self.maximum_local_disk_bytes != 0 - || self.maximum_file_descriptors != 0 - || self.maximum_bucket_calls != 0 - } - - /// Encodes the canonical threshold profile. - pub fn encode(&self) -> Result> { - self.validate()?; - serde_json::to_vec(self).map_err(Error::from) - } - - /// Decodes and revalidates one canonical threshold profile. - pub fn decode(bytes: &[u8]) -> Result { - let profile: Self = serde_json::from_slice(bytes)?; - profile.validate()?; - let mut canonical_end = bytes.len(); - while canonical_end > 0 && matches!(bytes[canonical_end - 1], b' ' | b'\n' | b'\r' | b'\t') - { - canonical_end -= 1; - } - if profile.encode()?.as_slice() != &bytes[..canonical_end] { - return Err(Error::Control("qualification profile is not canonical")); - } - Ok(profile) - } - - /// Returns the digest that must be bound to a candidate receipt/artifact. - pub fn digest(&self) -> Result { - Ok(Digest::from_bytes( - *blake3::hash(&self.encode()?).as_bytes(), - )) - } - - #[expect( - clippy::too_many_arguments, - reason = "built-in profiles keep every signed threshold explicit" - )] - pub(super) fn built_in( - name: &'static str, - minimum_cells: u64, - minimum_operations: u64, - minimum_duration_secs: u64, - maximum_p99_latency_ms: u64, - provider: &'static str, - topology: &'static str, - minimum_throughput_ops_per_sec: u64, - maximum_peak_rss_bytes: u64, - maximum_local_disk_bytes: u64, - maximum_file_descriptors: u64, - maximum_bucket_calls: u64, - ) -> Self { - Self { - schema_version: QUALIFICATION_PROFILE_SCHEMA_VERSION, - name: name.to_owned(), - minimum_cells, - minimum_operations, - minimum_duration_secs, - maximum_p99_latency_ms, - minimum_throughput_ops_per_sec, - maximum_peak_rss_bytes, - maximum_local_disk_bytes, - maximum_file_descriptors, - maximum_bucket_calls, - provider: provider.to_owned(), - topology: topology.to_owned(), - } - } - - pub(super) fn fault_for(name: &'static str, provider: &'static str) -> Self { - Self::built_in( - name, - 256, - 1_000_000, - 60, - 1_000, - provider, - "kubernetes", - 1, - 8 * 1024 * 1024 * 1024, - 20 * 1024 * 1024 * 1024, - 10_000, - 10_000_000, - ) - } - - pub(super) fn provider_for(name: &'static str, provider: &'static str) -> Self { - Self::built_in( - name, - 256, - 1_000_000, - 60, - 1_000, - provider, - "three-process", - 1, - 8 * 1024 * 1024 * 1024, - 20 * 1024 * 1024 * 1024, - 10_000, - 10_000_000, - ) - } - - pub(super) fn validate(&self) -> Result<()> { - if self.schema_version != QUALIFICATION_PROFILE_SCHEMA_VERSION - || self.minimum_cells == 0 - || self.minimum_operations == 0 - || self.minimum_duration_secs == 0 - || self.maximum_p99_latency_ms == 0 - || self.minimum_cells > MAX_QUALIFICATION_CELLS - || self.minimum_operations > MAX_QUALIFICATION_OPERATIONS - || self.minimum_duration_secs > MAX_QUALIFICATION_DURATION_SECS - || self.minimum_throughput_ops_per_sec > MAX_QUALIFICATION_OPERATIONS - || self.maximum_peak_rss_bytes > (u64::from(u32::MAX) << 32) - || self.maximum_local_disk_bytes > (u64::from(u32::MAX) << 32) - || self.maximum_file_descriptors > 1_000_000 - || self.maximum_bucket_calls > MAX_QUALIFICATION_OPERATIONS - { - return Err(Error::Control("invalid qualification profile")); - } - validate_label(&self.name, "qualification profile name")?; - if !self.provider.is_empty() { - validate_label(&self.provider, "qualification profile provider")?; - } - if !self.topology.is_empty() { - validate_label(&self.topology, "qualification profile topology")?; - } - Ok(()) - } -} - -/// Primitive rows exercised by the canonical mixed-load qualification driver. -pub const QUALIFICATION_PRIMITIVES: &[&str] = &[ - "sql", "kv", "blob", "queue", "cron", "workflow", "activity", "effects", -]; - -/// Resource measurements that protected run artifacts must carry. -pub const QUALIFICATION_RESOURCE_METRICS: &[(&str, &str)] = &[ - ("peak_rss_bytes", "bytes"), - ("peak_local_disk_bytes", "bytes"), - ("peak_file_descriptors", "count"), - ("bucket_calls", "count"), -]; - -/// Lifecycle cases that a canonical qualification schedule exposes to its -/// application-specific executor. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -#[repr(u8)] -pub enum QualificationCase { - /// The ordinary success path. - Happy = 0, - /// A delivery that must be retried before it succeeds. - Retry = 1, - /// A repeated delivery that must not apply twice. - Duplicate = 2, - /// An operation that arrives after its expiry. - Expiry = 3, - /// A caller that cancels an in-flight operation. - Cancellation = 4, - /// An owner that dies while work is outstanding. - OwnerLoss = 5, - /// A Cell recovered from a verified node-log witness. - Recovery = 6, -} - -impl QualificationCase { - /// Returns the stable bounded label used by workload adapters. - #[must_use] - pub const fn name(self) -> &'static str { - match self { - Self::Happy => "happy", - Self::Retry => "retry", - Self::Duplicate => "duplicate", - Self::Expiry => "expiry", - Self::Cancellation => "cancellation", - Self::OwnerLoss => "owner-loss", - Self::Recovery => "recovery", - } - } -} - -/// Lifecycle cases in the deterministic order used by the qualification driver. -pub const QUALIFICATION_CASES: &[QualificationCase] = &[ - QualificationCase::Happy, - QualificationCase::Retry, - QualificationCase::Duplicate, - QualificationCase::Expiry, - QualificationCase::Cancellation, - QualificationCase::OwnerLoss, - QualificationCase::Recovery, -]; - -/// Minimum generated schedule length that gives every primitive every case. -pub const QUALIFICATION_CASE_COVERAGE_OPERATIONS: u64 = - (QUALIFICATION_PRIMITIVES.len() * QUALIFICATION_CASES.len()) as u64; -/// Number of bytes needed to retain one bit for every primitive/lifecycle pair. -pub const QUALIFICATION_CASE_COVERAGE_BYTES: usize = - (QUALIFICATION_PRIMITIVES.len() * QUALIFICATION_CASES.len()).div_ceil(8); - -pub(super) fn case_coverage_index(operation: QualificationOperation) -> usize { - operation.primitive_index as usize * QUALIFICATION_CASES.len() + operation.case as usize -} - -pub(super) fn mark_case_coverage(coverage: &mut [u8], operation: QualificationOperation) { - let index = case_coverage_index(operation); - coverage[index / 8] |= 1 << (index % 8); -} - -pub(super) fn has_complete_case_coverage(coverage: &[u8]) -> bool { - (0..QUALIFICATION_PRIMITIVES.len()).all(|primitive| { - (0..QUALIFICATION_CASES.len()).all(|case| { - let index = primitive * QUALIFICATION_CASES.len() + case; - coverage[index / 8] & (1 << (index % 8)) != 0 - }) - }) -} - -pub(super) fn validate_label(value: &str, field: &'static str) -> Result<()> { - if value.is_empty() - || value.len() > MAX_LABEL_BYTES - || !value.is_ascii() - || value.bytes().any(|byte| byte.is_ascii_control()) - || value.contains("@") - || value.contains("://") - || value.to_ascii_lowercase().contains("secret") - || value.to_ascii_lowercase().contains("password") - { - return Err(Error::Control(field)); - } - Ok(()) -} - -pub(super) fn validate_path(value: &str, field: &'static str) -> Result<()> { - if value.is_empty() - || value.len() > MAX_LABEL_BYTES - || !value.is_ascii() - || value.bytes().any(|byte| byte.is_ascii_control()) - || value.contains('\\') - || value.contains("://") - || Path::new(value).is_absolute() - || Path::new(value) - .components() - .any(|component| matches!(component, std::path::Component::ParentDir)) - { - return Err(Error::Control(field)); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/qualification/receipt.rs b/crates/crab-cell-runtime/src/qualification/receipt.rs deleted file mode 100644 index a18c5edfd..000000000 --- a/crates/crab-cell-runtime/src/qualification/receipt.rs +++ /dev/null @@ -1,1212 +0,0 @@ -//! Qualification receipts, provider evidence, and the matrix manifest. - -use super::profile::{ - MAX_METRICS, MAX_RECEIPT_BYTES, has_complete_case_coverage, validate_label, validate_path, -}; -use super::workload::{qualification_run_outcome_digest, valid_primitive_counts}; -use super::*; - -pub(in crate::qualification) mod evidence; -mod matrix; -mod runner; - -pub use evidence::QualificationProviderEvidence; -use evidence::{verify_provider_evidence, verify_scale_evidence}; -pub use matrix::{QualificationMatrixEntry, QualificationMatrixManifest, QualificationMetric}; -pub(super) use matrix::{ - validate_metrics, validate_resource_metric_list, validate_resource_metric_units, - validate_run_latency_metrics, -}; -pub use runner::QualificationRunner; - -/// Schema for a measured, typed execution artifact bound to one workload. -pub const QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION: u32 = 4; - -/// Schema for canonical provider-semantics evidence. -pub const QUALIFICATION_PROVIDER_EVIDENCE_SCHEMA_VERSION: u32 = 1; - -/// Bounded measured outcome consumed by protected primitive qualification. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationRunArtifact { - pub(super) schema_version: u32, - pub(super) workload: QualificationWorkload, - pub(super) profile: String, - pub(super) profile_digest: [u8; 32], - pub(super) seed: u64, - pub(super) cells: u64, - pub(super) operations: u64, - pub(super) elapsed_ms: u64, - pub(super) primitive_counts: Vec, - pub(super) case_coverage: Vec, - pub(super) outcome_digest: [u8; 32], - pub(super) metrics: Vec, -} - -impl QualificationRunArtifact { - /// Decodes and verifies one canonical measured run artifact. - pub fn decode(bytes: &[u8]) -> Result { - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification run artifact exceeds limit")); - } - let artifact: Self = serde_json::from_slice(bytes)?; - artifact.validate()?; - if serde_json::to_vec(&artifact).map_err(Error::from)? != bytes { - return Err(Error::Control( - "qualification run artifact is not canonical", - )); - } - Ok(artifact) - } - - /// Encodes one measured result with stable field ordering. - pub fn encode(&self) -> Result> { - self.validate()?; - let bytes = serde_json::to_vec(self).map_err(Error::from)?; - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification run artifact exceeds limit")); - } - Ok(bytes) - } - - /// Verifies workload identity, measured counters, and profile thresholds. - pub fn verify_for_profile(&self, profile: &QualificationProfile) -> Result<()> { - self.validate()?; - self.workload.verify_for_profile(profile)?; - if self.profile != profile.name - || self.profile_digest() != profile.digest()? - || self.workload.profile() != profile.name - || self.seed != self.workload.seed() - || self.cells != self.workload.cells() - || self.operations != self.workload.operations() - { - return Err(Error::Control("qualification run profile identity")); - } - if profile.requires_lifecycle_case_coverage() - && !has_complete_case_coverage(&self.case_coverage) - { - return Err(Error::Control("qualification run lifecycle case coverage")); - } - if profile.requires_lifecycle_case_coverage() - && self - .primitive_counts - .iter() - .any(|counts| counts.verified != counts.acknowledged) - { - return Err(Error::Control( - "qualification run acknowledged outcomes are not fully verified", - )); - } - let duration_secs = self.elapsed_ms.saturating_add(999) / 1_000; - for (name, unit, expected) in [ - ("cells", "cells", self.cells), - ("operations", "operations", self.operations), - ("duration_secs", "seconds", duration_secs), - ( - "throughput_ops_per_sec", - "ops/s", - self.operations / duration_secs, - ), - ] { - if self.threshold_metric(name, unit)? != expected { - return Err(Error::Control("qualification run measured metrics")); - } - } - if self.elapsed_ms < profile.minimum_duration_secs().saturating_mul(1_000) - || self.threshold_metric("p99_latency_ms", "ms")? > profile.maximum_p99_latency_ms() - { - return Err(Error::Control("qualification run profile threshold failed")); - } - if profile.minimum_throughput_ops_per_sec() != 0 - && self.operations / duration_secs < profile.minimum_throughput_ops_per_sec() - { - return Err(Error::Control( - "qualification run throughput threshold failed", - )); - } - self.verify_resource_metrics(profile)?; - Ok(()) - } - - /// Returns the canonical workload this run measured. - #[must_use] - pub fn workload(&self) -> &QualificationWorkload { - &self.workload - } - - /// Returns the bounded metrics captured with this measured run. - #[must_use] - pub fn metrics(&self) -> &[QualificationMetric] { - &self.metrics - } - - /// Returns the digest of the measured outcome. - #[must_use] - pub const fn outcome_digest(&self) -> Digest { - Digest::from_bytes(self.outcome_digest) - } - - pub(super) fn profile_digest(&self) -> Digest { - Digest::from_bytes(self.profile_digest) - } - - pub(super) fn threshold_metric(&self, name: &str, unit: &str) -> Result { - let mut value = None; - for metric in &self.metrics { - if metric.name() != name { - continue; - } - if metric.unit() != unit || value.replace(metric.value()).is_some() { - return Err(Error::Control("qualification run threshold metric")); - } - } - value.ok_or(Error::Control("qualification run threshold metric missing")) - } - - pub(super) fn verify_resource_metrics(&self, profile: &QualificationProfile) -> Result<()> { - if !profile.requires_resource_measurements() { - return Ok(()); - } - for (name, unit) in QUALIFICATION_RESOURCE_METRICS { - let value = self.threshold_metric(name, unit)?; - if value == 0 { - return Err(Error::Control("qualification resource metric is zero")); - } - let maximum = match *name { - "peak_rss_bytes" => profile.maximum_peak_rss_bytes(), - "peak_local_disk_bytes" => profile.maximum_local_disk_bytes(), - "peak_file_descriptors" => profile.maximum_file_descriptors(), - "bucket_calls" => profile.maximum_bucket_calls(), - _ => return Err(Error::Control("unknown qualification resource metric")), - }; - if maximum != 0 && value > maximum { - return Err(Error::Control("qualification resource threshold failed")); - } - } - Ok(()) - } - - pub(super) fn validate(&self) -> Result<()> { - if self.schema_version != QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION - || self.profile_digest.iter().all(|byte| *byte == 0) - || self.outcome_digest.iter().all(|byte| *byte == 0) - || self.elapsed_ms == 0 - || self.primitive_counts.len() != QUALIFICATION_PRIMITIVES.len() - || self.case_coverage.len() != QUALIFICATION_CASE_COVERAGE_BYTES - { - return Err(Error::Control("invalid qualification run artifact")); - } - validate_label(&self.profile, "qualification run profile")?; - let expected_primitives = QUALIFICATION_PRIMITIVES - .iter() - .copied() - .collect::>(); - let actual_primitives = self - .primitive_counts - .iter() - .map(QualificationPrimitiveCounts::primitive) - .collect::>(); - if actual_primitives != expected_primitives { - return Err(Error::Control("qualification run counters")); - } - for counts in &self.primitive_counts { - let expected = self - .workload - .primitives - .iter() - .find(|expected| expected.primitive == counts.primitive) - .ok_or(Error::Control("qualification run primitive identity"))?; - if counts.attempted != expected.attempted - || !valid_primitive_counts(counts) - || counts.verified == 0 - { - return Err(Error::Control("qualification run counters")); - } - } - if self - .primitive_counts - .iter() - .try_fold(0_u64, |total, counts| total.checked_add(counts.attempted)) - != Some(self.operations) - { - return Err(Error::Control("qualification run counters")); - } - self.workload.validate()?; - if qualification_run_outcome_digest( - &self.workload, - &self.primitive_counts, - &self.case_coverage, - )? != self.outcome_digest() - { - return Err(Error::Control("qualification run outcome")); - } - validate_metrics(&self.metrics)?; - validate_resource_metric_units(&self.metrics)?; - validate_run_latency_metrics(&self.metrics)?; - Ok(()) - } -} - -/// Reproducible evidence record for one canonical Cell qualification run. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationReceipt { - pub(super) schema_version: u32, - pub(super) source_revision: String, - pub(super) image: [u8; 32], - pub(super) provider: String, - pub(super) workload: String, - pub(super) fault: String, - pub(super) metrics: Vec, - pub(super) artifact_digest: [u8; 32], - pub(super) passed: bool, - pub(super) dirty: bool, - pub(super) toolchain: String, - pub(super) execution_profile: String, - pub(super) profile: String, - pub(super) profile_digest: [u8; 32], - pub(super) topology: String, - pub(super) workload_seed: u64, - pub(super) bucket_calls: u64, - pub(super) peak_rss_bytes: u64, - pub(super) started_at_ms: u64, - pub(super) finished_at_ms: u64, - pub(super) fault_schedule_digest: [u8; 32], - pub(super) raw_artifact_digests: Vec<[u8; 32]>, - pub(super) ownership: Vec, - pub(super) signer: [u8; 32], - pub(super) signature: Vec, -} - -/// One bounded ownership proof observed during a qualification run. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationOwnership { - pub(super) epoch: u64, - pub(super) published_sequence: u64, - pub(super) root: [u8; 32], -} - -/// Non-secret execution identity and fault evidence supplied by a protected -/// qualification harness when it binds a measured run to a receipt. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationExecutionEvidence { - /// Provider identity recorded by the harness. - pub provider: String, - /// Logical workload row in the ten-row qualification matrix. - pub workload: String, - /// Named fault schedule, or `none` for a non-fault run. - pub fault: String, - /// Toolchain identity used by the harness. - pub toolchain: String, - /// Immutable execution image/profile identity. - pub execution_profile: String, - /// Topology identity required by the selected profile. - pub topology: String, - /// Wall-clock start timestamp in Unix milliseconds. - pub started_at_ms: u64, - /// Wall-clock finish timestamp in Unix milliseconds. - pub finished_at_ms: u64, - /// Canonical fault schedule bytes retained in the raw evidence bundle. - pub fault_schedule: Vec, - /// Ownership watermarks captured before/after any injected fault. - pub ownership: Vec, - /// Whether the harness observed a dirty source or workspace. - pub dirty: bool, -} - -impl QualificationExecutionEvidence { - /// Encodes the non-secret execution evidence in its canonical JSON form. - pub fn encode(&self) -> Result> { - self.validate()?; - let bytes = serde_json::to_vec(self).map_err(Error::from)?; - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control( - "qualification execution evidence exceeds limit", - )); - } - Ok(bytes) - } - - /// Decodes and validates one canonical execution evidence file. - pub fn decode(bytes: &[u8]) -> Result { - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control( - "qualification execution evidence exceeds limit", - )); - } - let evidence: Self = serde_json::from_slice(bytes)?; - evidence.validate()?; - if evidence.encode()? != bytes { - return Err(Error::Control( - "qualification execution evidence is not canonical", - )); - } - Ok(evidence) - } - - pub(super) fn validate(&self) -> Result<()> { - validate_label(&self.provider, "qualification evidence provider")?; - validate_label(&self.workload, "qualification evidence workload")?; - validate_label(&self.fault, "qualification evidence fault")?; - validate_label(&self.toolchain, "qualification evidence toolchain")?; - validate_label( - &self.execution_profile, - "qualification evidence execution profile", - )?; - validate_label(&self.topology, "qualification evidence topology")?; - if self.started_at_ms == 0 - || self.finished_at_ms < self.started_at_ms - || self.fault_schedule.is_empty() - || self.fault_schedule.len() > MAX_RECEIPT_BYTES - || self.ownership.len() > MAX_METRICS - || self - .ownership - .iter() - .any(|proof| proof.root.iter().all(|byte| *byte == 0)) - { - return Err(Error::Control("qualification execution evidence")); - } - Ok(()) - } -} - -impl QualificationOwnership { - /// Creates one ownership/commit watermark proof. - pub fn new(epoch: u64, published_sequence: u64, root: Digest) -> Self { - Self { - epoch, - published_sequence, - root: *root.as_bytes(), - } - } - - /// Returns the Cell ownership epoch observed at the watermark. - #[must_use] - pub const fn epoch(&self) -> u64 { - self.epoch - } - - /// Returns the highest published sequence at the watermark. - #[must_use] - pub const fn published_sequence(&self) -> u64 { - self.published_sequence - } - - /// Returns the published root digest at the watermark. - #[must_use] - pub const fn root(&self) -> Digest { - Digest::from_bytes(self.root) - } -} - -impl QualificationReceipt { - /// Creates an unsigned receipt for one measured run. - pub fn new( - source_revision: String, - image: Digest, - provider: String, - workload: String, - fault: String, - metrics: Vec, - artifact_digest: Digest, - passed: bool, - ) -> Result { - validate_label(&source_revision, "qualification source revision")?; - validate_label(&provider, "qualification provider")?; - validate_label(&workload, "qualification workload")?; - validate_label(&fault, "qualification fault")?; - validate_metrics(&metrics)?; - let fault_schedule_digest = *blake3::hash(fault.as_bytes()).as_bytes(); - let artifact_digest = *artifact_digest.as_bytes(); - Ok(Self { - schema_version: QUALIFICATION_SCHEMA_VERSION, - source_revision, - image: *image.as_bytes(), - provider, - workload, - fault, - metrics, - artifact_digest, - passed, - dirty: false, - toolchain: "unknown".into(), - execution_profile: "unknown".into(), - profile: "unqualified".into(), - profile_digest: [0; 32], - topology: "local".into(), - workload_seed: 0, - bucket_calls: 0, - peak_rss_bytes: 0, - started_at_ms: 1, - finished_at_ms: 1, - fault_schedule_digest, - raw_artifact_digests: vec![artifact_digest], - ownership: Vec::new(), - signer: [0; 32], - signature: vec![0; 64], - }) - } - - /// Adds bounded, non-secret execution identity to a receipt. - pub fn with_execution( - mut self, - toolchain: String, - execution_profile: String, - topology: String, - workload_seed: u64, - bucket_calls: u64, - peak_rss_bytes: u64, - dirty: bool, - ) -> Result { - validate_label(&toolchain, "qualification toolchain")?; - validate_label(&execution_profile, "qualification execution profile")?; - validate_label(&topology, "qualification topology")?; - self.toolchain = toolchain; - self.execution_profile = execution_profile; - self.topology = topology; - self.workload_seed = workload_seed; - self.bucket_calls = bucket_calls; - self.peak_rss_bytes = peak_rss_bytes; - self.dirty = dirty; - Ok(self) - } - - /// Binds the canonical threshold profile whose limits govern this run. - pub fn with_profile(mut self, profile: &QualificationProfile) -> Result { - profile.validate()?; - self.profile = profile.name.clone(); - self.profile_digest = *profile.digest()?.as_bytes(); - Ok(self) - } - - /// Binds timestamps, fault schedule bytes, and raw artifact/ownership evidence. - pub fn with_evidence( - mut self, - started_at_ms: u64, - finished_at_ms: u64, - fault_schedule: &[u8], - raw_artifact_digests: Vec, - ownership: Vec, - ) -> Result { - if started_at_ms == 0 || finished_at_ms < started_at_ms { - return Err(Error::Control("qualification evidence timestamps")); - } - if fault_schedule.is_empty() || fault_schedule.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification fault schedule")); - } - if raw_artifact_digests.is_empty() || raw_artifact_digests.len() > MAX_METRICS { - return Err(Error::Control("qualification artifact digest count")); - } - if ownership.len() > MAX_METRICS { - return Err(Error::Control("qualification ownership proof count")); - } - if ownership - .iter() - .any(|proof| proof.root.iter().all(|byte| *byte == 0)) - { - return Err(Error::Control("qualification ownership proof")); - } - self.started_at_ms = started_at_ms; - self.finished_at_ms = finished_at_ms; - self.fault_schedule_digest = *blake3::hash(fault_schedule).as_bytes(); - self.raw_artifact_digests = raw_artifact_digests - .into_iter() - .map(|digest| *digest.as_bytes()) - .collect(); - self.ownership = ownership; - Ok(self) - } - - /// Signs the exact canonical receipt with a qualification attestation key. - pub fn attest(mut self, signing_key: &SigningKey) -> Result { - self.signer = signing_key.verifying_key().to_bytes(); - self.signature = vec![0; 64]; - self.signature = signing_key.sign(&self.signing_bytes()?).to_bytes().to_vec(); - Ok(self) - } - - /// Returns the receipt schema version. - #[must_use] - pub const fn schema_version(&self) -> u32 { - self.schema_version - } - - /// Returns the source revision the run was built from. - #[must_use] - pub fn source_revision(&self) -> &str { - &self.source_revision - } - - /// Returns the image digest the run executed. - #[must_use] - pub const fn image(&self) -> Digest { - Digest::from_bytes(self.image) - } - - /// Returns the provider the run executed against. - #[must_use] - pub fn provider(&self) -> &str { - &self.provider - } - - /// Returns the workload name. - #[must_use] - pub fn workload(&self) -> &str { - &self.workload - } - - /// Returns the fault profile the run injected. - #[must_use] - pub fn fault(&self) -> &str { - &self.fault - } - - /// Returns the bounded metrics recorded for the run. - #[must_use] - pub fn metrics(&self) -> &[QualificationMetric] { - &self.metrics - } - - /// Returns the digest of the measured artifact. - #[must_use] - pub const fn artifact_digest(&self) -> Digest { - Digest::from_bytes(self.artifact_digest) - } - - /// Reports whether the run met every threshold. - #[must_use] - pub const fn passed(&self) -> bool { - self.passed - } - - /// Reports whether the run was built from a dirty tree. - #[must_use] - pub const fn dirty(&self) -> bool { - self.dirty - } - - /// Returns the toolchain the run used. - #[must_use] - pub fn toolchain(&self) -> &str { - &self.toolchain - } - - /// Returns the execution profile the harness applied. - #[must_use] - pub fn execution_profile(&self) -> &str { - &self.execution_profile - } - - /// Returns the threshold profile name. - #[must_use] - pub fn profile(&self) -> &str { - &self.profile - } - - /// Returns the digest of the exact threshold profile signed into this receipt. - #[must_use] - pub const fn profile_digest(&self) -> Digest { - Digest::from_bytes(self.profile_digest) - } - - /// Returns the topology the run used. - #[must_use] - pub fn topology(&self) -> &str { - &self.topology - } - - /// Returns the seed the workload was generated with. - #[must_use] - pub const fn workload_seed(&self) -> u64 { - self.workload_seed - } - - /// Returns the object-store bucket calls the run made. - #[must_use] - pub const fn bucket_calls(&self) -> u64 { - self.bucket_calls - } - - /// Returns the peak resident set size the run observed. - #[must_use] - pub const fn peak_rss_bytes(&self) -> u64 { - self.peak_rss_bytes - } - - /// Returns the wall-clock start time in milliseconds. - #[must_use] - pub const fn started_at_ms(&self) -> u64 { - self.started_at_ms - } - - /// Returns the wall-clock finish time in milliseconds. - #[must_use] - pub const fn finished_at_ms(&self) -> u64 { - self.finished_at_ms - } - - /// Returns the digest of the exact fault schedule. - #[must_use] - pub const fn fault_schedule_digest(&self) -> Digest { - Digest::from_bytes(self.fault_schedule_digest) - } - - /// Iterates the digests of the raw artifacts the run produced. - pub fn raw_artifact_digests(&self) -> impl Iterator + '_ { - self.raw_artifact_digests - .iter() - .copied() - .map(Digest::from_bytes) - } - - /// Returns the ownership watermarks the run proved. - #[must_use] - pub fn ownership(&self) -> &[QualificationOwnership] { - &self.ownership - } - - /// Returns the attestation key that signed the receipt. - #[must_use] - pub const fn signer(&self) -> [u8; 32] { - self.signer - } - - /// Encodes the canonical receipt, refusing one that violates its contract. - pub fn encode(&self) -> Result> { - self.validate_contract()?; - let bytes = serde_json::to_vec(self).map_err(Error::from)?; - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification receipt exceeds limit")); - } - Ok(bytes) - } - - /// Decodes and validates a canonical receipt. - pub fn decode(bytes: &[u8]) -> Result { - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification receipt exceeds limit")); - } - let receipt: Self = serde_json::from_slice(bytes)?; - if receipt.schema_version != QUALIFICATION_SCHEMA_VERSION { - return Err(Error::Control("qualification receipt schema version")); - } - receipt.validate_contract()?; - if receipt.dirty { - return Err(Error::Control("qualification receipt has dirty source")); - } - let key = VerifyingKey::from_bytes(&receipt.signer).map_err(Error::PeerSignature)?; - receipt.verify_signature(&key)?; - if receipt.encode()? != bytes { - return Err(Error::Control("qualification receipt is not canonical")); - } - Ok(receipt) - } - - /// Verifies a decoded, passing receipt against the expected release and artifact. - /// - /// Failed receipts remain retainable evidence but cannot satisfy this release gate. - pub fn verify_for(&self, source_revision: &str, image: Digest, artifact: &[u8]) -> Result<()> { - self.verify_for_artifacts(source_revision, image, &[artifact]) - } - - /// Verifies a receipt against the exact threshold profile used for the run. - pub fn verify_for_profile( - &self, - source_revision: &str, - image: Digest, - profile: &QualificationProfile, - artifacts: &[&[u8]], - ) -> Result<()> { - if self.profile != profile.name || self.profile_digest() != profile.digest()? { - return Err(Error::Control("qualification profile identity")); - } - if (!profile.provider.is_empty() && self.provider != profile.provider) - || (!profile.topology.is_empty() && self.topology != profile.topology) - { - return Err(Error::Control("qualification environment identity")); - } - self.verify_execution_environment(profile)?; - self.verify_for_artifacts(source_revision, image, artifacts) - } - - /// Verifies a receipt against a pinned qualification attestation key. - pub fn verify_for_trusted_signer( - &self, - source_revision: &str, - image: Digest, - artifact: &[u8], - trusted_signer: [u8; 32], - ) -> Result<()> { - self.verify_for_artifacts(source_revision, image, &[artifact])?; - self.verify_trusted_signer(trusted_signer) - } - - /// Verifies a receipt against a profile and a pinned attestation key. - pub fn verify_for_profile_with_signer( - &self, - source_revision: &str, - image: Digest, - profile: &QualificationProfile, - artifacts: &[&[u8]], - trusted_signer: [u8; 32], - ) -> Result<()> { - self.verify_for_profile(source_revision, image, profile, artifacts)?; - self.verify_profile_thresholds(profile)?; - self.verify_trusted_signer(trusted_signer) - } - - /// Rejects protected evidence that is stale or materially ahead of the verifier clock. - pub fn verify_fresh_at(&self, now_ms: u64) -> Result<()> { - if now_ms == 0 { - return Err(Error::Control("qualification verifier timestamp")); - } - if self.finished_at_ms - > now_ms.saturating_add(QUALIFICATION_PROTECTED_EVIDENCE_MAX_CLOCK_SKEW_MS) - { - return Err(Error::Control( - "qualification evidence timestamp is in the future", - )); - } - if now_ms.saturating_sub(self.finished_at_ms) > QUALIFICATION_PROTECTED_EVIDENCE_MAX_AGE_MS - { - return Err(Error::Control("qualification evidence is stale")); - } - Ok(()) - } - - /// Verifies the exact profile thresholds from measured receipt metrics. - pub fn verify_profile_thresholds(&self, profile: &QualificationProfile) -> Result<()> { - let cells = self.threshold_metric("cells", "cells")?; - let operations = self.threshold_metric("operations", "operations")?; - let duration_secs = self.threshold_metric("duration_secs", "seconds")?; - let p99_latency_ms = self.threshold_metric("p99_latency_ms", "ms")?; - if duration_secs == 0 { - return Err(Error::Control("qualification duration threshold is zero")); - } - if cells < profile.minimum_cells() - || operations < profile.minimum_operations() - || duration_secs < profile.minimum_duration_secs() - || p99_latency_ms > profile.maximum_p99_latency_ms() - { - return Err(Error::Control("qualification profile threshold failed")); - } - if profile.minimum_throughput_ops_per_sec() != 0 - && operations / duration_secs < profile.minimum_throughput_ops_per_sec() - { - return Err(Error::Control("qualification throughput threshold failed")); - } - let resource_envelope = profile.requires_resource_measurements(); - if (profile.maximum_peak_rss_bytes() != 0 || resource_envelope) - && (self.peak_rss_bytes == 0 - || (profile.maximum_peak_rss_bytes() != 0 - && self.peak_rss_bytes > profile.maximum_peak_rss_bytes())) - { - return Err(Error::Control("qualification RSS threshold failed")); - } - if profile.maximum_local_disk_bytes() != 0 || resource_envelope { - let peak_local_disk_bytes = self.threshold_metric("peak_local_disk_bytes", "bytes")?; - if peak_local_disk_bytes == 0 - || (profile.maximum_local_disk_bytes() != 0 - && peak_local_disk_bytes > profile.maximum_local_disk_bytes()) - { - return Err(Error::Control("qualification disk threshold failed")); - } - } - if profile.maximum_file_descriptors() != 0 || resource_envelope { - let peak_file_descriptors = self.threshold_metric("peak_file_descriptors", "count")?; - if peak_file_descriptors == 0 - || (profile.maximum_file_descriptors() != 0 - && peak_file_descriptors > profile.maximum_file_descriptors()) - { - return Err(Error::Control( - "qualification file-descriptor threshold failed", - )); - } - } - if (profile.maximum_bucket_calls() != 0 || resource_envelope) - && (self.bucket_calls == 0 - || (profile.maximum_bucket_calls() != 0 - && self.bucket_calls > profile.maximum_bucket_calls())) - { - return Err(Error::Control( - "qualification object-store threshold failed", - )); - } - Ok(()) - } - - /// Verifies the receipt's Ed25519 signature against an expected public key. - pub fn verify_trusted_signer(&self, trusted_signer: [u8; 32]) -> Result<()> { - if self.signer != trusted_signer { - return Err(Error::Control("qualification receipt signer")); - } - let key = VerifyingKey::from_bytes(&trusted_signer).map_err(Error::PeerSignature)?; - self.verify_signature(&key) - } - - /// Verifies every raw artifact digest listed by a passing receipt. - pub fn verify_for_artifacts( - &self, - source_revision: &str, - image: Digest, - artifacts: &[&[u8]], - ) -> Result<()> { - if !self.passed { - return Err(Error::Control("qualification receipt is not passed")); - } - validate_label(source_revision, "qualification source revision")?; - if self.source_revision != source_revision || self.image() != image { - return Err(Error::Control("qualification release identity")); - } - if artifacts.len() != self.raw_artifact_digests.len() || artifacts.is_empty() { - return Err(Error::Control("qualification artifact evidence count")); - } - for (expected, artifact) in self.raw_artifact_digests().zip(artifacts.iter().copied()) { - if expected != Digest::from_bytes(*blake3::hash(artifact).as_bytes()) { - return Err(Error::Control("qualification artifact digest")); - } - } - if self.artifact_digest() != Digest::from_bytes(*blake3::hash(artifacts[0]).as_bytes()) { - return Err(Error::Control("qualification primary artifact digest")); - } - let encoded = self.encode()?; - let decoded = Self::decode(&encoded)?; - if decoded != *self { - return Err(Error::Control( - "qualification receipt changed during verification", - )); - } - Ok(()) - } - - /// Verifies one complete matrix against an exact source/image identity. - pub fn verify_matrix( - source_revision: &str, - image: Digest, - evidence: &[(&str, &QualificationReceipt, &[&[u8]])], - ) -> Result<()> { - if evidence.len() != QUALIFICATION_MATRIX_ROWS.len() { - return Err(Error::Control("qualification matrix evidence count")); - } - let expected = QUALIFICATION_MATRIX_ROWS - .iter() - .copied() - .collect::>(); - let actual = evidence - .iter() - .map(|(workload, _, _)| *workload) - .collect::>(); - if actual != expected || actual.len() != evidence.len() { - return Err(Error::Control("qualification matrix evidence rows")); - } - for (workload, receipt, artifacts) in evidence { - if *workload != receipt.workload() { - return Err(Error::Control("qualification matrix receipt workload")); - } - receipt.verify_for_artifacts(source_revision, image, artifacts)?; - } - Ok(()) - } - - /// Verifies a complete matrix and requires one exact threshold profile. - pub fn verify_matrix_for_profile( - source_revision: &str, - image: Digest, - profile: &QualificationProfile, - evidence: &[(&str, &QualificationReceipt, &[&[u8]])], - ) -> Result<()> { - if evidence.len() != QUALIFICATION_MATRIX_ROWS.len() { - return Err(Error::Control("qualification matrix evidence count")); - } - let expected = QUALIFICATION_MATRIX_ROWS - .iter() - .copied() - .collect::>(); - let actual = evidence - .iter() - .map(|(workload, _, _)| *workload) - .collect::>(); - if actual != expected || actual.len() != evidence.len() { - return Err(Error::Control("qualification matrix evidence rows")); - } - for (workload, receipt, artifacts) in evidence { - if *workload != receipt.workload() { - return Err(Error::Control("qualification matrix receipt workload")); - } - receipt.verify_for_profile(source_revision, image, profile, artifacts)?; - receipt.verify_profile_thresholds(profile)?; - if *workload == "primitives" { - receipt.verify_primitive_workload(profile, artifacts)?; - } - } - Ok(()) - } - - /// Verifies a complete profile matrix with pinned attestation and thresholds. - pub fn verify_matrix_for_profile_with_signer( - source_revision: &str, - image: Digest, - profile: &QualificationProfile, - evidence: &[(&str, &QualificationReceipt, &[&[u8]])], - trusted_signer: [u8; 32], - ) -> Result<()> { - if evidence.len() != QUALIFICATION_MATRIX_ROWS.len() { - return Err(Error::Control("qualification matrix evidence count")); - } - let expected = QUALIFICATION_MATRIX_ROWS - .iter() - .copied() - .collect::>(); - let actual = evidence - .iter() - .map(|(workload, _, _)| *workload) - .collect::>(); - if actual != expected || actual.len() != evidence.len() { - return Err(Error::Control("qualification matrix evidence rows")); - } - for (workload, receipt, artifacts) in evidence { - if *workload != receipt.workload() { - return Err(Error::Control("qualification matrix receipt workload")); - } - receipt.verify_for_profile_with_signer( - source_revision, - image, - profile, - artifacts, - trusted_signer, - )?; - if *workload == "primitives" { - receipt.verify_primitive_workload(profile, artifacts)?; - receipt.verify_primitive_run_artifact(profile, artifacts)?; - } - } - Ok(()) - } - - /// Verifies a signed protected matrix and requires every row to be recent. - pub fn verify_matrix_for_profile_with_signer_fresh_at( - source_revision: &str, - image: Digest, - profile: &QualificationProfile, - evidence: &[(&str, &QualificationReceipt, &[&[u8]])], - trusted_signer: [u8; 32], - now_ms: u64, - ) -> Result<()> { - Self::verify_matrix_for_profile_with_signer( - source_revision, - image, - profile, - evidence, - trusted_signer, - )?; - for (_, receipt, _) in evidence { - receipt.verify_fresh_at(now_ms)?; - } - Ok(()) - } - - pub(super) fn validate_contract(&self) -> Result<()> { - if self.schema_version != QUALIFICATION_SCHEMA_VERSION - || self.dirty - || self.signer.iter().all(|byte| *byte == 0) - || self.signature.len() != 64 - || self.signature.iter().all(|byte| *byte == 0) - { - return Err(Error::Control("qualification receipt is not attested")); - } - validate_label(&self.source_revision, "qualification source revision")?; - validate_label(&self.provider, "qualification provider")?; - validate_label(&self.workload, "qualification workload")?; - validate_label(&self.fault, "qualification fault")?; - validate_label(&self.toolchain, "qualification toolchain")?; - validate_label(&self.execution_profile, "qualification execution profile")?; - validate_label(&self.profile, "qualification profile")?; - validate_label(&self.topology, "qualification topology")?; - validate_metrics(&self.metrics)?; - if self.started_at_ms == 0 - || self.finished_at_ms < self.started_at_ms - || self.fault_schedule_digest.iter().all(|byte| *byte == 0) - || self.profile_digest.iter().all(|byte| *byte == 0) - || self.raw_artifact_digests.is_empty() - || self.raw_artifact_digests.len() > MAX_METRICS - || self.ownership.len() > MAX_METRICS - || !self.raw_artifact_digests.contains(&self.artifact_digest) - { - return Err(Error::Control("qualification evidence is incomplete")); - } - if self - .ownership - .iter() - .any(|proof| proof.root.iter().all(|byte| *byte == 0)) - { - return Err(Error::Control("qualification ownership proof")); - } - Ok(()) - } - - pub(super) fn signing_bytes(&self) -> Result> { - let mut unsigned = self.clone(); - unsigned.signature = vec![0; 64]; - serde_json::to_vec(&unsigned).map_err(Error::from) - } - - pub(super) fn verify_signature(&self, key: &VerifyingKey) -> Result<()> { - let signature: [u8; 64] = self - .signature - .as_slice() - .try_into() - .map_err(|_| Error::Control("qualification receipt signature length"))?; - key.verify(&self.signing_bytes()?, &Signature::from_bytes(&signature)) - .map_err(Error::PeerSignature) - } - - pub(super) fn threshold_metric(&self, name: &str, unit: &str) -> Result { - let mut value = None; - for metric in &self.metrics { - if metric.name() != name { - continue; - } - if metric.unit() != unit || value.replace(metric.value()).is_some() { - return Err(Error::Control("qualification threshold metric")); - } - } - value.ok_or(Error::Control("qualification threshold metric missing")) - } - - pub(super) fn verify_primitive_workload( - &self, - profile: &QualificationProfile, - artifacts: &[&[u8]], - ) -> Result<()> { - let mut matching = artifacts.iter().filter_map(|artifact| { - QualificationWorkload::decode(artifact) - .ok() - .filter(|workload| workload.verify_for_profile(profile).is_ok()) - }); - let workload = matching.next().ok_or(Error::Control( - "qualification matrix is missing its canonical workload", - ))?; - if matching.next().is_some() || self.workload_seed != workload.seed() { - return Err(Error::Control( - "qualification receipt does not bind the canonical workload seed", - )); - } - Ok(()) - } - - pub(super) fn verify_primitive_run_artifact( - &self, - profile: &QualificationProfile, - artifacts: &[&[u8]], - ) -> Result<()> { - let mut matching = artifacts.iter().filter_map(|artifact| { - QualificationRunArtifact::decode(artifact) - .ok() - .filter(|run| run.verify_for_profile(profile).is_ok()) - }); - let run = matching.next().ok_or(Error::Control( - "protected primitive evidence is missing its measured run artifact", - ))?; - if matching.next().is_some() || self.workload_seed != run.workload().seed() { - return Err(Error::Control( - "qualification receipt does not bind the measured workload seed", - )); - } - let mut workloads = artifacts.iter().filter_map(|artifact| { - QualificationWorkload::decode(artifact) - .ok() - .filter(|workload| workload.verify_for_profile(profile).is_ok()) - }); - if let Some(workload) = workloads.next() - && (workloads.next().is_some() || workload != *run.workload()) - { - return Err(Error::Control( - "qualification receipt does not bind the measured workload", - )); - } - verify_provider_evidence(profile, &run, artifacts)?; - verify_scale_evidence(profile, &run, artifacts)?; - for (name, unit) in [ - ("cells", "cells"), - ("operations", "operations"), - ("duration_secs", "seconds"), - ("throughput_ops_per_sec", "ops/s"), - ("p50_latency_ms", "ms"), - ("p95_latency_ms", "ms"), - ("p99_latency_ms", "ms"), - ("max_latency_ms", "ms"), - ] { - if self.threshold_metric(name, unit)? != run.threshold_metric(name, unit)? { - return Err(Error::Control( - "qualification receipt does not bind measured run metrics", - )); - } - } - if profile.requires_resource_measurements() { - for (name, unit) in QUALIFICATION_RESOURCE_METRICS { - let receipt_value = match *name { - "peak_rss_bytes" => self.peak_rss_bytes, - "bucket_calls" => self.bucket_calls, - "peak_local_disk_bytes" | "peak_file_descriptors" => { - self.threshold_metric(name, unit)? - } - _ => return Err(Error::Control("unknown qualification resource metric")), - }; - if receipt_value != run.threshold_metric(name, unit)? { - return Err(Error::Control( - "qualification receipt does not bind resource measurements", - )); - } - } - } - Ok(()) - } - - pub(super) fn verify_execution_environment( - &self, - profile: &QualificationProfile, - ) -> Result<()> { - if !profile.requires_protected_evidence() { - return Ok(()); - } - if self.finished_at_ms.saturating_sub(self.started_at_ms) - < profile.minimum_duration_secs().saturating_mul(1_000) - { - return Err(Error::Control( - "protected qualification evidence duration is below profile minimum", - )); - } - if self.topology == "local" - || self.toolchain == "unknown" - || self.execution_profile != "release" - || self.bucket_calls == 0 - || self.peak_rss_bytes == 0 - || self.ownership.is_empty() - { - return Err(Error::Control( - "protected qualification evidence lacks measured environment proof", - )); - } - if profile.requires_fault_injection() - && (self.fault.eq_ignore_ascii_case("none") - || self.fault_schedule_digest == *blake3::hash(b"none").as_bytes() - || self.ownership.len() < 2 - // Two repeated samples prove no failover; require a real watermark advance. - || !self.ownership.windows(2).any(|observations| { - observations[1].epoch > observations[0].epoch - || observations[1].published_sequence > observations[0].published_sequence - }) - || self.ownership.windows(2).any(|observations| { - observations[1].epoch < observations[0].epoch - || observations[1].published_sequence < observations[0].published_sequence - })) - { - return Err(Error::Control( - "fault qualification evidence lacks an injected ownership transition", - )); - } - Ok(()) - } -} diff --git a/crates/crab-cell-runtime/src/qualification/receipt/evidence.rs b/crates/crab-cell-runtime/src/qualification/receipt/evidence.rs deleted file mode 100644 index e54972888..000000000 --- a/crates/crab-cell-runtime/src/qualification/receipt/evidence.rs +++ /dev/null @@ -1,407 +0,0 @@ -//! Provider-semantics evidence bound to a protected primitives run. - -use super::*; - -/// Canonical provider-semantics evidence bound to a protected primitives run. -/// -/// Provider profiles require one such artifact proving conditional mutation, -/// bounded range reads, and multipart behavior. The artifact is raw signed -/// evidence; the receipt only stores its digest. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationProviderEvidence { - pub(super) schema_version: u32, - pub(super) provider: String, - pub(super) profile: String, - pub(super) profile_digest: [u8; 32], - pub(super) workload_seed: u64, - pub(super) conditional: bool, - pub(super) range: bool, - pub(super) multipart: bool, -} - -impl QualificationProviderEvidence { - /// Creates one provider-semantics artifact for a measured workload. - pub fn new( - profile: &QualificationProfile, - workload_seed: u64, - conditional: bool, - range: bool, - multipart: bool, - ) -> Result { - profile.validate()?; - if profile.required_provider().is_empty() { - return Err(Error::Control( - "provider evidence requires a named qualification provider", - )); - } - let evidence = Self { - schema_version: QUALIFICATION_PROVIDER_EVIDENCE_SCHEMA_VERSION, - provider: profile.required_provider().to_owned(), - profile: profile.name.clone(), - profile_digest: *profile.digest()?.as_bytes(), - workload_seed, - conditional, - range, - multipart, - }; - evidence.validate()?; - Ok(evidence) - } - - /// Encodes canonical provider-semantics evidence. - pub fn encode(&self) -> Result> { - self.validate()?; - let bytes = serde_json::to_vec(self).map_err(Error::from)?; - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control( - "qualification provider evidence exceeds limit", - )); - } - Ok(bytes) - } - - /// Decodes and validates canonical provider-semantics evidence. - pub fn decode(bytes: &[u8]) -> Result { - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control( - "qualification provider evidence exceeds limit", - )); - } - let evidence: Self = serde_json::from_slice(bytes)?; - evidence.validate()?; - if evidence.encode()? != bytes { - return Err(Error::Control( - "qualification provider evidence is not canonical", - )); - } - Ok(evidence) - } - - pub(super) fn verify_for( - &self, - profile: &QualificationProfile, - workload_seed: u64, - ) -> Result<()> { - if !profile.requires_provider_evidence() - || self.provider != profile.required_provider() - || self.profile != profile.name - || self.profile_digest != *profile.digest()?.as_bytes() - || self.workload_seed != workload_seed - || !self.conditional - || !self.range - || !self.multipart - { - return Err(Error::Control( - "qualification provider semantics are incomplete or mismatched", - )); - } - Ok(()) - } - - pub(super) fn validate(&self) -> Result<()> { - if self.schema_version != QUALIFICATION_PROVIDER_EVIDENCE_SCHEMA_VERSION - || self.profile_digest.iter().all(|byte| *byte == 0) - { - return Err(Error::Control("invalid qualification provider evidence")); - } - validate_label(&self.provider, "qualification provider evidence provider")?; - validate_label(&self.profile, "qualification provider evidence profile")?; - Ok(()) - } -} - -pub(super) fn verify_provider_evidence( - profile: &QualificationProfile, - run: &QualificationRunArtifact, - artifacts: &[&[u8]], -) -> Result<()> { - if !profile.requires_provider_evidence() { - return Ok(()); - } - let mut matches = Vec::new(); - for artifact in artifacts { - let Some(candidate) = provider_evidence_candidate(artifact)? else { - continue; - }; - matches.push(candidate); - } - let mut matches = matches.into_iter(); - let evidence = matches.next().ok_or(Error::Control( - "protected provider evidence is missing its semantics artifact", - ))?; - if matches.next().is_some() { - return Err(Error::Control( - "protected provider evidence has multiple semantics artifacts", - )); - } - evidence.verify_for(profile, run.workload().seed()) -} - -fn provider_evidence_candidate(bytes: &[u8]) -> Result> { - // Raw artifacts may use arbitrary formats, but a JSON object that starts - // claiming provider semantics must decode completely or fail the receipt; - // otherwise a partial duplicate could hide beside a valid artifact. - let Ok(value) = serde_json::from_slice::(bytes) else { - return Ok(None); - }; - let Some(object) = value.as_object() else { - return Ok(None); - }; - if !["conditional", "range", "multipart"] - .iter() - .any(|field| object.contains_key(*field)) - { - return Ok(None); - } - QualificationProviderEvidence::decode(bytes).map(Some) -} - -const SCALE_EVIDENCE_SCHEMA_VERSION: u32 = 1; -const SCALE_STATES: [&str; 5] = [ - "empty", - "sparse", - "resident", - "pending-publication", - "churned", -]; -const SCALE_CELL_COUNTS: [u64; 3] = [1_000, 5_000, 10_000]; - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -pub(in crate::qualification) struct ScaleEvidence { - schema_version: u32, - profile: String, - profile_digest: [u8; 32], - workload_seed: u64, - cell_samples: Vec, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct ScaleSample { - state: String, - target_cells: u64, - before: ScaleSnapshot, - after: ScaleSnapshot, - peak: ScaleSnapshot, -} - -impl ScaleSample { - fn charges_open_cells(&self) -> bool { - self.after - .admitted_resident_bytes - .checked_sub(self.before.admitted_resident_bytes) - .is_some_and(|bytes| { - bytes >= self.target_cells * crate::fleet::resource::ACTIVE_CELL_NATIVE_BYTES as u64 - }) - && self - .after - .admitted_file_descriptors - .checked_sub(self.before.admitted_file_descriptors) - .is_some_and(|descriptors| { - descriptors - >= self.target_cells - * crate::fleet::resource::ACTIVE_CELL_FILE_DESCRIPTORS as u64 - }) - } - - fn slope_fits_admission(&self, smaller: &Self) -> bool { - let Some(extra_cells) = self.target_cells.checked_sub(smaller.target_cells) else { - return false; - }; - let increment = |sample: &Self, field: fn(&ScaleSnapshot) -> u64| { - i128::from(field(&sample.after)) - i128::from(field(&sample.before)) - }; - let slope = - |field: fn(&ScaleSnapshot) -> u64| increment(self, field) - increment(smaller, field); - let cache_charge = - i128::from(extra_cells) * i128::from(crate::cell::worker::ACTIVE_CELL_PAGE_CACHE_BYTES); - let memory_charge = i128::from(extra_cells) - * crate::fleet::resource::ACTIVE_CELL_NATIVE_BYTES as i128 - + cache_charge - + slope(|snapshot| snapshot.retained_bytes); - // Comparing two sample sizes cancels fixed process overhead, which - // has its own node reserve and cannot be charged to each Cell. - slope(|snapshot| snapshot.rss_bytes) <= memory_charge - && slope(|snapshot| snapshot.allocator_bytes) <= memory_charge - && slope(|snapshot| snapshot.sqlite_cache_bytes) <= cache_charge - && slope(|snapshot| snapshot.file_descriptors) - <= i128::from(extra_cells) - * crate::fleet::resource::ACTIVE_CELL_FILE_DESCRIPTORS as i128 - } -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct ScaleSnapshot { - active_cells: u64, - rss_bytes: u64, - allocator_bytes: u64, - threads: u64, - file_descriptors: u64, - sqlite_cache_bytes: u64, - admitted_resident_bytes: u64, - admitted_file_descriptors: u64, - retained_bytes: u64, - local_disk_reserved_bytes: u64, - local_disk_bytes: u64, -} - -impl ScaleSnapshot { - fn values(&self) -> [u64; 11] { - [ - self.active_cells, - self.rss_bytes, - self.allocator_bytes, - self.threads, - self.file_descriptors, - self.sqlite_cache_bytes, - self.admitted_resident_bytes, - self.admitted_file_descriptors, - self.retained_bytes, - self.local_disk_reserved_bytes, - self.local_disk_bytes, - ] - } - - fn covers(&self, other: &Self) -> bool { - self.values() - .into_iter() - .zip(other.values()) - .all(|(peak, observed)| peak >= observed) - } -} - -impl ScaleEvidence { - fn decode(bytes: &[u8]) -> Result { - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification scale evidence exceeds limit")); - } - let evidence: Self = serde_json::from_slice(bytes)?; - if serde_json::to_vec(&evidence).map_err(Error::from)? != bytes { - return Err(Error::Control( - "qualification scale evidence is not canonical", - )); - } - Ok(evidence) - } - - fn verify_for( - &self, - profile: &QualificationProfile, - run: &QualificationRunArtifact, - ) -> Result<()> { - if self.schema_version != SCALE_EVIDENCE_SCHEMA_VERSION - || self.profile != profile.name() - || self.profile_digest != *profile.digest()?.as_bytes() - || self.workload_seed != run.workload().seed() - || run.workload().cells() < 10_000 - || self.cell_samples.len() != SCALE_STATES.len() * SCALE_CELL_COUNTS.len() - { - return Err(Error::Control( - "qualification scale evidence identity or samples", - )); - } - for ((state, count), sample) in SCALE_STATES - .into_iter() - .flat_map(|state| { - SCALE_CELL_COUNTS - .into_iter() - .map(move |count| (state, count)) - }) - .zip(&self.cell_samples) - { - if sample.state != state - || sample.target_cells != count - || sample.before.active_cells != 0 - || sample.after.active_cells != count - || sample.peak.active_cells < count - || sample.before.rss_bytes == 0 - || sample.before.threads == 0 - || sample.before.file_descriptors == 0 - || sample.after.rss_bytes == 0 - || sample.after.allocator_bytes == 0 - || sample.after.sqlite_cache_bytes == 0 - || sample.after.local_disk_bytes == 0 - || !sample.charges_open_cells() - || !sample.peak.covers(&sample.before) - || !sample.peak.covers(&sample.after) - { - return Err(Error::Control("qualification scale sample is incomplete")); - } - } - for samples in self - .cell_samples - .as_chunks::<{ SCALE_CELL_COUNTS.len() }>() - .0 - { - for pair in samples.windows(2) { - if !pair[1].slope_fits_admission(&pair[0]) { - return Err(Error::Control( - "qualification scale slope exceeds admission", - )); - } - } - } - for (name, unit, observed) in [ - ( - "peak_rss_bytes", - "bytes", - self.cell_samples - .iter() - .map(|sample| sample.peak.rss_bytes) - .max(), - ), - ( - "peak_local_disk_bytes", - "bytes", - self.cell_samples - .iter() - .map(|sample| sample.peak.local_disk_bytes) - .max(), - ), - ( - "peak_file_descriptors", - "count", - self.cell_samples - .iter() - .map(|sample| sample.peak.file_descriptors) - .max(), - ), - ] { - if run.threshold_metric(name, unit)? < observed.unwrap_or(0) { - return Err(Error::Control( - "qualification scale peaks exceed measured run", - )); - } - } - Ok(()) - } -} - -pub(super) fn verify_scale_evidence( - profile: &QualificationProfile, - run: &QualificationRunArtifact, - artifacts: &[&[u8]], -) -> Result<()> { - if profile.name() != "scale-v1" { - return Ok(()); - } - let mut matches = artifacts.iter().filter_map(|artifact| { - let value = serde_json::from_slice::(artifact).ok()?; - value - .as_object()? - .contains_key("cell_samples") - .then_some(*artifact) - }); - let evidence = matches.next().ok_or(Error::Control( - "protected scale evidence is missing its Cell samples", - ))?; - if matches.next().is_some() { - return Err(Error::Control( - "protected scale evidence has multiple sample artifacts", - )); - } - ScaleEvidence::decode(evidence)?.verify_for(profile, run) -} diff --git a/crates/crab-cell-runtime/src/qualification/receipt/matrix.rs b/crates/crab-cell-runtime/src/qualification/receipt/matrix.rs deleted file mode 100644 index a755671fc..000000000 --- a/crates/crab-cell-runtime/src/qualification/receipt/matrix.rs +++ /dev/null @@ -1,242 +0,0 @@ -//! Bounded metrics and the matrix manifest attached to a receipt. - -use super::*; - -/// One bounded named measurement attached to a qualification receipt. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationMetric { - pub(in crate::qualification) name: String, - pub(in crate::qualification) value: u64, - pub(in crate::qualification) unit: String, -} - -/// One receipt/artifact pair in a qualification matrix manifest. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationMatrixEntry { - pub(in crate::qualification) workload: String, - pub(in crate::qualification) receipt: String, - pub(in crate::qualification) artifacts: Vec, -} - -impl QualificationMatrixEntry { - /// Creates one manifest entry using paths relative to the manifest file. - pub fn new(workload: String, receipt: String, artifacts: Vec) -> Result { - validate_label(&workload, "qualification matrix workload")?; - validate_path(&receipt, "qualification matrix receipt path")?; - if artifacts.is_empty() || artifacts.len() > MAX_METRICS { - return Err(Error::Control("qualification matrix artifact count")); - } - for artifact in &artifacts { - validate_path(artifact, "qualification matrix artifact path")?; - } - Ok(Self { - workload, - receipt, - artifacts, - }) - } - - /// Returns the required workload row name. - #[must_use] - pub fn workload(&self) -> &str { - &self.workload - } - - /// Returns the receipt path relative to the manifest. - #[must_use] - pub fn receipt(&self) -> &str { - &self.receipt - } - - /// Returns raw-artifact paths relative to the manifest. - #[must_use] - pub fn artifacts(&self) -> &[String] { - &self.artifacts - } -} - -/// Complete, bounded manifest for release qualification evidence. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationMatrixManifest { - pub(in crate::qualification) schema_version: u32, - pub(in crate::qualification) entries: Vec, -} - -impl QualificationMatrixManifest { - /// Builds and validates a complete matrix manifest. - pub fn new(entries: Vec) -> Result { - let manifest = Self { - schema_version: QUALIFICATION_MATRIX_SCHEMA_VERSION, - entries, - }; - manifest.validate_contract()?; - Ok(manifest) - } - - /// Decodes canonical JSON and rejects incomplete or duplicate rows. - pub fn decode(bytes: &[u8]) -> Result { - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification matrix exceeds limit")); - } - let manifest: Self = serde_json::from_slice(bytes)?; - manifest.validate_contract()?; - if serde_json::to_vec(&manifest).map_err(Error::from)? != bytes { - return Err(Error::Control("qualification matrix is not canonical")); - } - Ok(manifest) - } - - /// Encodes canonical JSON for a release artifact. - pub fn encode(&self) -> Result> { - self.validate_contract()?; - let bytes = serde_json::to_vec(self).map_err(Error::from)?; - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification matrix exceeds limit")); - } - Ok(bytes) - } - - /// Returns the entries in their declared manifest order. - #[must_use] - pub fn entries(&self) -> &[QualificationMatrixEntry] { - &self.entries - } - - pub(in crate::qualification) fn validate_contract(&self) -> Result<()> { - if self.schema_version != QUALIFICATION_MATRIX_SCHEMA_VERSION - || self.entries.len() != QUALIFICATION_MATRIX_ROWS.len() - { - return Err(Error::Control("qualification matrix schema or row count")); - } - let expected = QUALIFICATION_MATRIX_ROWS - .iter() - .copied() - .collect::>(); - let actual = self - .entries - .iter() - .map(QualificationMatrixEntry::workload) - .collect::>(); - if actual.len() != self.entries.len() || actual != expected { - return Err(Error::Control("qualification matrix rows")); - } - if self - .entries - .iter() - .zip(QUALIFICATION_MATRIX_ROWS.iter().copied()) - .any(|(entry, expected)| entry.workload != expected) - { - return Err(Error::Control("qualification matrix row order")); - } - for entry in &self.entries { - QualificationMatrixEntry::new( - entry.workload.clone(), - entry.receipt.clone(), - entry.artifacts.clone(), - )?; - } - Ok(()) - } -} - -impl QualificationMetric { - /// Creates one measured metric, validating its name and unit labels. - pub fn new(name: String, value: u64, unit: String) -> Result { - validate_label(&name, "qualification metric name")?; - validate_label(&unit, "qualification metric unit")?; - Ok(Self { name, value, unit }) - } - - /// Returns the metric name. - #[must_use] - pub fn name(&self) -> &str { - &self.name - } - - /// Returns the measured value. - #[must_use] - pub const fn value(&self) -> u64 { - self.value - } - - /// Returns the unit the value is expressed in. - #[must_use] - pub fn unit(&self) -> &str { - &self.unit - } -} - -pub(in crate::qualification) fn validate_metrics(metrics: &[QualificationMetric]) -> Result<()> { - if metrics.len() > MAX_METRICS { - return Err(Error::Control("qualification metric count exceeds limit")); - } - let mut identities = BTreeSet::new(); - for metric in metrics { - validate_label(&metric.name, "qualification metric name")?; - validate_label(&metric.unit, "qualification metric unit")?; - if !identities.insert((metric.name.as_str(), metric.unit.as_str())) { - return Err(Error::Control("duplicate qualification metric")); - } - } - Ok(()) -} - -pub(in crate::qualification) fn validate_resource_metric_list( - metrics: &[QualificationMetric], -) -> Result<()> { - validate_metrics(metrics)?; - if metrics.iter().any(|metric| { - !QUALIFICATION_RESOURCE_METRICS - .iter() - .any(|(name, unit)| metric.name() == *name && metric.unit() == *unit) - }) { - return Err(Error::Control("unknown qualification resource metric")); - } - Ok(()) -} - -pub(in crate::qualification) fn validate_resource_metric_units( - metrics: &[QualificationMetric], -) -> Result<()> { - for metric in metrics { - if let Some((_, expected_unit)) = QUALIFICATION_RESOURCE_METRICS - .iter() - .find(|(name, _)| metric.name() == *name) - && metric.unit() != *expected_unit - { - return Err(Error::Control("qualification resource metric unit")); - } - } - Ok(()) -} - -pub(in crate::qualification) fn validate_run_latency_metrics( - metrics: &[QualificationMetric], -) -> Result<()> { - let mut latency = [0_u64; 4]; - for (index, (name, unit)) in [ - ("p50_latency_ms", "ms"), - ("p95_latency_ms", "ms"), - ("p99_latency_ms", "ms"), - ("max_latency_ms", "ms"), - ] - .into_iter() - .enumerate() - { - let mut matches = metrics.iter().filter(|metric| metric.name() == name); - let Some(metric) = matches.next() else { - return Err(Error::Control("qualification run latency metrics")); - }; - if matches.next().is_some() || metric.unit() != unit { - return Err(Error::Control("qualification run latency metrics")); - } - latency[index] = metric.value(); - } - if !latency.windows(2).all(|pair| pair[0] <= pair[1]) { - return Err(Error::Control("qualification run latency percentile order")); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/qualification/receipt/runner.rs b/crates/crab-cell-runtime/src/qualification/receipt/runner.rs deleted file mode 100644 index 5774fa2ee..000000000 --- a/crates/crab-cell-runtime/src/qualification/receipt/runner.rs +++ /dev/null @@ -1,320 +0,0 @@ -//! Deterministic receipt emitter used by the qualification harnesses. - -use super::*; - -/// Deterministic receipt emitter used by local and protected qualification -/// harnesses. The harness owns workload/fault execution; this type only binds -/// its measured outputs to exact source, image and artifact bytes. -pub struct QualificationRunner { - pub(in crate::qualification) signing_key: SigningKey, -} - -impl QualificationRunner { - /// Creates a runner that signs receipts with the given attestation key. - #[must_use] - pub fn new(signing_key: SigningKey) -> Self { - Self { signing_key } - } - - /// Binds a verified typed run artifact to one signed receipt. - /// - /// The run artifact must be the first raw artifact and a canonical - /// workload artifact must be present in the remaining list. This keeps the - /// protected `primitives` row reproducible while allowing the harness to - /// retain additional raw provider/fault evidence in the same receipt. - pub fn emit_protected_run( - &self, - profile: &QualificationProfile, - source_revision: String, - image: Digest, - evidence: QualificationExecutionEvidence, - run: &QualificationRunArtifact, - artifacts: &[&[u8]], - ) -> Result { - evidence.validate()?; - if evidence.dirty { - return Err(Error::Control("qualification run source is dirty")); - } - run.verify_for_profile(profile)?; - if artifacts.is_empty() { - return Err(Error::Control("qualification run artifacts are empty")); - } - if evidence.workload != "primitives" { - return Err(Error::Control("qualification run workload row")); - } - if (!profile.required_provider().is_empty() - && evidence.provider != profile.required_provider()) - || (!profile.required_topology().is_empty() - && evidence.topology != profile.required_topology()) - { - return Err(Error::Control("qualification run environment identity")); - } - if profile.requires_protected_evidence() - && (evidence.topology == "local" - || evidence.toolchain == "unknown" - || evidence.execution_profile != "release" - || evidence.ownership.is_empty() - || evidence - .finished_at_ms - .saturating_sub(evidence.started_at_ms) - < run.elapsed_ms) - { - return Err(Error::Control( - "qualification run lacks protected execution proof", - )); - } - if profile.requires_fault_injection() - && (evidence.fault.eq_ignore_ascii_case("none") - || evidence.fault_schedule == b"none" - || evidence.ownership.len() < 2 - || !evidence.ownership.windows(2).any(|observations| { - observations[1].epoch() > observations[0].epoch() - || observations[1].published_sequence() - > observations[0].published_sequence() - }) - || evidence.ownership.windows(2).any(|observations| { - observations[1].epoch() < observations[0].epoch() - || observations[1].published_sequence() - < observations[0].published_sequence() - })) - { - return Err(Error::Control( - "qualification run lacks protected fault transition proof", - )); - } - let encoded_run = run.encode()?; - if Digest::from_bytes(*blake3::hash(artifacts[0]).as_bytes()) - != Digest::from_bytes(*blake3::hash(&encoded_run).as_bytes()) - || QualificationRunArtifact::decode(artifacts[0])? != *run - { - return Err(Error::Control("qualification run primary artifact differs")); - } - let mut workload_artifacts = artifacts.iter().filter_map(|artifact| { - QualificationWorkload::decode(artifact) - .ok() - .filter(|workload| workload.verify_for_profile(profile).is_ok()) - }); - let Some(workload) = workload_artifacts.next() else { - return Err(Error::Control("qualification run workload artifact count")); - }; - if workload_artifacts.next().is_some() || workload != *run.workload() { - return Err(Error::Control("qualification run workload identity")); - } - verify_provider_evidence(profile, run, artifacts)?; - verify_scale_evidence(profile, run, artifacts)?; - let bucket_calls = run.threshold_metric("bucket_calls", "count")?; - let peak_rss_bytes = run.threshold_metric("peak_rss_bytes", "bytes")?; - let artifact_digests = artifacts - .iter() - .map(|artifact| Digest::from_bytes(*blake3::hash(artifact).as_bytes())) - .collect::>(); - let receipt = QualificationReceipt::new( - source_revision, - image, - evidence.provider, - evidence.workload, - evidence.fault, - run.metrics().to_vec(), - artifact_digests[0], - true, - )? - .with_execution( - evidence.toolchain, - evidence.execution_profile, - evidence.topology, - run.workload().seed(), - bucket_calls, - peak_rss_bytes, - evidence.dirty, - )? - .with_profile(profile)? - .with_evidence( - evidence.started_at_ms, - evidence.finished_at_ms, - &evidence.fault_schedule, - artifact_digests, - evidence.ownership, - )? - .attest(&self.signing_key)?; - Ok(receipt) - } - - /// Builds, signs, and returns one receipt for a verified run artifact. - #[expect( - clippy::too_many_arguments, - reason = "receipt identity is intentionally explicit and fully bound" - )] - pub fn emit( - &self, - source_revision: String, - image: Digest, - provider: String, - workload: String, - fault: String, - metrics: Vec, - artifact: &[u8], - passed: bool, - execution: (String, String, String, u64, u64, u64, bool), - ) -> Result { - self.emit_with_profile( - &QualificationProfile::pr_contract(), - source_revision, - image, - provider, - workload, - fault, - metrics, - artifact, - passed, - execution, - ) - } - - /// Emits evidence while binding the exact threshold profile used by the harness. - #[expect( - clippy::too_many_arguments, - reason = "receipt identity is intentionally explicit and fully bound" - )] - pub fn emit_with_profile( - &self, - profile: &QualificationProfile, - source_revision: String, - image: Digest, - provider: String, - workload: String, - fault: String, - metrics: Vec, - artifact: &[u8], - passed: bool, - execution: (String, String, String, u64, u64, u64, bool), - ) -> Result { - let artifact_digest = Digest::from_bytes(*blake3::hash(artifact).as_bytes()); - let fault_schedule = fault.clone(); - let (toolchain, execution_profile, topology, seed, bucket_calls, peak_rss_bytes, dirty) = - execution; - QualificationReceipt::new( - source_revision, - image, - provider, - workload, - fault, - metrics, - artifact_digest, - passed, - )? - .with_execution( - toolchain, - execution_profile, - topology, - seed, - bucket_calls, - peak_rss_bytes, - dirty, - )? - .with_profile(profile)? - .with_evidence( - 1, - 1, - fault_schedule.as_bytes(), - vec![artifact_digest], - Vec::new(), - )? - .attest(&self.signing_key) - } - - /// Emits a receipt with externally captured timing, fault, artifact, and ownership evidence. - #[expect( - clippy::too_many_arguments, - reason = "qualification evidence is intentionally bound in one receipt" - )] - pub fn emit_with_evidence( - &self, - source_revision: String, - image: Digest, - provider: String, - workload: String, - fault: String, - metrics: Vec, - artifact: &[u8], - passed: bool, - execution: (String, String, String, u64, u64, u64, bool), - started_at_ms: u64, - finished_at_ms: u64, - fault_schedule: &[u8], - raw_artifact_digests: Vec, - ownership: Vec, - ) -> Result { - self.emit_with_profile_and_evidence( - &QualificationProfile::pr_contract(), - source_revision, - image, - provider, - workload, - fault, - metrics, - artifact, - passed, - execution, - started_at_ms, - finished_at_ms, - fault_schedule, - raw_artifact_digests, - ownership, - ) - } - - /// Emits evidence with an explicit profile and externally captured proof. - #[expect( - clippy::too_many_arguments, - reason = "qualification evidence is intentionally bound in one receipt" - )] - pub fn emit_with_profile_and_evidence( - &self, - profile: &QualificationProfile, - source_revision: String, - image: Digest, - provider: String, - workload: String, - fault: String, - metrics: Vec, - artifact: &[u8], - passed: bool, - execution: (String, String, String, u64, u64, u64, bool), - started_at_ms: u64, - finished_at_ms: u64, - fault_schedule: &[u8], - raw_artifact_digests: Vec, - ownership: Vec, - ) -> Result { - let (toolchain, execution_profile, topology, seed, bucket_calls, peak_rss_bytes, dirty) = - execution; - QualificationReceipt::new( - source_revision, - image, - provider, - workload, - fault, - metrics, - Digest::from_bytes(*blake3::hash(artifact).as_bytes()), - passed, - )? - .with_execution( - toolchain, - execution_profile, - topology, - seed, - bucket_calls, - peak_rss_bytes, - dirty, - )? - .with_profile(profile)? - .with_evidence( - started_at_ms, - finished_at_ms, - fault_schedule, - raw_artifact_digests, - ownership, - )? - .attest(&self.signing_key) - } -} diff --git a/crates/crab-cell-runtime/src/qualification/tests.rs b/crates/crab-cell-runtime/src/qualification/tests.rs deleted file mode 100644 index 86c317dc5..000000000 --- a/crates/crab-cell-runtime/src/qualification/tests.rs +++ /dev/null @@ -1,77 +0,0 @@ -use super::profile::{ - MAX_QUALIFICATION_CELLS, MAX_QUALIFICATION_CONCURRENCY, MAX_QUALIFICATION_DURATION_SECS, - MAX_QUALIFICATION_OPERATIONS, MAX_RECEIPT_BYTES, -}; -use super::workload::qualification_run_outcome_digest; -use super::*; - -// Capability modules keep the receipt suite navigable; the shared imports -// stay here. -mod evidence; -mod matrix; -mod profile; -mod receipt; -mod workload; - -struct ContractExecutor { - calls: u64, - case_coverage: bool, -} - -impl QualificationOperationExecutor for ContractExecutor { - type Future<'a> = std::future::Ready>; - - fn execute<'a>(&'a mut self, operation: QualificationOperation) -> Self::Future<'a> { - self.calls = self.calls.saturating_add(1); - let execution = if operation.rejection_hint() { - QualificationExecution::rejected() - } else if operation.ambiguous_hint() { - QualificationExecution::ambiguous(u64::from(operation.retry_hint())) - } else { - QualificationExecution::acknowledged(true) - .with_retries(u64::from(operation.retry_hint())) - }; - let execution = if self.case_coverage { - execution.with_case(operation.case()) - } else { - execution - }; - std::future::ready(Ok(execution)) - } -} - -#[derive(Clone)] -struct ConcurrentExecutor { - active: std::sync::Arc, - maximum: std::sync::Arc, - fail_at: Option, -} - -impl QualificationOperationExecutor for ConcurrentExecutor { - type Future<'a> = - std::pin::Pin> + Send + 'a>>; - - fn execute<'a>(&'a mut self, operation: QualificationOperation) -> Self::Future<'a> { - let active = std::sync::Arc::clone(&self.active); - let maximum = std::sync::Arc::clone(&self.maximum); - let fail_at = self.fail_at; - Box::pin(async move { - let current = active.fetch_add(1, std::sync::atomic::Ordering::AcqRel) + 1; - maximum.fetch_max(current, std::sync::atomic::Ordering::AcqRel); - tokio::task::yield_now().await; - active.fetch_sub(1, std::sync::atomic::Ordering::AcqRel); - if Some(operation.index()) == fail_at { - return Err(Error::Control("qualification executor failed")); - } - let execution = if operation.rejection_hint() { - QualificationExecution::rejected() - } else if operation.ambiguous_hint() { - QualificationExecution::ambiguous(u64::from(operation.retry_hint())) - } else { - QualificationExecution::acknowledged(true) - .with_retries(u64::from(operation.retry_hint())) - }; - Ok(execution) - }) - } -} diff --git a/crates/crab-cell-runtime/src/qualification/tests/evidence.rs b/crates/crab-cell-runtime/src/qualification/tests/evidence.rs deleted file mode 100644 index dec2831a9..000000000 --- a/crates/crab-cell-runtime/src/qualification/tests/evidence.rs +++ /dev/null @@ -1,160 +0,0 @@ -//! Execution evidence and ownership watermarks. - -use super::*; - -#[test] -fn execution_evidence_round_trip_is_canonical_and_bounded() { - let evidence = QualificationExecutionEvidence { - provider: "s3".into(), - workload: "primitives".into(), - fault: "owner-loss".into(), - toolchain: "rustc-1.90".into(), - execution_profile: "release".into(), - topology: "kubernetes".into(), - started_at_ms: 10, - finished_at_ms: 20, - fault_schedule: b"owner-loss-before-commit".to_vec(), - ownership: vec![QualificationOwnership::new( - 4, - 8, - Digest::from_bytes([7; 32]), - )], - dirty: false, - }; - let encoded = evidence.encode().expect("evidence encoding"); - assert_eq!( - QualificationExecutionEvidence::decode(&encoded).expect("evidence decoding"), - evidence - ); - let mut noncanonical = encoded; - noncanonical.push(b'\n'); - assert!(QualificationExecutionEvidence::decode(&noncanonical).is_err()); -} - -#[test] -fn evidence_binds_fault_artifacts_and_ownership_watermarks() { - let artifact = b"raw qualification output"; - let runner = QualificationRunner::new(SigningKey::from_bytes(&[6; 32])); - let receipt = runner - .emit_with_evidence( - "abc".into(), - Digest::from_bytes([1; 32]), - "rustfs".into(), - "failover".into(), - "lost-release-reply".into(), - Vec::new(), - artifact, - true, - ( - "rustc".into(), - "release".into(), - "three-node".into(), - 7, - 8, - 9, - false, - ), - 10, - 20, - b"seed=7;fault=lost-release-reply", - vec![Digest::from_bytes(*blake3::hash(artifact).as_bytes())], - vec![QualificationOwnership::new( - 4, - 12, - Digest::from_bytes([3; 32]), - )], - ) - .unwrap(); - let encoded = receipt.encode().unwrap(); - let decoded = QualificationReceipt::decode(&encoded).unwrap(); - assert_eq!(decoded.started_at_ms(), 10); - assert_eq!(decoded.finished_at_ms(), 20); - assert_eq!(decoded.ownership()[0].epoch(), 4); - assert_eq!(decoded.ownership()[0].published_sequence(), 12); - assert!( - decoded - .raw_artifact_digests() - .any(|digest| { digest == Digest::from_bytes(*blake3::hash(artifact).as_bytes()) }) - ); -} - -#[test] -fn evidence_requires_bounded_fault_schedule_and_nonzero_ownership_roots() { - let valid = QualificationExecutionEvidence { - provider: "rustfs".into(), - workload: "failover".into(), - fault: "owner-kill".into(), - toolchain: "rustc".into(), - execution_profile: "release".into(), - topology: "three-process".into(), - started_at_ms: 1, - finished_at_ms: 2, - fault_schedule: b"owner-kill".to_vec(), - ownership: Vec::new(), - dirty: false, - }; - assert!(valid.encode().is_ok()); - - let mut empty = valid.clone(); - empty.fault_schedule.clear(); - assert!(empty.encode().is_err()); - - let mut zero_root = valid.clone(); - zero_root.ownership = vec![QualificationOwnership::new( - 1, - 1, - Digest::from_bytes([0; 32]), - )]; - assert!(zero_root.encode().is_err()); - - let mut oversized = valid; - oversized.fault_schedule = vec![0; MAX_RECEIPT_BYTES + 1]; - assert!(oversized.encode().is_err()); - - let artifact = b"raw qualification output"; - let digest = Digest::from_bytes(*blake3::hash(artifact).as_bytes()); - let receipt = QualificationReceipt::new( - "source".into(), - Digest::from_bytes([1; 32]), - "rustfs".into(), - "failover".into(), - "owner-kill".into(), - Vec::new(), - digest, - true, - ) - .expect("receipt identity"); - assert!( - receipt - .clone() - .with_evidence(1, 2, b"", vec![digest], Vec::new()) - .is_err() - ); - assert!( - receipt - .clone() - .with_evidence( - 1, - 2, - b"owner-kill", - vec![digest], - vec![QualificationOwnership::new( - 1, - 1, - Digest::from_bytes([0; 32]), - )] - ) - .is_err() - ); - assert!( - receipt - .with_evidence( - 1, - 2, - &vec![0; MAX_RECEIPT_BYTES + 1], - vec![digest], - Vec::new() - ) - .is_err() - ); -} diff --git a/crates/crab-cell-runtime/src/qualification/tests/matrix.rs b/crates/crab-cell-runtime/src/qualification/tests/matrix.rs deleted file mode 100644 index b4101a091..000000000 --- a/crates/crab-cell-runtime/src/qualification/tests/matrix.rs +++ /dev/null @@ -1,451 +0,0 @@ -//! Matrix manifests and row recomputation. - -use super::*; - -#[test] -fn matrix_manifest_requires_each_bounded_workload_once() { - let entries = QUALIFICATION_MATRIX_ROWS - .iter() - .map(|workload| { - QualificationMatrixEntry::new( - (*workload).to_owned(), - format!("receipts/{workload}.json"), - vec![format!("artifacts/{workload}.json")], - ) - .unwrap() - }) - .collect::>(); - let manifest = QualificationMatrixManifest::new(entries).unwrap(); - assert_eq!( - QualificationMatrixManifest::decode(&manifest.encode().unwrap()).unwrap(), - manifest - ); - - let mut incomplete = manifest.entries().to_vec(); - incomplete.pop(); - assert!(QualificationMatrixManifest::new(incomplete).is_err()); - - let mut reordered = manifest.entries().to_vec(); - reordered.swap(0, 1); - assert!(QualificationMatrixManifest::new(reordered).is_err()); - - let mut duplicate = manifest.entries().to_vec(); - duplicate[0] = QualificationMatrixEntry::new( - duplicate[1].workload().to_owned(), - duplicate[0].receipt().to_owned(), - duplicate[0].artifacts().to_vec(), - ) - .unwrap(); - assert!(QualificationMatrixManifest::new(duplicate).is_err()); - - assert!( - QualificationMatrixEntry::new( - "protocol".into(), - "../receipt.json".into(), - vec!["artifact.bin".into()], - ) - .is_err() - ); - assert!( - QualificationMatrixEntry::new( - "protocol".into(), - "/tmp/receipt.json".into(), - vec!["artifact.bin".into()], - ) - .is_err() - ); -} - -#[test] -fn matrix_verifier_recomputes_every_row_artifact() { - let image = Digest::from_bytes([7; 32]); - let runner = QualificationRunner::new(SigningKey::from_bytes(&[8; 32])); - let mut receipts = Vec::new(); - let mut artifacts = Vec::new(); - for workload in QUALIFICATION_MATRIX_ROWS { - let artifact = format!("artifact-{workload}").into_bytes(); - receipts.push( - runner - .emit( - "source".into(), - image, - "local".into(), - (*workload).into(), - "none".into(), - Vec::new(), - &artifact, - true, - ( - "rustc".into(), - "test".into(), - "local".into(), - 1, - 0, - 0, - false, - ), - ) - .unwrap(), - ); - artifacts.push(artifact); - } - let artifact_views = artifacts - .iter() - .map(|artifact| vec![artifact.as_slice()]) - .collect::>(); - let evidence = QUALIFICATION_MATRIX_ROWS - .iter() - .enumerate() - .map(|(index, workload)| { - ( - *workload, - &receipts[index], - artifact_views[index].as_slice(), - ) - }) - .collect::>(); - QualificationReceipt::verify_matrix("source", image, &evidence).unwrap(); - - let mut forged_artifact = artifacts[0].clone(); - forged_artifact.push(b'!'); - let forged_views = [vec![forged_artifact.as_slice()]]; - let forged_evidence = evidence - .iter() - .enumerate() - .map(|(index, (workload, receipt, row_artifacts))| { - if index == 0 { - (*workload, *receipt, forged_views[0].as_slice()) - } else { - (*workload, *receipt, *row_artifacts) - } - }) - .collect::>(); - assert!(QualificationReceipt::verify_matrix("source", image, &forged_evidence).is_err()); -} - -#[test] -fn primitive_matrix_binds_the_receipt_to_the_workload_seed() { - let profile = QualificationProfile::pr_contract(); - let image = Digest::from_bytes([17; 32]); - let workload = QualificationWorkload::generate(&profile, 7).unwrap(); - let workload_artifact = workload.encode().unwrap(); - let runner = QualificationRunner::new(SigningKey::from_bytes(&[18; 32])); - let threshold_metrics = || { - vec![ - QualificationMetric::new("cells".into(), 1, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 1, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 1, "seconds".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 1, "ms".into()).unwrap(), - ] - }; - let mut receipts = Vec::new(); - let mut artifacts = Vec::new(); - for workload_name in QUALIFICATION_MATRIX_ROWS { - let artifact = if *workload_name == "primitives" { - workload_artifact.clone() - } else { - format!("artifact-{workload_name}").into_bytes() - }; - let seed = if *workload_name == "primitives" { - workload.seed() - } else { - 0 - }; - receipts.push( - runner - .emit_with_profile( - &profile, - "source".into(), - image, - "local".into(), - (*workload_name).into(), - "none".into(), - threshold_metrics(), - &artifact, - true, - ( - "rustc".into(), - "test".into(), - "local".into(), - seed, - 0, - 0, - false, - ), - ) - .unwrap(), - ); - artifacts.push(artifact); - } - let artifact_views = artifacts - .iter() - .map(|artifact| vec![artifact.as_slice()]) - .collect::>(); - let evidence = QUALIFICATION_MATRIX_ROWS - .iter() - .enumerate() - .map(|(index, workload_name)| { - ( - *workload_name, - &receipts[index], - artifact_views[index].as_slice(), - ) - }) - .collect::>(); - QualificationReceipt::verify_matrix_for_profile("source", image, &profile, &evidence).unwrap(); - - let below_threshold = runner - .emit_with_profile( - &profile, - "source".into(), - image, - "local".into(), - "protocol".into(), - "none".into(), - vec![ - QualificationMetric::new("cells".into(), 1, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 1, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 1, "seconds".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 5_001, "ms".into()).unwrap(), - ], - b"artifact-protocol", - true, - ( - "rustc".into(), - "test".into(), - "local".into(), - 0, - 0, - 0, - false, - ), - ) - .unwrap(); - let mut below_threshold_evidence = evidence.clone(); - let below_threshold_artifacts = [b"artifact-protocol" as &[u8]]; - below_threshold_evidence[0] = ("protocol", &below_threshold, &below_threshold_artifacts); - assert!( - QualificationReceipt::verify_matrix_for_profile( - "source", - image, - &profile, - &below_threshold_evidence, - ) - .is_err() - ); - - let bad = runner - .emit_with_profile( - &profile, - "source".into(), - image, - "local".into(), - "primitives".into(), - "none".into(), - threshold_metrics(), - &workload_artifact, - true, - ( - "rustc".into(), - "test".into(), - "local".into(), - workload.seed() + 1, - 0, - 0, - false, - ), - ) - .unwrap(); - let mut bad_evidence = evidence; - bad_evidence[7] = ("primitives", &bad, artifact_views[7].as_slice()); - assert!( - QualificationReceipt::verify_matrix_for_profile("source", image, &profile, &bad_evidence,) - .is_err() - ); -} - -#[test] -fn protected_primitive_matrix_requires_measured_run_artifact() { - let profile = QualificationProfile::pr_contract(); - let image = Digest::from_bytes([27; 32]); - let key = SigningKey::from_bytes(&[28; 32]); - let runner = QualificationRunner::new(key.clone()); - let workload = QualificationWorkload::generate_with_size(&profile, 7, 1, 8, 1).unwrap(); - let workload_artifact = workload.encode().unwrap(); - let run_artifact = QualificationRunArtifact { - schema_version: QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION, - workload: workload.clone(), - profile: profile.name.clone(), - profile_digest: *profile.digest().unwrap().as_bytes(), - seed: workload.seed(), - cells: workload.cells(), - operations: workload.operations(), - elapsed_ms: 1_000, - primitive_counts: workload.primitives.clone(), - case_coverage: vec![0; QUALIFICATION_CASE_COVERAGE_BYTES], - outcome_digest: *qualification_run_outcome_digest( - &workload, - &workload.primitives, - &[0; QUALIFICATION_CASE_COVERAGE_BYTES], - ) - .unwrap() - .as_bytes(), - metrics: vec![ - QualificationMetric::new("cells".into(), 1, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 8, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 1, "seconds".into()).unwrap(), - QualificationMetric::new("throughput_ops_per_sec".into(), 8, "ops/s".into()).unwrap(), - QualificationMetric::new("p50_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p95_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("max_latency_ms".into(), 1, "ms".into()).unwrap(), - ], - } - .encode() - .unwrap(); - let threshold_metrics = || { - vec![ - QualificationMetric::new("cells".into(), 1, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 8, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 1, "seconds".into()).unwrap(), - QualificationMetric::new("throughput_ops_per_sec".into(), 8, "ops/s".into()).unwrap(), - QualificationMetric::new("p50_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p95_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("max_latency_ms".into(), 1, "ms".into()).unwrap(), - ] - }; - let mut receipts = Vec::new(); - let mut artifacts = Vec::new(); - for workload_name in QUALIFICATION_MATRIX_ROWS { - let (primary, row_artifacts, seed) = if *workload_name == "primitives" { - ( - workload_artifact.clone(), - vec![workload_artifact.clone(), run_artifact.clone()], - workload.seed(), - ) - } else { - let artifact = format!("artifact-{workload_name}").into_bytes(); - (artifact.clone(), vec![artifact], 0) - }; - let raw_digests = row_artifacts - .iter() - .map(|artifact| Digest::from_bytes(*blake3::hash(artifact).as_bytes())) - .collect(); - receipts.push( - runner - .emit_with_profile_and_evidence( - &profile, - "source".into(), - image, - "local".into(), - (*workload_name).into(), - "none".into(), - threshold_metrics(), - &primary, - true, - ( - "rustc".into(), - "test".into(), - "local".into(), - seed, - 0, - 0, - false, - ), - 1, - 2, - b"none", - raw_digests, - Vec::new(), - ) - .unwrap(), - ); - artifacts.push(row_artifacts); - } - let artifact_views = artifacts - .iter() - .map(|row| row.iter().map(Vec::as_slice).collect::>()) - .collect::>(); - let evidence = QUALIFICATION_MATRIX_ROWS - .iter() - .enumerate() - .map(|(index, workload_name)| { - ( - *workload_name, - &receipts[index], - artifact_views[index].as_slice(), - ) - }) - .collect::>(); - QualificationReceipt::verify_matrix_for_profile_with_signer( - "source", - image, - &profile, - &evidence, - key.verifying_key().to_bytes(), - ) - .unwrap(); - - let mut mismatched_receipt = receipts[7].clone(); - mismatched_receipt - .metrics - .iter_mut() - .find(|metric| metric.name() == "p95_latency_ms") - .unwrap() - .value = 2; - let mismatched_receipt = mismatched_receipt.attest(&key).unwrap(); - let mut mismatched_evidence = evidence.clone(); - mismatched_evidence[7] = ( - "primitives", - &mismatched_receipt, - artifact_views[7].as_slice(), - ); - assert!( - QualificationReceipt::verify_matrix_for_profile_with_signer( - "source", - image, - &profile, - &mismatched_evidence, - key.verifying_key().to_bytes(), - ) - .is_err() - ); - - let missing_artifacts = artifacts - .iter() - .enumerate() - .map(|(index, row)| { - if index == 7 { - vec![row[0].clone()] - } else { - row.clone() - } - }) - .collect::>(); - let missing_views = missing_artifacts - .iter() - .map(|row| row.iter().map(Vec::as_slice).collect::>()) - .collect::>(); - let missing_run = QUALIFICATION_MATRIX_ROWS - .iter() - .enumerate() - .map(|(index, workload_name)| { - ( - *workload_name, - &receipts[index], - missing_views[index].as_slice(), - ) - }) - .collect::>(); - assert!( - QualificationReceipt::verify_matrix_for_profile_with_signer( - "source", - image, - &profile, - &missing_run, - key.verifying_key().to_bytes(), - ) - .is_err() - ); -} diff --git a/crates/crab-cell-runtime/src/qualification/tests/profile.rs b/crates/crab-cell-runtime/src/qualification/tests/profile.rs deleted file mode 100644 index 9f404a81c..000000000 --- a/crates/crab-cell-runtime/src/qualification/tests/profile.rs +++ /dev/null @@ -1,463 +0,0 @@ -//! Threshold, coverage, and environment profiles. - -use super::*; - -#[test] -fn threshold_profiles_are_canonical_and_distinct() { - let contract = QualificationProfile::pr_contract(); - let local = QualificationProfile::local_provider(); - let scale = QualificationProfile::scale(); - let fault = QualificationProfile::fault(); - let provider = QualificationProfile::provider(); - let compatibility = QualificationProfile::compatibility(); - assert_eq!( - contract.schema_version, - QUALIFICATION_PROFILE_SCHEMA_VERSION - ); - assert!(contract.minimum_operations() < local.minimum_operations()); - assert!(local.minimum_cells() < scale.minimum_cells()); - assert!(!contract.requires_lifecycle_case_coverage()); - assert!(local.requires_lifecycle_case_coverage()); - assert!(scale.requires_lifecycle_case_coverage()); - assert!(fault.requires_lifecycle_case_coverage()); - assert!(provider.requires_lifecycle_case_coverage()); - assert!(compatibility.requires_lifecycle_case_coverage()); - assert_eq!(fault.name(), "fault-v1"); - assert_eq!(provider.name(), "provider-v1"); - assert_eq!(compatibility.name(), "compatibility-v1"); - for (profile, name, provider, topology) in [ - ( - QualificationProfile::provider_s3(), - "provider-s3-v1", - "s3", - "three-process", - ), - ( - QualificationProfile::provider_gcs(), - "provider-gcs-v1", - "gcs", - "three-process", - ), - ( - QualificationProfile::provider_azure(), - "provider-azure-v1", - "azure", - "three-process", - ), - ( - QualificationProfile::fault_s3(), - "fault-s3-v1", - "s3", - "kubernetes", - ), - ( - QualificationProfile::fault_gcs(), - "fault-gcs-v1", - "gcs", - "kubernetes", - ), - ( - QualificationProfile::fault_azure(), - "fault-azure-v1", - "azure", - "kubernetes", - ), - ] { - assert_eq!(profile.name(), name); - assert_eq!(profile.required_provider(), provider); - assert_eq!(profile.required_topology(), topology); - assert!(profile.requires_protected_evidence()); - } - assert_ne!(contract.digest().unwrap(), local.digest().unwrap()); - assert_eq!( - QualificationProfile::decode(&contract.encode().unwrap()).unwrap(), - contract - ); -} - -#[test] -fn threshold_profile_constructor_rejects_unbounded_workloads() { - assert!( - QualificationProfile::new( - "too-many-cells".into(), - MAX_QUALIFICATION_CELLS + 1, - 1, - 1, - 1, - ) - .is_err() - ); - assert!( - QualificationProfile::new( - "too-many-operations".into(), - 1, - MAX_QUALIFICATION_OPERATIONS + 1, - 1, - 1, - ) - .is_err() - ); - assert!( - QualificationProfile::new( - "too-long".into(), - 1, - 1, - MAX_QUALIFICATION_DURATION_SECS + 1, - 1, - ) - .is_err() - ); -} - -#[test] -fn tracked_profile_newline_does_not_change_its_canonical_digest() { - let profile = QualificationProfile::scale(); - let mut encoded = profile.encode().unwrap(); - encoded.extend_from_slice(b"\n"); - assert_eq!(QualificationProfile::decode(&encoded).unwrap(), profile); - encoded.extend_from_slice(b" "); - assert_eq!(QualificationProfile::decode(&encoded).unwrap(), profile); -} - -#[test] -fn qualification_schedule_reaches_each_outcome_hint() { - let profile = QualificationProfile::new("hint-run".into(), 1, 8, 1, 1_000).unwrap(); - let workload = QualificationWorkload::generate_with_size(&profile, 41, 1, 2_048, 1).unwrap(); - let operations = workload.iter_operations().collect::>(); - assert!(operations.iter().any(|operation| operation.retry_hint())); - assert!( - operations - .iter() - .any(|operation| operation.rejection_hint()) - ); - assert!( - operations - .iter() - .any(|operation| operation.ambiguous_hint()) - ); - assert!( - operations - .iter() - .filter(|operation| operation.ambiguous_hint()) - .all(|operation| !operation.rejection_hint()) - ); -} - -#[test] -fn qualification_schedule_covers_each_lifecycle_case_per_primitive() { - let profile = QualificationProfile::new("case-run".into(), 1, 1, 1, 1_000).unwrap(); - let workload = QualificationWorkload::generate(&profile, 41).unwrap(); - assert_eq!( - workload.operations(), - QUALIFICATION_CASE_COVERAGE_OPERATIONS - ); - for primitive in QUALIFICATION_PRIMITIVES { - for case in QUALIFICATION_CASES { - assert!( - workload - .iter_operations() - .any(|operation| operation.primitive() == *primitive - && operation.case() == *case), - "missing {} case for {primitive}", - case.name() - ); - } - } -} - -#[test] -fn protected_profiles_require_measured_nonlocal_environment() { - let profile = QualificationProfile::scale(); - assert!(profile.requires_protected_evidence()); - let renamed_profile = - QualificationProfile::new("pr-contract-v1".into(), 2, 1, 1, 5_000).unwrap(); - assert!(renamed_profile.requires_protected_evidence()); - let image = Digest::from_bytes([31; 32]); - let key = SigningKey::from_bytes(&[32; 32]); - let artifact = b"scale-evidence"; - let artifact_digest = Digest::from_bytes(*blake3::hash(artifact).as_bytes()); - let metrics = vec![ - QualificationMetric::new("cells".into(), 10_000, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 10_000_000, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 3_600, "seconds".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 500, "ms".into()).unwrap(), - QualificationMetric::new("peak_local_disk_bytes".into(), 1, "bytes".into()).unwrap(), - QualificationMetric::new("peak_file_descriptors".into(), 1, "count".into()).unwrap(), - ]; - let local = QualificationRunner::new(key.clone()) - .emit_with_profile_and_evidence( - &profile, - "source".into(), - image, - "rustfs".into(), - "storage".into(), - "none".into(), - metrics.clone(), - artifact, - true, - ( - "rustc".into(), - "release".into(), - "local".into(), - 7, - 1, - 1, - false, - ), - 1, - 3_600_001, - b"none", - vec![artifact_digest], - Vec::new(), - ) - .unwrap(); - assert!( - local - .verify_for_profile_with_signer( - "source", - image, - &profile, - &[artifact], - key.verifying_key().to_bytes(), - ) - .is_err() - ); - - let protected = QualificationRunner::new(key.clone()) - .emit_with_profile_and_evidence( - &profile, - "source".into(), - image, - "rustfs".into(), - "storage".into(), - "none".into(), - metrics, - artifact, - true, - ( - "rustc".into(), - "release".into(), - "dedicated-hosts".into(), - 7, - 1, - 1, - false, - ), - 1, - 3_600_001, - b"none", - vec![artifact_digest], - vec![QualificationOwnership::new( - 2, - 3, - Digest::from_bytes([33; 32]), - )], - ) - .unwrap(); - protected - .verify_for_profile_with_signer( - "source", - image, - &profile, - &[artifact], - key.verifying_key().to_bytes(), - ) - .unwrap(); - - let debug_execution = protected - .clone() - .with_execution( - "rustc".into(), - "debug".into(), - "dedicated-hosts".into(), - 7, - 1, - 1, - false, - ) - .unwrap() - .attest(&key) - .unwrap(); - assert!( - debug_execution - .verify_for_profile_with_signer( - "source", - image, - &profile, - &[artifact], - key.verifying_key().to_bytes(), - ) - .is_err(), - "protected evidence must come from a release execution profile" - ); - - for metric_name in ["peak_local_disk_bytes", "peak_file_descriptors"] { - let mut missing_measurement = protected.clone(); - missing_measurement - .metrics - .iter_mut() - .find(|metric| metric.name() == metric_name) - .expect("resource metric") - .value = 0; - assert!( - missing_measurement - .verify_profile_thresholds(&profile) - .is_err(), - "zero {metric_name} must not stand in for a protected measurement" - ); - } -} - -#[test] -fn fault_profiles_require_injected_schedule_and_monotonic_ownership() { - let profile = QualificationProfile::fault_s3(); - assert!(profile.requires_fault_injection()); - let image = Digest::from_bytes([34; 32]); - let artifact = b"fault-evidence"; - let artifact_digest = Digest::from_bytes(*blake3::hash(artifact).as_bytes()); - let metrics = vec![ - QualificationMetric::new("cells".into(), 256, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 1_000_000, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 60, "seconds".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("peak_local_disk_bytes".into(), 1, "bytes".into()).unwrap(), - QualificationMetric::new("peak_file_descriptors".into(), 1, "count".into()).unwrap(), - ]; - let key = SigningKey::from_bytes(&[35; 32]); - let runner = QualificationRunner::new(key.clone()); - let incomplete = runner - .emit_with_profile_and_evidence( - &profile, - "fault-source".into(), - image, - "s3".into(), - "failover".into(), - "none".into(), - metrics.clone(), - artifact, - true, - ( - "rustc".into(), - "release".into(), - "kubernetes".into(), - 7, - 1, - 1, - false, - ), - 1, - 60_001, - b"none", - vec![artifact_digest], - vec![QualificationOwnership::new( - 1, - 1, - Digest::from_bytes([36; 32]), - )], - ) - .unwrap(); - assert!( - incomplete - .verify_for_profile_with_signer( - "fault-source", - image, - &profile, - &[artifact], - key.verifying_key().to_bytes(), - ) - .is_err() - ); - - let complete = runner - .emit_with_profile_and_evidence( - &profile, - "fault-source".into(), - image, - "s3".into(), - "failover".into(), - "owner-kill".into(), - metrics, - artifact, - true, - ( - "rustc".into(), - "release".into(), - "kubernetes".into(), - 7, - 1, - 1, - false, - ), - 1, - 60_001, - b"fault=owner-kill;phase=after-publication", - vec![artifact_digest], - vec![ - QualificationOwnership::new(1, 1, Digest::from_bytes([36; 32])), - QualificationOwnership::new(2, 2, Digest::from_bytes([37; 32])), - ], - ) - .unwrap(); - complete - .verify_for_profile_with_signer( - "fault-source", - image, - &profile, - &[artifact], - key.verifying_key().to_bytes(), - ) - .unwrap(); - - let no_transition = runner - .emit_with_profile_and_evidence( - &profile, - "fault-source".into(), - image, - "s3".into(), - "failover".into(), - "owner-kill".into(), - vec![ - QualificationMetric::new("cells".into(), 256, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 1_000_000, "operations".into()) - .unwrap(), - QualificationMetric::new("duration_secs".into(), 60, "seconds".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("peak_local_disk_bytes".into(), 1, "bytes".into()) - .unwrap(), - QualificationMetric::new("peak_file_descriptors".into(), 1, "count".into()) - .unwrap(), - ], - artifact, - true, - ( - "rustc".into(), - "release".into(), - "kubernetes".into(), - 7, - 1, - 1, - false, - ), - 1, - 60_001, - b"fault=owner-kill;phase=after-publication", - vec![artifact_digest], - vec![ - QualificationOwnership::new(1, 1, Digest::from_bytes([36; 32])), - QualificationOwnership::new(1, 1, Digest::from_bytes([36; 32])), - ], - ) - .unwrap(); - assert!( - no_transition - .verify_for_profile_with_signer( - "fault-source", - image, - &profile, - &[artifact], - key.verifying_key().to_bytes(), - ) - .is_err(), - "fault evidence must contain an actual ownership transition" - ); -} diff --git a/crates/crab-cell-runtime/src/qualification/tests/receipt.rs b/crates/crab-cell-runtime/src/qualification/tests/receipt.rs deleted file mode 100644 index 9ef1c2947..000000000 --- a/crates/crab-cell-runtime/src/qualification/tests/receipt.rs +++ /dev/null @@ -1,7 +0,0 @@ -//! Receipts, runners, and protected verification. - -use super::*; - -mod artifacts; -mod receipts; -mod verification; diff --git a/crates/crab-cell-runtime/src/qualification/tests/receipt/artifacts.rs b/crates/crab-cell-runtime/src/qualification/tests/receipt/artifacts.rs deleted file mode 100644 index 321804f08..000000000 --- a/crates/crab-cell-runtime/src/qualification/tests/receipt/artifacts.rs +++ /dev/null @@ -1,735 +0,0 @@ -//! Run-artifact binding, workload identity, and lifecycle coverage. - -use super::*; - -#[test] -fn measured_run_artifact_binds_the_canonical_workload_and_thresholds() { - let profile = QualificationProfile::pr_contract(); - let workload = QualificationWorkload::generate_with_size(&profile, 19, 1, 64, 1).unwrap(); - let artifact = QualificationRunArtifact { - schema_version: QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION, - workload: workload.clone(), - profile: profile.name.clone(), - profile_digest: *profile.digest().unwrap().as_bytes(), - seed: workload.seed(), - cells: workload.cells(), - operations: workload.operations(), - elapsed_ms: 1_000, - primitive_counts: workload.primitives.clone(), - case_coverage: vec![0; QUALIFICATION_CASE_COVERAGE_BYTES], - outcome_digest: *qualification_run_outcome_digest( - &workload, - &workload.primitives, - &[0; QUALIFICATION_CASE_COVERAGE_BYTES], - ) - .unwrap() - .as_bytes(), - metrics: vec![ - QualificationMetric::new("cells".into(), 1, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 64, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 1, "seconds".into()).unwrap(), - QualificationMetric::new("throughput_ops_per_sec".into(), 64, "ops/s".into()).unwrap(), - QualificationMetric::new("p50_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p95_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("max_latency_ms".into(), 1, "ms".into()).unwrap(), - ], - }; - let encoded = artifact.encode().unwrap(); - assert_eq!( - QualificationRunArtifact::decode(&encoded).unwrap(), - artifact - ); - artifact.verify_for_profile(&profile).unwrap(); - - let mut missing_latency = artifact.clone(); - missing_latency - .metrics - .retain(|metric| metric.name() != "p95_latency_ms"); - assert!(missing_latency.encode().is_err()); - - let mut unordered_latency = artifact.clone(); - unordered_latency - .metrics - .iter_mut() - .find(|metric| metric.name() == "p50_latency_ms") - .unwrap() - .value = 2; - assert!(unordered_latency.encode().is_err()); - - let mut all_acknowledged = artifact.clone(); - for counts in &mut all_acknowledged.primitive_counts { - counts.acknowledged = counts.attempted; - counts.rejected = 0; - counts.ambiguous = 0; - counts.retried = 0; - counts.verified = counts.attempted; - } - all_acknowledged.outcome_digest = *qualification_run_outcome_digest( - &all_acknowledged.workload, - &all_acknowledged.primitive_counts, - &all_acknowledged.case_coverage, - ) - .unwrap() - .as_bytes(); - all_acknowledged.verify_for_profile(&profile).unwrap(); - - let mut throughput_profile = - QualificationProfile::new("throughput-run".into(), 1, 8, 1, 1_000).unwrap(); - throughput_profile.minimum_throughput_ops_per_sec = 8; - let throughput_workload = - QualificationWorkload::generate_with_size(&throughput_profile, 19, 1, 8, 1).unwrap(); - let mut slow = QualificationRunArtifact { - schema_version: QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION, - workload: throughput_workload.clone(), - profile: throughput_profile.name.clone(), - profile_digest: *throughput_profile.digest().unwrap().as_bytes(), - seed: throughput_workload.seed(), - cells: throughput_workload.cells(), - operations: throughput_workload.operations(), - elapsed_ms: 2_000, - primitive_counts: throughput_workload.primitives.clone(), - case_coverage: vec![0; QUALIFICATION_CASE_COVERAGE_BYTES], - outcome_digest: *qualification_run_outcome_digest( - &throughput_workload, - &throughput_workload.primitives, - &[0; QUALIFICATION_CASE_COVERAGE_BYTES], - ) - .unwrap() - .as_bytes(), - metrics: vec![ - QualificationMetric::new("cells".into(), 1, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 8, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 2, "seconds".into()).unwrap(), - QualificationMetric::new("throughput_ops_per_sec".into(), 4, "ops/s".into()).unwrap(), - QualificationMetric::new("p50_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p95_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("max_latency_ms".into(), 1, "ms".into()).unwrap(), - ], - }; - assert!(slow.verify_for_profile(&throughput_profile).is_err()); - slow.elapsed_ms = 1_000; - slow.metrics[2].value = 1; - slow.metrics[3].value = 8; - slow.verify_for_profile(&throughput_profile).unwrap(); - - let mut forged = artifact.clone(); - forged.seed = forged.seed.saturating_add(1); - assert!(forged.verify_for_profile(&profile).is_err()); - - let mut forged_digest = artifact.clone(); - forged_digest.outcome_digest[0] ^= 1; - assert!(forged_digest.encode().is_err()); - - let mut unverified = artifact.clone(); - unverified.primitive_counts[0].verified = 0; - assert!(unverified.encode().is_err()); - - let mut redistributed = artifact.clone(); - let donor = redistributed - .primitive_counts - .iter() - .position(|counts| counts.acknowledged >= 2) - .unwrap(); - let recipient = redistributed - .primitive_counts - .iter() - .position(|counts| counts.primitive != redistributed.primitive_counts[donor].primitive) - .unwrap(); - redistributed.primitive_counts[donor].attempted -= 1; - redistributed.primitive_counts[donor].acknowledged -= 1; - redistributed.primitive_counts[donor].verified -= 1; - redistributed.primitive_counts[recipient].attempted += 1; - redistributed.primitive_counts[recipient].acknowledged += 1; - redistributed.primitive_counts[recipient].verified += 1; - assert!(redistributed.encode().is_err()); -} -#[tokio::test] -async fn protected_run_artifact_binds_receipt_resource_measurements() { - let mut profile = - QualificationProfile::new("protected-resource-contract".into(), 1, 8, 1, 5_000).unwrap(); - profile.maximum_peak_rss_bytes = 1_000; - profile.maximum_local_disk_bytes = 2_000; - profile.maximum_file_descriptors = 3_000; - profile.maximum_bucket_calls = 4_000; - let workload = QualificationWorkload::generate_with_size(&profile, 23, 1, 8, 1).unwrap(); - let mut executor = ContractExecutor { - calls: 0, - case_coverage: false, - }; - let summary = workload.run(&mut executor).await.unwrap(); - assert!( - summary - .artifact(&workload) - .unwrap() - .verify_for_profile(&profile) - .is_err() - ); - - let resources = [ - QualificationMetric::new("peak_rss_bytes".into(), 19, "bytes".into()).unwrap(), - QualificationMetric::new("peak_local_disk_bytes".into(), 29, "bytes".into()).unwrap(), - QualificationMetric::new("peak_file_descriptors".into(), 39, "count".into()).unwrap(), - QualificationMetric::new("bucket_calls".into(), 49, "count".into()).unwrap(), - ]; - let mut run = summary - .artifact_with_resource_metrics(&workload, &resources) - .unwrap(); - run.elapsed_ms = 1_000; - for (name, value) in [("duration_secs", 1), ("throughput_ops_per_sec", 8)] { - run.metrics - .iter_mut() - .find(|metric| metric.name() == name) - .unwrap() - .value = value; - } - run.verify_for_profile(&profile).unwrap(); - let encoded = run.encode().unwrap(); - - let mut receipt_metrics = summary.metrics().unwrap(); - receipt_metrics.extend(resources.iter().cloned()); - let key = SigningKey::from_bytes(&[93; 32]); - let receipt = QualificationRunner::new(key) - .emit_with_profile_and_evidence( - &profile, - "resource-source".into(), - Digest::from_bytes([94; 32]), - "provider".into(), - "primitives".into(), - "none".into(), - receipt_metrics, - &encoded, - true, - ( - "rustc".into(), - "release".into(), - "three-process".into(), - workload.seed(), - 49, - 19, - false, - ), - 1, - 1_001, - b"none", - vec![Digest::from_bytes(*blake3::hash(&encoded).as_bytes())], - vec![QualificationOwnership::new( - 1, - 1, - Digest::from_bytes([95; 32]), - )], - ) - .unwrap(); - receipt.verify_profile_thresholds(&profile).unwrap(); - receipt - .verify_primitive_run_artifact(&profile, &[&encoded]) - .unwrap(); - - let mut forged = receipt; - forged.bucket_calls += 1; - assert!( - forged - .verify_primitive_run_artifact(&profile, &[&encoded]) - .is_err() - ); -} -#[tokio::test] -async fn protected_run_binder_requires_one_canonical_run_and_workload() { - let mut profile = - QualificationProfile::new("protected-binder-contract".into(), 1, 8, 1, 5_000).unwrap(); - profile.maximum_peak_rss_bytes = 1_000; - profile.maximum_local_disk_bytes = 2_000; - profile.maximum_file_descriptors = 3_000; - profile.maximum_bucket_calls = 4_000; - profile.provider = "rustfs".into(); - let workload = QualificationWorkload::generate_with_size(&profile, 29, 1, 8, 1).unwrap(); - let mut executor = ContractExecutor { - calls: 0, - case_coverage: false, - }; - let summary = workload.run(&mut executor).await.unwrap(); - let resources = [ - QualificationMetric::new("peak_rss_bytes".into(), 19, "bytes".into()).unwrap(), - QualificationMetric::new("peak_local_disk_bytes".into(), 29, "bytes".into()).unwrap(), - QualificationMetric::new("peak_file_descriptors".into(), 39, "count".into()).unwrap(), - QualificationMetric::new("bucket_calls".into(), 49, "count".into()).unwrap(), - ]; - let mut run = summary - .artifact_with_resource_metrics(&workload, &resources) - .unwrap(); - run.elapsed_ms = 1_000; - for (name, value) in [("duration_secs", 1), ("throughput_ops_per_sec", 8)] { - run.metrics - .iter_mut() - .find(|metric| metric.name() == name) - .unwrap() - .value = value; - } - run.verify_for_profile(&profile).unwrap(); - let run_bytes = run.encode().unwrap(); - let workload_bytes = workload.encode().unwrap(); - let provider_evidence = - QualificationProviderEvidence::new(&profile, workload.seed(), true, true, true) - .unwrap() - .encode() - .unwrap(); - let key = SigningKey::from_bytes(&[96; 32]); - let runner = QualificationRunner::new(key.clone()); - let evidence = QualificationExecutionEvidence { - provider: "rustfs".into(), - workload: "primitives".into(), - fault: "none".into(), - toolchain: "rustc".into(), - execution_profile: "release".into(), - topology: "three-process".into(), - started_at_ms: 1, - finished_at_ms: 1_001, - fault_schedule: b"none".to_vec(), - ownership: vec![QualificationOwnership::new( - 1, - 1, - Digest::from_bytes([98; 32]), - )], - dirty: false, - }; - assert!( - runner - .emit_protected_run( - &profile, - "binder-source".into(), - Digest::from_bytes([97; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes], - ) - .is_err() - ); - let partial_provider_evidence = - QualificationProviderEvidence::new(&profile, workload.seed(), true, true, false) - .unwrap() - .encode() - .unwrap(); - assert!( - runner - .emit_protected_run( - &profile, - "binder-source".into(), - Digest::from_bytes([97; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes, &partial_provider_evidence], - ) - .is_err() - ); - let malformed_provider_evidence = br#"{"schema_version":1,"provider":"rustfs","profile":"protected-binder-contract","conditional":true}"#; - assert!( - runner - .emit_protected_run( - &profile, - "binder-source".into(), - Digest::from_bytes([97; 32]), - evidence.clone(), - &run, - &[ - &run_bytes, - &workload_bytes, - &provider_evidence, - malformed_provider_evidence - ], - ) - .is_err() - ); - assert!( - runner - .emit_protected_run( - &profile, - "binder-source".into(), - Digest::from_bytes([97; 32]), - evidence.clone(), - &run, - &[ - &run_bytes, - &workload_bytes, - &provider_evidence, - &provider_evidence, - ], - ) - .is_err() - ); - let receipt = runner - .emit_protected_run( - &profile, - "binder-source".into(), - Digest::from_bytes([97; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes, &provider_evidence], - ) - .unwrap(); - receipt - .verify_for_profile_with_signer( - "binder-source", - Digest::from_bytes([97; 32]), - &profile, - &[&run_bytes, &workload_bytes, &provider_evidence], - key.verifying_key().to_bytes(), - ) - .unwrap(); - receipt - .verify_primitive_workload(&profile, &[&run_bytes, &workload_bytes, &provider_evidence]) - .unwrap(); - receipt - .verify_primitive_run_artifact(&profile, &[&run_bytes, &workload_bytes, &provider_evidence]) - .unwrap(); - let mut short_evidence = evidence.clone(); - short_evidence.finished_at_ms = 2; - assert!( - runner - .emit_protected_run( - &profile, - "binder-source".into(), - Digest::from_bytes([97; 32]), - short_evidence, - &run, - &[&run_bytes, &workload_bytes, &provider_evidence], - ) - .is_err() - ); - let mut dirty_evidence = evidence.clone(); - dirty_evidence.dirty = true; - assert!( - runner - .emit_protected_run( - &profile, - "binder-source".into(), - Digest::from_bytes([97; 32]), - dirty_evidence, - &run, - &[&run_bytes, &workload_bytes, &provider_evidence], - ) - .is_err() - ); - - let mismatched_workload = - QualificationWorkload::generate_with_size(&profile, 29, 2, 8, 1).unwrap(); - let mismatched_workload_bytes = mismatched_workload.encode().unwrap(); - assert!( - runner - .emit_protected_run( - &profile, - "binder-source".into(), - Digest::from_bytes([97; 32]), - evidence, - &run, - &[&run_bytes, &mismatched_workload_bytes, &provider_evidence], - ) - .is_err() - ); -} -#[test] -fn protected_run_artifacts_require_complete_lifecycle_case_coverage() { - let mut profile = - QualificationProfile::new("protected-case-coverage".into(), 1, 8, 1, 1_000).unwrap(); - profile.provider = "rustfs".into(); - profile.topology = "three-process".into(); - let workload = QualificationWorkload::generate_with_size(&profile, 19, 1, 56, 1).unwrap(); - let artifact = QualificationRunArtifact { - schema_version: QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION, - workload: workload.clone(), - profile: profile.name.clone(), - profile_digest: *profile.digest().unwrap().as_bytes(), - seed: workload.seed(), - cells: workload.cells(), - operations: workload.operations(), - elapsed_ms: 1_000, - primitive_counts: workload.primitives.clone(), - case_coverage: vec![0; QUALIFICATION_CASE_COVERAGE_BYTES], - outcome_digest: *qualification_run_outcome_digest( - &workload, - &workload.primitives, - &[0; QUALIFICATION_CASE_COVERAGE_BYTES], - ) - .unwrap() - .as_bytes(), - metrics: vec![ - QualificationMetric::new("cells".into(), 1, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 56, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 1, "seconds".into()).unwrap(), - QualificationMetric::new("throughput_ops_per_sec".into(), 56, "ops/s".into()).unwrap(), - QualificationMetric::new("p50_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p95_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("max_latency_ms".into(), 1, "ms".into()).unwrap(), - ], - }; - artifact.encode().unwrap(); - assert!(artifact.verify_for_profile(&profile).is_err()); - - let mut partial_verification = artifact; - partial_verification.case_coverage.fill(u8::MAX); - partial_verification.primitive_counts[0].verified -= 1; - partial_verification.outcome_digest = *qualification_run_outcome_digest( - &partial_verification.workload, - &partial_verification.primitive_counts, - &partial_verification.case_coverage, - ) - .unwrap() - .as_bytes(); - partial_verification.encode().unwrap(); - assert!(partial_verification.verify_for_profile(&profile).is_err()); -} - -#[tokio::test] -async fn scale_receipt_requires_observed_open_cells_and_binds_their_resource_peaks() { - use crate::qualification::receipt::evidence::ScaleEvidence; - - let mut profile = QualificationProfile::new("scale-v1".into(), 10_000, 8, 1, 5_000).unwrap(); - profile.maximum_peak_rss_bytes = 5_000_000_000; - profile.maximum_local_disk_bytes = 100_000_000; - profile.maximum_file_descriptors = 100_000; - profile.maximum_bucket_calls = 10; - let workload = QualificationWorkload::generate_with_size(&profile, 73, 10_000, 8, 1).unwrap(); - let summary = workload - .run(&mut ContractExecutor { - calls: 0, - case_coverage: false, - }) - .await - .unwrap(); - let resources = [ - QualificationMetric::new("peak_rss_bytes".into(), 3_000_000_000, "bytes".into()).unwrap(), - QualificationMetric::new("peak_local_disk_bytes".into(), 50_000_000, "bytes".into()) - .unwrap(), - QualificationMetric::new("peak_file_descriptors".into(), 90_000, "count".into()).unwrap(), - QualificationMetric::new("bucket_calls".into(), 1, "count".into()).unwrap(), - ]; - let mut run = summary - .artifact_with_resource_metrics(&workload, &resources) - .unwrap(); - run.elapsed_ms = 1_000; - for (name, value) in [("duration_secs", 1), ("throughput_ops_per_sec", 8)] { - run.metrics - .iter_mut() - .find(|metric| metric.name() == name) - .unwrap() - .value = value; - } - run.verify_for_profile(&profile).unwrap(); - let run_bytes = run.encode().unwrap(); - let workload_bytes = workload.encode().unwrap(); - let samples = [ - "empty", - "sparse", - "resident", - "pending-publication", - "churned", - ] - .into_iter() - .flat_map(|state| { - [1_000, 5_000, 10_000].into_iter().map(move |count| { - let after = scale_snapshot(count); - serde_json::json!({ - "state": state, - "target_cells": count, - "before": scale_snapshot(0), - "after": after, - "peak": after, - }) - }) - }) - .collect::>(); - let mut artifact = serde_json::json!({ - "schema_version": 1, - "profile": profile.name(), - "profile_digest": profile.digest().unwrap().as_bytes(), - "workload_seed": workload.seed(), - "cell_samples": samples, - }); - // Fixed process setup can exceed the 1,000-Cell marginal charge. - artifact["cell_samples"][0]["after"]["rss_bytes"] = serde_json::json!(400_000_000); - artifact["cell_samples"][0]["peak"]["rss_bytes"] = serde_json::json!(400_000_000); - let canonical = |value| { - serde_json::to_vec(&serde_json::from_value::(value).unwrap()).unwrap() - }; - let scale_bytes = canonical(artifact.clone()); - let key = SigningKey::from_bytes(&[74; 32]); - let runner = QualificationRunner::new(key.clone()); - let evidence = QualificationExecutionEvidence { - provider: "rustfs".into(), - workload: "primitives".into(), - fault: "none".into(), - toolchain: "rustc".into(), - execution_profile: "release".into(), - topology: "dedicated-hosts".into(), - started_at_ms: 1, - finished_at_ms: 1_001, - fault_schedule: b"none".to_vec(), - ownership: vec![QualificationOwnership::new( - 1, - 1, - Digest::from_bytes([75; 32]), - )], - dirty: false, - }; - assert!( - runner - .emit_protected_run( - &profile, - "scale-source".into(), - Digest::from_bytes([76; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes], - ) - .is_err() - ); - let mut false_open = artifact; - false_open["cell_samples"][0]["after"]["active_cells"] = serde_json::json!(999); - let false_open_bytes = canonical(false_open); - assert!( - runner - .emit_protected_run( - &profile, - "scale-source".into(), - Digest::from_bytes([76; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes, &false_open_bytes], - ) - .is_err() - ); - let mut undercharged_descriptors = - serde_json::from_slice::(&scale_bytes).unwrap(); - undercharged_descriptors["cell_samples"][1]["after"]["file_descriptors"] = - serde_json::json!(60_001); - undercharged_descriptors["cell_samples"][1]["peak"]["file_descriptors"] = - serde_json::json!(60_001); - let undercharged_descriptors_bytes = canonical(undercharged_descriptors); - assert!( - runner - .emit_protected_run( - &profile, - "scale-source".into(), - Digest::from_bytes([76; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes, &undercharged_descriptors_bytes], - ) - .is_err() - ); - let mut undercharged_memory = - serde_json::from_slice::(&scale_bytes).unwrap(); - undercharged_memory["cell_samples"][1]["after"]["rss_bytes"] = serde_json::json!(2_000_000_000); - undercharged_memory["cell_samples"][1]["peak"]["rss_bytes"] = serde_json::json!(2_000_000_000); - let undercharged_memory_bytes = canonical(undercharged_memory); - assert!( - runner - .emit_protected_run( - &profile, - "scale-source".into(), - Digest::from_bytes([76; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes, &undercharged_memory_bytes], - ) - .is_err() - ); - let mut undercharged_cache = serde_json::from_slice::(&scale_bytes).unwrap(); - undercharged_cache["cell_samples"][1]["after"]["sqlite_cache_bytes"] = - serde_json::json!(1_000_000_000); - undercharged_cache["cell_samples"][1]["peak"]["sqlite_cache_bytes"] = - serde_json::json!(1_000_000_000); - let undercharged_cache_bytes = canonical(undercharged_cache); - assert!( - runner - .emit_protected_run( - &profile, - "scale-source".into(), - Digest::from_bytes([76; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes, &undercharged_cache_bytes], - ) - .is_err() - ); - let mut underreported_peak = serde_json::from_slice::(&scale_bytes).unwrap(); - underreported_peak["cell_samples"][14]["peak"]["rss_bytes"] = - serde_json::json!(3_000_000_001_u64); - let underreported_peak_bytes = canonical(underreported_peak); - assert!( - runner - .emit_protected_run( - &profile, - "scale-source".into(), - Digest::from_bytes([76; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes, &underreported_peak_bytes], - ) - .is_err() - ); - assert!( - runner - .emit_protected_run( - &profile, - "scale-source".into(), - Digest::from_bytes([76; 32]), - evidence.clone(), - &run, - &[&run_bytes, &workload_bytes, &scale_bytes, &scale_bytes], - ) - .is_err() - ); - let receipt = runner - .emit_protected_run( - &profile, - "scale-source".into(), - Digest::from_bytes([76; 32]), - evidence, - &run, - &[&run_bytes, &workload_bytes, &scale_bytes], - ) - .unwrap(); - receipt - .verify_for_profile_with_signer( - "scale-source", - Digest::from_bytes([76; 32]), - &profile, - &[&run_bytes, &workload_bytes, &scale_bytes], - key.verifying_key().to_bytes(), - ) - .unwrap(); - assert!( - receipt - .verify_primitive_run_artifact( - &profile, - &[&run_bytes, &workload_bytes, &false_open_bytes], - ) - .is_err() - ); - assert!( - receipt - .verify_primitive_run_artifact( - &profile, - &[&run_bytes, &workload_bytes, &undercharged_memory_bytes], - ) - .is_err() - ); -} - -fn scale_snapshot(cells: u64) -> serde_json::Value { - serde_json::json!({ - "active_cells": cells, - "rss_bytes": 100 + cells * 100, - "allocator_bytes": 100 + cells * 50, - "threads": 1, - "file_descriptors": 1 + cells * 8, - "sqlite_cache_bytes": cells * 1_024, - "admitted_resident_bytes": cells * 65_536, - "admitted_file_descriptors": cells * 8, - "retained_bytes": 0, - "local_disk_reserved_bytes": cells * 4_096, - "local_disk_bytes": cells * 4_096, - }) -} diff --git a/crates/crab-cell-runtime/src/qualification/tests/receipt/receipts.rs b/crates/crab-cell-runtime/src/qualification/tests/receipt/receipts.rs deleted file mode 100644 index 1664c4a2e..000000000 --- a/crates/crab-cell-runtime/src/qualification/tests/receipt/receipts.rs +++ /dev/null @@ -1,146 +0,0 @@ -//! Receipt encoding, measurement identities, and byte bounds. - -use super::*; - -#[test] -fn receipt_round_trips_and_binds_measurements() { - let receipt = QualificationReceipt::new( - "abc".into(), - Digest::from_bytes([1; 32]), - "memory".into(), - "warm-route".into(), - "none".into(), - vec![QualificationMetric::new("p99".into(), 7, "ms".into()).unwrap()], - Digest::from_bytes([2; 32]), - true, - ) - .unwrap() - .with_execution( - "rustc".into(), - "unit".into(), - "local".into(), - 7, - 0, - 1024, - false, - ) - .unwrap() - .with_profile(&QualificationProfile::pr_contract()) - .unwrap() - .attest(&SigningKey::from_bytes(&[9; 32])) - .unwrap(); - assert_eq!(receipt.schema_version(), QUALIFICATION_SCHEMA_VERSION); - let decoded = QualificationReceipt::decode(&receipt.encode().unwrap()).unwrap(); - assert_eq!(decoded, receipt); -} -#[test] -fn qualification_metrics_reject_duplicate_identities() { - let duplicate = vec![ - QualificationMetric::new("p99".into(), 7, "ms".into()).unwrap(), - QualificationMetric::new("p99".into(), 8, "ms".into()).unwrap(), - ]; - assert!( - QualificationReceipt::new( - "abc".into(), - Digest::from_bytes([1; 32]), - "memory".into(), - "warm-route".into(), - "none".into(), - duplicate, - Digest::from_bytes([2; 32]), - true, - ) - .is_err() - ); - - let profile = QualificationProfile::pr_contract(); - let workload = QualificationWorkload::generate(&profile, 19).unwrap(); - let artifact = QualificationRunArtifact { - schema_version: QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION, - workload: workload.clone(), - profile: profile.name.clone(), - profile_digest: *profile.digest().unwrap().as_bytes(), - seed: workload.seed(), - cells: workload.cells(), - operations: workload.operations(), - elapsed_ms: 1_000, - primitive_counts: workload.primitives.clone(), - case_coverage: vec![0; QUALIFICATION_CASE_COVERAGE_BYTES], - outcome_digest: *qualification_run_outcome_digest( - &workload, - &workload.primitives, - &[0; QUALIFICATION_CASE_COVERAGE_BYTES], - ) - .unwrap() - .as_bytes(), - metrics: vec![ - QualificationMetric::new("p99_latency_ms".into(), 1, "ms".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 2, "ms".into()).unwrap(), - ], - }; - assert!(artifact.encode().is_err()); -} -#[test] -fn multi_artifact_receipts_require_all_raw_bytes() { - let primary = b"primary"; - let secondary = b"secondary"; - let runner = QualificationRunner::new(SigningKey::from_bytes(&[10; 32])); - let receipt = runner - .emit_with_evidence( - "source".into(), - Digest::from_bytes([11; 32]), - "local".into(), - "storage".into(), - "none".into(), - Vec::new(), - primary, - true, - ( - "rustc".into(), - "test".into(), - "local".into(), - 0, - 0, - 0, - false, - ), - 1, - 2, - b"none", - vec![ - Digest::from_bytes(*blake3::hash(primary).as_bytes()), - Digest::from_bytes(*blake3::hash(secondary).as_bytes()), - ], - Vec::new(), - ) - .unwrap(); - receipt - .verify_for_artifacts( - "source", - Digest::from_bytes([11; 32]), - &[primary, secondary], - ) - .unwrap(); - assert!( - receipt - .verify_for("source", Digest::from_bytes([11; 32]), primary) - .is_err() - ); -} -#[test] -fn receipt_rejects_unbounded_or_non_ascii_identity() { - assert!( - QualificationReceipt::new( - "\n".into(), - Digest::from_bytes([1; 32]), - "provider".into(), - "workload".into(), - "fault".into(), - Vec::new(), - Digest::from_bytes([2; 32]), - false, - ) - .is_err() - ); - assert!(QualificationReceipt::decode(&vec![b' '; MAX_RECEIPT_BYTES + 1]).is_err()); -} diff --git a/crates/crab-cell-runtime/src/qualification/tests/receipt/verification.rs b/crates/crab-cell-runtime/src/qualification/tests/receipt/verification.rs deleted file mode 100644 index 01e72d0a5..000000000 --- a/crates/crab-cell-runtime/src/qualification/tests/receipt/verification.rs +++ /dev/null @@ -1,171 +0,0 @@ -//! Runner and protected verification over signed receipts. - -use super::*; - -#[test] -fn runner_binds_artifact_and_rejects_forged_or_dirty_receipts() { - let runner = QualificationRunner::new(SigningKey::from_bytes(&[4; 32])); - let receipt = runner - .emit( - "abc".into(), - Digest::from_bytes([1; 32]), - "rustfs".into(), - "warm".into(), - "none".into(), - Vec::new(), - b"artifact", - true, - ( - "rustc".into(), - "ci".into(), - "three-node".into(), - 1, - 2, - 3, - false, - ), - ) - .unwrap(); - assert_eq!( - receipt.artifact_digest(), - Digest::from_bytes(*blake3::hash(b"artifact").as_bytes()) - ); - let encoded = receipt.encode().unwrap(); - assert_eq!(QualificationReceipt::decode(&encoded).unwrap(), receipt); - assert_eq!( - receipt.profile_digest(), - QualificationProfile::pr_contract().digest().unwrap() - ); - receipt - .verify_for("abc", Digest::from_bytes([1; 32]), b"artifact") - .unwrap(); - let failed = QualificationReceipt::new( - "abc".into(), - Digest::from_bytes([1; 32]), - "rustfs".into(), - "warm".into(), - "none".into(), - Vec::new(), - Digest::from_bytes(*blake3::hash(b"artifact").as_bytes()), - false, - ) - .unwrap() - .with_execution( - "rustc".into(), - "ci".into(), - "three-node".into(), - 1, - 2, - 3, - false, - ) - .unwrap() - .with_profile(&QualificationProfile::pr_contract()) - .unwrap() - .attest(&SigningKey::from_bytes(&[4; 32])) - .unwrap(); - assert!( - failed - .verify_for("abc", Digest::from_bytes([1; 32]), b"artifact") - .is_err() - ); - assert!( - receipt - .verify_for("other", Digest::from_bytes([1; 32]), b"artifact") - .is_err() - ); - let mut forged = receipt.clone(); - forged.bucket_calls = forged.bucket_calls.saturating_add(1); - assert!(QualificationReceipt::decode(&forged.encode().unwrap()).is_err()); - assert!( - receipt - .verify_for("abc", Digest::from_bytes([1; 32]), b"other") - .is_err() - ); - let dirty = runner - .emit( - "abc".into(), - Digest::from_bytes([1; 32]), - "rustfs".into(), - "warm".into(), - "none".into(), - Vec::new(), - b"artifact", - true, - ( - "rustc".into(), - "ci".into(), - "three-node".into(), - 1, - 2, - 3, - true, - ), - ) - .unwrap(); - assert!(dirty.encode().is_err()); -} -#[test] -fn protected_verification_pins_signer_and_threshold_metrics() { - let profile = QualificationProfile::pr_contract(); - let key = SigningKey::from_bytes(&[12; 32]); - let metrics = vec![ - QualificationMetric::new("cells".into(), 1, "cells".into()).unwrap(), - QualificationMetric::new("operations".into(), 1, "operations".into()).unwrap(), - QualificationMetric::new("duration_secs".into(), 1, "seconds".into()).unwrap(), - QualificationMetric::new("p99_latency_ms".into(), 5_000, "ms".into()).unwrap(), - ]; - let receipt = QualificationRunner::new(key.clone()) - .emit_with_profile_and_evidence( - &profile, - "source".into(), - Digest::from_bytes([13; 32]), - "provider".into(), - "primitives".into(), - "none".into(), - metrics, - b"artifact", - true, - ( - "rustc".into(), - "release".into(), - "local".into(), - 7, - 0, - 0, - false, - ), - 1, - 1, - b"none", - vec![Digest::from_bytes(*blake3::hash(b"artifact").as_bytes())], - Vec::new(), - ) - .unwrap(); - receipt - .verify_for_profile_with_signer( - "source", - Digest::from_bytes([13; 32]), - &profile, - &[b"artifact"], - key.verifying_key().to_bytes(), - ) - .unwrap(); - assert!( - receipt - .verify_for_profile_with_signer( - "source", - Digest::from_bytes([13; 32]), - &profile, - &[b"artifact"], - SigningKey::from_bytes(&[14; 32]).verifying_key().to_bytes(), - ) - .is_err() - ); - let mut below_threshold = receipt.clone(); - below_threshold.metrics[1].value = 0; - assert!(below_threshold.verify_profile_thresholds(&profile).is_err()); - below_threshold.metrics[1].value = 1; - below_threshold.metrics[2].value = 0; - assert!(below_threshold.verify_profile_thresholds(&profile).is_err()); -} diff --git a/crates/crab-cell-runtime/src/qualification/tests/workload.rs b/crates/crab-cell-runtime/src/qualification/tests/workload.rs deleted file mode 100644 index 5d17548d4..000000000 --- a/crates/crab-cell-runtime/src/qualification/tests/workload.rs +++ /dev/null @@ -1,173 +0,0 @@ -//! Workload generation, streaming, and bounded execution. - -use super::*; - -#[test] -fn mixed_workload_is_seed_deterministic_and_covers_every_primitive() { - let profile = QualificationProfile::pr_contract(); - let first = QualificationWorkload::generate(&profile, 7).unwrap(); - let second = QualificationWorkload::generate(&profile, 7).unwrap(); - let changed = QualificationWorkload::generate(&profile, 8).unwrap(); - assert_eq!(first, second); - assert_ne!(first.outcome_digest(), changed.outcome_digest()); - assert_eq!(first.primitives().len(), QUALIFICATION_PRIMITIVES.len()); - assert!( - first - .primitives() - .iter() - .all(|counts| counts.attempted() > 0 && counts.verified() > 0) - ); - for seed in 0..32 { - let workload = QualificationWorkload::generate(&profile, seed).unwrap(); - assert!( - workload - .primitives() - .iter() - .all(|counts| counts.attempted() > 0) - ); - } - first.verify_for_profile(&profile).unwrap(); - let encoded = first.encode().unwrap(); - assert_eq!(QualificationWorkload::decode(&encoded).unwrap(), first); - - let mut forged = first.clone(); - forged.outcome_digest[0] ^= 1; - assert!(forged.encode().is_err()); -} - -#[test] -fn workload_bounds_and_counter_overflow_fail_closed() { - let profile = QualificationProfile::pr_contract(); - let mut workload = QualificationWorkload::generate(&profile, 7).unwrap(); - workload.operations = MAX_QUALIFICATION_OPERATIONS + 1; - assert!(workload.encode().is_err()); - - workload.operations = 8; - workload.primitives[0].attempted = u64::MAX; - workload.primitives[0].acknowledged = u64::MAX; - workload.primitives[0].rejected = u64::MAX; - assert!(workload.encode().is_err()); -} - -#[tokio::test] -async fn workload_iterator_and_executor_are_streaming_and_reproducible() { - let profile = QualificationProfile::new("contract-run".into(), 1, 32, 1, 1_000).unwrap(); - let workload = QualificationWorkload::generate_with_size(&profile, 41, 3, 32, 1).unwrap(); - let operations = workload.iter_operations().collect::>(); - assert_eq!(operations.len(), 32); - for (index, operation) in operations.iter().copied().enumerate() { - assert_eq!(workload.operation_at(index as u64).unwrap(), operation); - } - let mut executor = ContractExecutor { - calls: 0, - case_coverage: false, - }; - let summary = workload.run(&mut executor).await.unwrap(); - assert_eq!(executor.calls, workload.operations()); - assert_eq!(summary.operations(), workload.operations()); - assert_eq!(summary.cells(), workload.cells()); - assert!( - summary - .primitive_counts() - .iter() - .all(|counts| counts.attempted() > 0 && counts.verified() > 0) - ); - assert!( - summary - .metrics() - .unwrap() - .iter() - .any(|metric| { metric.name() == "p99_latency_ms" && metric.unit() == "ms" }) - ); - assert_ne!(summary.outcome_digest(), workload.outcome_digest()); -} - -#[tokio::test] -async fn measured_artifact_verifies_across_fractional_second_boundaries() { - let profile = QualificationProfile::pr_contract(); - let workload = QualificationWorkload::generate_with_size(&profile, 41, 1, 64, 1).unwrap(); - let mut executor = ContractExecutor { - calls: 0, - case_coverage: false, - }; - let mut summary = workload.run(&mut executor).await.unwrap(); - for nanos in [ - 999_999_999, - 1_000_000_000, - 1_000_000_001, - 1_000_999_999, - 1_001_000_000, - 7_000_000_000, - 7_000_000_001, - 7_000_999_999, - 7_001_000_000, - ] { - summary.elapsed = std::time::Duration::from_nanos(nanos); - let artifact = summary.artifact(&workload).unwrap(); - let decoded = QualificationRunArtifact::decode(&artifact.encode().unwrap()).unwrap(); - assert_eq!( - decoded.verify_for_profile(&profile).is_ok(), - nanos >= 1_000_000_000, - "elapsed {nanos} ns must preserve the profile's minimum duration" - ); - } -} - -#[tokio::test] -async fn case_coverage_runner_rejects_unverified_lifecycle_hints() { - let profile = QualificationProfile::pr_contract(); - let workload = QualificationWorkload::generate(&profile, 41).unwrap(); - let mut missing = ContractExecutor { - calls: 0, - case_coverage: false, - }; - assert!(workload.run_with_case_coverage(&mut missing).await.is_err()); - - let mut covered = ContractExecutor { - calls: 0, - case_coverage: true, - }; - let summary = workload.run_with_case_coverage(&mut covered).await.unwrap(); - assert!(summary.case_coverage().iter().all(|byte| *byte == u8::MAX)); - summary.artifact(&workload).unwrap(); -} - -#[tokio::test] -async fn concurrent_workload_runner_bounds_inflight_operations() { - let profile = QualificationProfile::new("concurrent-run".into(), 1, 32, 1, 1_000).unwrap(); - let workload = QualificationWorkload::generate_with_size(&profile, 41, 2, 32, 1).unwrap(); - let active = std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)); - let maximum = std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)); - let executor = ConcurrentExecutor { - active: std::sync::Arc::clone(&active), - maximum: std::sync::Arc::clone(&maximum), - fail_at: None, - }; - assert!(workload.run_concurrent(executor.clone(), 0).await.is_err()); - assert!( - workload - .run_concurrent(executor.clone(), MAX_QUALIFICATION_CONCURRENCY + 1) - .await - .is_err() - ); - let summary = workload.run_concurrent(executor, 4).await.unwrap(); - assert_eq!(summary.operations(), workload.operations()); - assert!(maximum.load(std::sync::atomic::Ordering::Acquire) >= 2); - assert!(maximum.load(std::sync::atomic::Ordering::Acquire) <= 4); - summary.artifact(&workload).unwrap().encode().unwrap(); -} - -#[tokio::test] -async fn concurrent_workload_runner_drains_inflight_operations_after_failure() { - let profile = QualificationProfile::new("concurrent-failure".into(), 1, 32, 1, 1_000).unwrap(); - let workload = QualificationWorkload::generate_with_size(&profile, 41, 2, 32, 1).unwrap(); - let active = std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)); - let maximum = std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)); - let executor = ConcurrentExecutor { - active: std::sync::Arc::clone(&active), - maximum, - fail_at: Some(0), - }; - assert!(workload.run_concurrent(executor, 4).await.is_err()); - assert_eq!(active.load(std::sync::atomic::Ordering::Acquire), 0); -} diff --git a/crates/crab-cell-runtime/src/qualification/workload.rs b/crates/crab-cell-runtime/src/qualification/workload.rs deleted file mode 100644 index 287334fa1..000000000 --- a/crates/crab-cell-runtime/src/qualification/workload.rs +++ /dev/null @@ -1,489 +0,0 @@ -//! Deterministic qualification workloads and execution accounting. - -use super::profile::{ - MAX_QUALIFICATION_CELLS, MAX_QUALIFICATION_CONCURRENCY, MAX_QUALIFICATION_DURATION_SECS, - MAX_QUALIFICATION_OPERATIONS, MAX_RECEIPT_BYTES, mark_case_coverage, validate_label, -}; -use super::*; - -mod operation; -mod summary; - -use operation::{LCG_SEED_XOR, lcg_state_at, operation_from_state}; -pub use operation::{ - QualificationExecution, QualificationOperation, QualificationOperationIter, - QualificationOutcome, QualificationPrimitiveCounts, -}; -pub(super) use operation::{qualification_workload_outcome_digest, valid_primitive_counts}; -pub(super) use summary::qualification_run_outcome_digest; -pub use summary::{QualificationLatencyHistogram, QualificationRunSummary}; - -/// Async application boundary used by [`QualificationWorkload::run`]. -pub trait QualificationOperationExecutor { - /// Future returned for one operation. - type Future<'a>: Future> + Send + 'a - where - Self: 'a; - - /// Runs one workload operation. - fn execute<'a>(&'a mut self, operation: QualificationOperation) -> Self::Future<'a>; -} - -/// Deterministic logical workload artifact for PR, provider, and scale tiers. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationWorkload { - pub(super) schema_version: u32, - pub(super) profile: String, - pub(super) profile_digest: [u8; 32], - pub(super) seed: u64, - pub(super) cells: u64, - pub(super) operations: u64, - pub(super) duration_secs: u64, - pub(super) primitives: Vec, - pub(super) outcome_digest: [u8; 32], -} - -impl QualificationWorkload { - /// Generates the canonical operation schedule and logical outcome digest. - pub fn generate(profile: &QualificationProfile, seed: u64) -> Result { - profile.validate()?; - Self::generate_with_size( - profile, - seed, - profile.minimum_cells(), - profile - .minimum_operations() - .max(QUALIFICATION_CASE_COVERAGE_OPERATIONS), - profile.minimum_duration_secs(), - ) - } - - /// Generates a bounded custom-size schedule for fast contract tests. - pub fn generate_with_size( - profile: &QualificationProfile, - seed: u64, - cells: u64, - operations: u64, - duration_secs: u64, - ) -> Result { - profile.validate()?; - let minimum_operations = profile - .minimum_operations() - .max(QUALIFICATION_PRIMITIVES.len() as u64); - if cells == 0 - || cells > MAX_QUALIFICATION_CELLS - || operations == 0 - || operations > MAX_QUALIFICATION_OPERATIONS - || duration_secs == 0 - || duration_secs > MAX_QUALIFICATION_DURATION_SECS - || cells < profile.minimum_cells() - || operations < minimum_operations - || duration_secs < profile.minimum_duration_secs() - { - return Err(Error::Control( - "qualification workload is below profile minimum", - )); - } - let mut primitives = QUALIFICATION_PRIMITIVES - .iter() - .map(|primitive| QualificationPrimitiveCounts::new(primitive)) - .collect::>(); - for operation in QualificationOperationIter::new(seed, cells, operations) { - let primitive_index = operation.primitive_index as usize; - let counts = &mut primitives[primitive_index]; - counts.attempted += 1; - if operation.rejection_hint { - counts.rejected += 1; - } else if operation.ambiguous_hint { - counts.ambiguous += 1; - if operation.retry_hint { - counts.retried += 1; - } - } else { - counts.acknowledged += 1; - if operation.retry_hint { - counts.retried += 1; - } - counts.verified += 1; - } - } - let outcome_digest = qualification_workload_outcome_digest(seed, cells, operations); - Ok(Self { - schema_version: 1, - profile: profile.name.clone(), - profile_digest: *profile.digest()?.as_bytes(), - seed, - cells, - operations, - duration_secs, - primitives, - outcome_digest: *outcome_digest.as_bytes(), - }) - } - - /// Decodes and validates a canonical workload artifact. - pub fn decode(bytes: &[u8]) -> Result { - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification workload exceeds limit")); - } - let workload: Self = serde_json::from_slice(bytes)?; - workload.validate()?; - if serde_json::to_vec(&workload).map_err(Error::from)? != bytes { - return Err(Error::Control("qualification workload is not canonical")); - } - Ok(workload) - } - - /// Encodes the workload artifact with stable field ordering. - pub fn encode(&self) -> Result> { - self.validate()?; - let bytes = serde_json::to_vec(self).map_err(Error::from)?; - if bytes.len() > MAX_RECEIPT_BYTES { - return Err(Error::Control("qualification workload exceeds limit")); - } - Ok(bytes) - } - - /// Returns the threshold profile the workload was built for. - #[must_use] - pub fn profile(&self) -> &str { - &self.profile - } - - /// Returns the digest of that threshold profile. - #[must_use] - pub const fn profile_digest(&self) -> Digest { - Digest::from_bytes(self.profile_digest) - } - - /// Returns the seed the operation schedule is derived from. - #[must_use] - pub const fn seed(&self) -> u64 { - self.seed - } - - /// Returns how many Cells the workload addresses. - #[must_use] - pub const fn cells(&self) -> u64 { - self.cells - } - - /// Returns how many operations the schedule contains. - #[must_use] - pub const fn operations(&self) -> u64 { - self.operations - } - - /// Returns the target duration in seconds. - #[must_use] - pub const fn duration_secs(&self) -> u64 { - self.duration_secs - } - - /// Returns the per-primitive operation counts. - #[must_use] - pub fn primitives(&self) -> &[QualificationPrimitiveCounts] { - &self.primitives - } - - /// Returns a streaming iterator over the canonical operation schedule. - #[must_use] - pub fn iter_operations(&self) -> QualificationOperationIter { - QualificationOperationIter::new(self.seed, self.cells, self.operations) - } - - /// Returns one operation by index without materializing the schedule. - pub fn operation_at(&self, index: u64) -> Result { - if index >= self.operations { - return Err(Error::Control("qualification operation index")); - } - let state = lcg_state_at(self.seed ^ LCG_SEED_XOR, index + 1); - Ok(operation_from_state(index, state, self.cells)) - } - - /// Returns the digest of the measured outcome. - #[must_use] - pub const fn outcome_digest(&self) -> Digest { - Digest::from_bytes(self.outcome_digest) - } - - /// Executes every scheduled operation through a typed application adapter. - /// - /// The executor owns request construction, retries, and post-commit - /// verification. This method only retains bounded counters and a fixed - /// latency histogram, making it suitable for million-operation runs. - pub async fn run(&self, executor: &mut E) -> Result - where - E: QualificationOperationExecutor, - { - self.run_internal(executor, false).await - } - - /// Executes the schedule and fails unless every executor result identifies - /// the lifecycle case assigned to that operation. - pub async fn run_with_case_coverage( - &self, - executor: &mut E, - ) -> Result - where - E: QualificationOperationExecutor, - { - self.run_internal(executor, true).await - } - - pub(super) async fn run_internal( - &self, - executor: &mut E, - require_case: bool, - ) -> Result - where - E: QualificationOperationExecutor, - { - let started = Instant::now(); - let mut counts = QUALIFICATION_PRIMITIVES - .iter() - .map(|primitive| QualificationPrimitiveCounts::new(primitive)) - .collect::>(); - let mut case_coverage = vec![0; QUALIFICATION_CASE_COVERAGE_BYTES]; - let mut latency = QualificationLatencyHistogram::default(); - for operation in self.iter_operations() { - let operation_started = Instant::now(); - let execution = executor.execute(operation).await?; - if require_case && execution.case() != Some(operation.case()) { - return Err(Error::Control( - "qualification executor did not report lifecycle case", - )); - } - Self::record_execution( - &mut counts, - &mut case_coverage, - &mut latency, - operation, - execution, - operation_started.elapsed(), - ); - } - self.summary(counts, case_coverage, latency, started.elapsed()) - } - - /// Executes the schedule with bounded in-flight operations through cloned - /// typed executors. The executor must make operations independent or - /// idempotent when it opts into concurrency; aggregate counters and the - /// logical outcome digest remain schedule-order independent. - pub async fn run_concurrent( - &self, - executor: E, - concurrency: usize, - ) -> Result - where - E: QualificationOperationExecutor + Clone + Send + 'static, - { - self.run_concurrent_internal(executor, concurrency, false) - .await - } - - /// Executes a concurrent schedule and requires case identity from every - /// completed operation before emitting a measured result. - pub async fn run_concurrent_with_case_coverage( - &self, - executor: E, - concurrency: usize, - ) -> Result - where - E: QualificationOperationExecutor + Clone + Send + 'static, - { - self.run_concurrent_internal(executor, concurrency, true) - .await - } - - pub(super) async fn run_concurrent_internal( - &self, - executor: E, - concurrency: usize, - require_case: bool, - ) -> Result - where - E: QualificationOperationExecutor + Clone + Send + 'static, - { - if concurrency == 0 || concurrency > MAX_QUALIFICATION_CONCURRENCY { - return Err(Error::Control("qualification concurrency is out of bounds")); - } - let started = Instant::now(); - let mut counts = QUALIFICATION_PRIMITIVES - .iter() - .map(|primitive| QualificationPrimitiveCounts::new(primitive)) - .collect::>(); - let mut case_coverage = vec![0; QUALIFICATION_CASE_COVERAGE_BYTES]; - let mut latency = QualificationLatencyHistogram::default(); - let mut operations = self.iter_operations(); - let mut pending = FuturesUnordered::new(); - let mut first_error = None; - - loop { - while first_error.is_none() && pending.len() < concurrency { - let Some(operation) = operations.next() else { - break; - }; - let mut worker = executor.clone(); - pending.push(async move { - let operation_started = Instant::now(); - let execution = worker.execute(operation).await?; - Ok::<_, Error>((operation, execution, operation_started.elapsed())) - }); - } - let Some(result) = pending.next().await else { - break; - }; - match result { - Ok((operation, execution, elapsed)) => { - if require_case && execution.case() != Some(operation.case()) { - if first_error.is_none() { - first_error = Some(Error::Control( - "qualification executor did not report lifecycle case", - )); - } - continue; - } - Self::record_execution( - &mut counts, - &mut case_coverage, - &mut latency, - operation, - execution, - elapsed, - ); - } - Err(error) => { - if first_error.is_none() { - first_error = Some(error); - } - } - } - } - if let Some(error) = first_error { - return Err(error); - } - self.summary(counts, case_coverage, latency, started.elapsed()) - } - - pub(super) fn record_execution( - counts: &mut [QualificationPrimitiveCounts], - case_coverage: &mut [u8], - latency: &mut QualificationLatencyHistogram, - operation: QualificationOperation, - execution: QualificationExecution, - elapsed: Duration, - ) { - if execution.case() == Some(operation.case()) { - mark_case_coverage(case_coverage, operation); - } - latency.record(elapsed); - let primitive_index = usize::from(operation.primitive_index); - let primitive = &mut counts[primitive_index]; - primitive.attempted = primitive.attempted.saturating_add(1); - match execution.outcome { - QualificationOutcome::Acknowledged => { - primitive.acknowledged = primitive.acknowledged.saturating_add(1); - if execution.verified { - primitive.verified = primitive.verified.saturating_add(1); - } - } - QualificationOutcome::Rejected => { - primitive.rejected = primitive.rejected.saturating_add(1); - } - QualificationOutcome::Ambiguous => { - primitive.ambiguous = primitive.ambiguous.saturating_add(1); - } - } - if execution.retries != 0 { - primitive.retried = primitive.retried.saturating_add(1); - } - } - - pub(super) fn summary( - &self, - primitive_counts: Vec, - case_coverage: Vec, - latency: QualificationLatencyHistogram, - elapsed: Duration, - ) -> Result { - let outcome_digest = - qualification_run_outcome_digest(self, &primitive_counts, &case_coverage)?; - Ok(QualificationRunSummary { - profile: self.profile.clone(), - profile_digest: self.profile_digest(), - seed: self.seed, - cells: self.cells, - operations: self.operations, - elapsed, - primitive_counts, - case_coverage, - outcome_digest, - latency, - }) - } - - /// Recomputes the schedule and requires an exact profile/seed/size match. - pub fn verify_for_profile(&self, profile: &QualificationProfile) -> Result<()> { - if self.profile != profile.name || self.profile_digest() != profile.digest()? { - return Err(Error::Control("qualification workload profile identity")); - } - let expected = Self::generate_with_size( - profile, - self.seed, - self.cells, - self.operations, - self.duration_secs, - )?; - if expected != *self { - return Err(Error::Control("qualification workload outcome")); - } - Ok(()) - } - - pub(super) fn validate(&self) -> Result<()> { - let expected_primitives = QUALIFICATION_PRIMITIVES - .iter() - .copied() - .collect::>(); - if self.schema_version != 1 - || self.cells == 0 - || self.cells > MAX_QUALIFICATION_CELLS - || self.operations == 0 - || self.operations > MAX_QUALIFICATION_OPERATIONS - || self.duration_secs == 0 - || self.duration_secs > MAX_QUALIFICATION_DURATION_SECS - || self.profile_digest.iter().all(|byte| *byte == 0) - || self.outcome_digest.iter().all(|byte| *byte == 0) - || self.primitives.len() != QUALIFICATION_PRIMITIVES.len() - || self - .primitives - .iter() - .map(QualificationPrimitiveCounts::primitive) - .collect::>() - != expected_primitives - { - return Err(Error::Control("invalid qualification workload")); - } - validate_label(&self.profile, "qualification workload profile")?; - if self - .primitives - .iter() - .any(|counts| counts.attempted == 0 || !valid_primitive_counts(counts)) - { - return Err(Error::Control("qualification workload counters")); - } - if self - .primitives - .iter() - .try_fold(0_u64, |total, counts| total.checked_add(counts.attempted)) - != Some(self.operations) - || qualification_workload_outcome_digest(self.seed, self.cells, self.operations) - != self.outcome_digest() - { - return Err(Error::Control("qualification workload outcome")); - } - Ok(()) - } -} diff --git a/crates/crab-cell-runtime/src/qualification/workload/operation.rs b/crates/crab-cell-runtime/src/qualification/workload/operation.rs deleted file mode 100644 index 6d637a4ba..000000000 --- a/crates/crab-cell-runtime/src/qualification/workload/operation.rs +++ /dev/null @@ -1,390 +0,0 @@ -//! Operation vocabulary and the deterministic schedule that produces it. - -use serde::{Deserialize, Serialize}; - -use crate::identity::Digest; -use crate::qualification::profile::{ - QUALIFICATION_CASE_COVERAGE_OPERATIONS, QUALIFICATION_CASES, QUALIFICATION_PRIMITIVES, - QualificationCase, -}; - -/// Bounded per-primitive counters emitted by the deterministic workload driver. -#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct QualificationPrimitiveCounts { - pub(in crate::qualification) primitive: String, - pub(in crate::qualification) attempted: u64, - pub(in crate::qualification) acknowledged: u64, - pub(in crate::qualification) rejected: u64, - pub(in crate::qualification) ambiguous: u64, - pub(in crate::qualification) retried: u64, - pub(in crate::qualification) verified: u64, -} - -impl QualificationPrimitiveCounts { - pub(in crate::qualification) fn new(primitive: &str) -> Self { - Self { - primitive: primitive.to_owned(), - attempted: 0, - acknowledged: 0, - rejected: 0, - ambiguous: 0, - retried: 0, - verified: 0, - } - } - - /// Returns the primitive row name. - #[must_use] - pub fn primitive(&self) -> &str { - &self.primitive - } - - /// Operations attempted for this primitive. - #[must_use] - pub const fn attempted(&self) -> u64 { - self.attempted - } - - /// Attempts the destination acknowledged. - #[must_use] - pub const fn acknowledged(&self) -> u64 { - self.acknowledged - } - - /// Attempts the destination rejected. - #[must_use] - pub const fn rejected(&self) -> u64 { - self.rejected - } - - /// Attempts whose outcome stayed unknown. - #[must_use] - pub const fn ambiguous(&self) -> u64 { - self.ambiguous - } - - /// Attempts that needed a retry. - #[must_use] - pub const fn retried(&self) -> u64 { - self.retried - } - - /// Attempts whose recorded result was verified. - #[must_use] - pub const fn verified(&self) -> u64 { - self.verified - } -} - -/// One deterministic operation in a qualification workload. -/// -/// The iterator exposes only bounded scalar identity. Payloads and primitive -/// requests remain owned by the application-specific executor, so a scale run -/// does not retain millions of operation bodies in memory. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct QualificationOperation { - pub(in crate::qualification) index: u64, - pub(in crate::qualification) primitive_index: u8, - pub(in crate::qualification) cell_index: u64, - pub(in crate::qualification) nonce: u64, - pub(in crate::qualification) case: QualificationCase, - pub(in crate::qualification) retry_hint: bool, - pub(in crate::qualification) rejection_hint: bool, - pub(in crate::qualification) ambiguous_hint: bool, -} - -impl QualificationOperation { - /// Returns the monotonically increasing logical operation number. - #[must_use] - pub const fn index(self) -> u64 { - self.index - } - - /// Returns the primitive selected by the canonical schedule. - #[must_use] - pub fn primitive(self) -> &'static str { - QUALIFICATION_PRIMITIVES[self.primitive_index as usize] - } - - /// Returns the deterministic Cell ordinal selected by the schedule. - #[must_use] - pub const fn cell_index(self) -> u64 { - self.cell_index - } - - /// Returns the deterministic nonce for request identity/payload seeding. - #[must_use] - pub const fn nonce(self) -> u64 { - self.nonce - } - - /// Returns the bounded lifecycle case assigned to this operation. - #[must_use] - pub const fn case(self) -> QualificationCase { - self.case - } - - /// Returns whether this operation exercises producer/delivery duplicate handling. - #[must_use] - pub const fn duplicate_hint(self) -> bool { - matches!(self.case, QualificationCase::Duplicate) - } - - /// Returns whether this operation exercises expiry handling. - #[must_use] - pub const fn expiry_hint(self) -> bool { - matches!(self.case, QualificationCase::Expiry) - } - - /// Returns whether this operation exercises cancellation handling. - #[must_use] - pub const fn cancellation_hint(self) -> bool { - matches!(self.case, QualificationCase::Cancellation) - } - - /// Returns whether this operation exercises owner-loss handling. - #[must_use] - pub const fn owner_loss_hint(self) -> bool { - matches!(self.case, QualificationCase::OwnerLoss) - } - - /// Returns whether this operation exercises post-takeover recovery. - #[must_use] - pub const fn recovery_hint(self) -> bool { - matches!(self.case, QualificationCase::Recovery) - } - - /// Returns whether this operation is scheduled to exercise a retry path. - #[must_use] - pub const fn retry_hint(self) -> bool { - self.retry_hint - } - - /// Returns whether this operation is scheduled to exercise rejection. - #[must_use] - pub const fn rejection_hint(self) -> bool { - self.rejection_hint - } - - /// Returns whether this operation is scheduled to exercise ambiguity. - #[must_use] - pub const fn ambiguous_hint(self) -> bool { - self.ambiguous_hint - } -} - -/// Actual outcome reported by one application-specific operation executor. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum QualificationOutcome { - /// The destination accepted the operation. - Acknowledged, - /// The destination rejected the operation. - Rejected, - /// The outcome could not be determined. - Ambiguous, -} - -/// Bounded result returned by a typed workload executor. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct QualificationExecution { - pub(in crate::qualification) outcome: QualificationOutcome, - pub(in crate::qualification) verified: bool, - pub(in crate::qualification) retries: u64, - pub(in crate::qualification) case: Option, -} - -impl QualificationExecution { - /// Records an acknowledged operation and whether its state was verified. - #[must_use] - pub const fn acknowledged(verified: bool) -> Self { - Self { - outcome: QualificationOutcome::Acknowledged, - verified, - retries: 0, - case: None, - } - } - - /// Records a rejected operation. - #[must_use] - pub const fn rejected() -> Self { - Self { - outcome: QualificationOutcome::Rejected, - verified: false, - retries: 0, - case: None, - } - } - - /// Records an ambiguous operation with the number of resolution attempts. - #[must_use] - pub const fn ambiguous(retries: u64) -> Self { - Self { - outcome: QualificationOutcome::Ambiguous, - verified: false, - retries, - case: None, - } - } - - /// Adds a retry count to an execution result. - #[must_use] - pub const fn with_retries(mut self, retries: u64) -> Self { - self.retries = retries; - self - } - - /// Binds the executor result to the lifecycle case it actually exercised. - #[must_use] - pub const fn with_case(mut self, case: QualificationCase) -> Self { - self.case = Some(case); - self - } - - /// Returns the outcome the executor reported. - #[must_use] - pub const fn outcome(self) -> QualificationOutcome { - self.outcome - } - - /// Reports whether the result was verified against the record. - #[must_use] - pub const fn verified(self) -> bool { - self.verified - } - - /// Returns how many retries the operation needed. - #[must_use] - pub const fn retries(self) -> u64 { - self.retries - } - - /// Returns the matrix case the operation belongs to, when any. - #[must_use] - pub const fn case(self) -> Option { - self.case - } -} - -pub(in crate::qualification) fn qualification_workload_outcome_digest( - seed: u64, - cells: u64, - operations: u64, -) -> Digest { - let mut hasher = blake3::Hasher::new(); - for operation in QualificationOperationIter::new(seed, cells, operations) { - hasher.update(&operation.index.to_be_bytes()); - hasher.update(&u64::from(operation.primitive_index).to_be_bytes()); - hasher.update(&operation.cell_index.to_be_bytes()); - hasher.update(&operation.nonce.to_be_bytes()); - hasher.update(&[operation.case as u8]); - } - Digest::from_bytes(*hasher.finalize().as_bytes()) -} - -pub(in crate::qualification) fn valid_primitive_counts( - counts: &QualificationPrimitiveCounts, -) -> bool { - counts - .acknowledged - .checked_add(counts.rejected) - .and_then(|total| total.checked_add(counts.ambiguous)) - == Some(counts.attempted) - && counts.retried <= counts.attempted - && counts.verified <= counts.acknowledged -} - -pub(in crate::qualification) const LCG_MULTIPLIER: u64 = 6_364_136_223_846_793_005; -pub(in crate::qualification) const LCG_INCREMENT: u64 = 1_442_695_040_888_963_407; -pub(in crate::qualification) const LCG_SEED_XOR: u64 = 0x9e37_79b9_7f4a_7c15; - -/// Streaming operation iterator; it retains only the LCG state and counters. -pub struct QualificationOperationIter { - pub(in crate::qualification) index: u64, - pub(in crate::qualification) cells: u64, - pub(in crate::qualification) operations: u64, - pub(in crate::qualification) state: u64, -} - -impl QualificationOperationIter { - pub(in crate::qualification) fn new(seed: u64, cells: u64, operations: u64) -> Self { - Self { - index: 0, - cells, - operations, - state: seed ^ LCG_SEED_XOR, - } - } -} - -impl Iterator for QualificationOperationIter { - type Item = QualificationOperation; - - fn next(&mut self) -> Option { - if self.index >= self.operations { - return None; - } - self.state = self - .state - .wrapping_mul(LCG_MULTIPLIER) - .wrapping_add(LCG_INCREMENT); - let operation = operation_from_state(self.index, self.state, self.cells); - self.index = self.index.saturating_add(1); - Some(operation) - } -} - -pub(in crate::qualification) fn operation_from_state( - index: u64, - state: u64, - cells: u64, -) -> QualificationOperation { - let primitive_count = QUALIFICATION_PRIMITIVES.len() as u64; - let coverage_operations = QUALIFICATION_CASE_COVERAGE_OPERATIONS; - let (primitive_index, case_index) = if index < coverage_operations { - ( - (index % primitive_count) as usize, - (index / primitive_count) as usize, - ) - } else { - ( - (state as usize) % QUALIFICATION_PRIMITIVES.len(), - (state.rotate_left(19) as usize) % QUALIFICATION_CASES.len(), - ) - }; - QualificationOperation { - index, - primitive_index: primitive_index as u8, - cell_index: state % cells, - nonce: state, - case: QUALIFICATION_CASES[case_index], - retry_hint: state & 0x1f == 0, - rejection_hint: state & 0x3ff == 0, - ambiguous_hint: state & 0x7ff == 0x200, - } -} - -pub(in crate::qualification) fn lcg_state_at(seed: u64, steps: u64) -> u64 { - let mut result = (1_u64, 0_u64); - let mut base = (LCG_MULTIPLIER, LCG_INCREMENT); - let mut steps = steps; - while steps != 0 { - if steps & 1 != 0 { - result = compose_affine(base, result); - } - base = compose_affine(base, base); - steps >>= 1; - } - result.0.wrapping_mul(seed).wrapping_add(result.1) -} - -pub(in crate::qualification) fn compose_affine( - after: (u64, u64), - before: (u64, u64), -) -> (u64, u64) { - ( - after.0.wrapping_mul(before.0), - after.0.wrapping_mul(before.1).wrapping_add(after.1), - ) -} diff --git a/crates/crab-cell-runtime/src/qualification/workload/summary.rs b/crates/crab-cell-runtime/src/qualification/workload/summary.rs deleted file mode 100644 index cd9f4104f..000000000 --- a/crates/crab-cell-runtime/src/qualification/workload/summary.rs +++ /dev/null @@ -1,257 +0,0 @@ -//! Measured run summary and latency histogram. - -use std::time::Duration; - -use crate::identity::Digest; -use crate::qualification::profile::{QUALIFICATION_CASE_COVERAGE_BYTES, QUALIFICATION_PRIMITIVES}; -use crate::qualification::receipt::{ - QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION, QualificationMetric, QualificationRunArtifact, - validate_metrics, validate_resource_metric_list, -}; -use crate::{Error, Result}; - -use super::QualificationWorkload; -use super::operation::QualificationPrimitiveCounts; - -/// Bounded latency histogram used by a qualification run. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct QualificationLatencyHistogram { - pub(in crate::qualification) buckets: [u64; 64], - pub(in crate::qualification) samples: u64, -} - -impl Default for QualificationLatencyHistogram { - fn default() -> Self { - Self { - buckets: [0; 64], - samples: 0, - } - } -} - -impl QualificationLatencyHistogram { - pub(in crate::qualification) fn record(&mut self, latency: Duration) { - let micros = latency.as_micros().max(1).min(u128::from(u64::MAX)) as u64; - let bucket = (u64::BITS - micros.leading_zeros() - 1) as usize; - self.buckets[bucket.min(self.buckets.len() - 1)] = - self.buckets[bucket.min(self.buckets.len() - 1)].saturating_add(1); - self.samples = self.samples.saturating_add(1); - } - - pub(in crate::qualification) fn percentile_ms(&self, percentile: u64) -> u64 { - if self.samples == 0 { - return 0; - } - let rank = self.samples.saturating_mul(percentile).saturating_add(99) / 100; - let mut seen = 0_u64; - for (bucket, count) in self.buckets.iter().copied().enumerate() { - seen = seen.saturating_add(count); - if seen >= rank { - let micros = 1_u64 << bucket.min(63); - return micros.saturating_add(999) / 1_000; - } - } - u64::MAX / 1_000 - } -} - -/// Measured result of executing a canonical workload through typed APIs. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct QualificationRunSummary { - pub(in crate::qualification) profile: String, - pub(in crate::qualification) profile_digest: Digest, - pub(in crate::qualification) seed: u64, - pub(in crate::qualification) cells: u64, - pub(in crate::qualification) operations: u64, - pub(in crate::qualification) elapsed: Duration, - pub(in crate::qualification) primitive_counts: Vec, - pub(in crate::qualification) case_coverage: Vec, - pub(in crate::qualification) outcome_digest: Digest, - pub(in crate::qualification) latency: QualificationLatencyHistogram, -} - -impl QualificationRunSummary { - /// Returns the threshold profile the run was measured against. - #[must_use] - pub fn profile(&self) -> &str { - &self.profile - } - - /// Returns the digest of that threshold profile. - #[must_use] - pub const fn profile_digest(&self) -> Digest { - self.profile_digest - } - - /// Returns the seed the schedule was generated with. - #[must_use] - pub const fn seed(&self) -> u64 { - self.seed - } - - /// Returns how many Cells the run addressed. - #[must_use] - pub const fn cells(&self) -> u64 { - self.cells - } - - /// Returns how many operations the run executed. - #[must_use] - pub const fn operations(&self) -> u64 { - self.operations - } - - /// Returns the wall-clock duration the run took. - #[must_use] - pub const fn elapsed(&self) -> Duration { - self.elapsed - } - - /// Returns the per-primitive operation counts. - #[must_use] - pub fn primitive_counts(&self) -> &[QualificationPrimitiveCounts] { - &self.primitive_counts - } - - /// Returns one bounded bitset of observed primitive/lifecycle pairs. - #[must_use] - pub fn case_coverage(&self) -> &[u8] { - &self.case_coverage - } - - /// Returns the digest of the measured outcome. - #[must_use] - pub const fn outcome_digest(&self) -> Digest { - self.outcome_digest - } - - /// Returns the measured latency percentile rounded up to milliseconds. - #[must_use] - pub fn latency_percentile_ms(&self, percentile: u64) -> u64 { - self.latency.percentile_ms(percentile) - } - - /// Returns receipt-compatible bounded metrics from this run. - pub fn metrics(&self) -> Result> { - // Verification recomputes these metrics from serialized milliseconds. - // Rounding nanoseconds independently rejects runs just past a whole second. - let duration_secs = self.elapsed_ms().saturating_add(999) / 1_000; - let throughput = self.operations / duration_secs; - Ok(vec![ - QualificationMetric::new("cells".into(), self.cells, "cells".into())?, - QualificationMetric::new("operations".into(), self.operations, "operations".into())?, - QualificationMetric::new("duration_secs".into(), duration_secs, "seconds".into())?, - QualificationMetric::new("throughput_ops_per_sec".into(), throughput, "ops/s".into())?, - QualificationMetric::new( - "p50_latency_ms".into(), - self.latency_percentile_ms(50), - "ms".into(), - )?, - QualificationMetric::new( - "p95_latency_ms".into(), - self.latency_percentile_ms(95), - "ms".into(), - )?, - QualificationMetric::new( - "p99_latency_ms".into(), - self.latency_percentile_ms(99), - "ms".into(), - )?, - QualificationMetric::new( - "max_latency_ms".into(), - self.latency_percentile_ms(100), - "ms".into(), - )?, - ]) - } - - /// Encodes the measured result together with the exact canonical workload. - pub fn artifact(&self, workload: &QualificationWorkload) -> Result { - self.artifact_with_resource_metrics(workload, &[]) - } - - /// Encodes a measured result and the resource observations captured for the same run. - /// - /// Protected profiles require every [`crate::qualification::QUALIFICATION_RESOURCE_METRICS`] - /// observation. The receipt verifier compares these values with the signed - /// execution measurements, so a harness cannot substitute a different - /// machine's resource envelope after the workload has completed. - pub fn artifact_with_resource_metrics( - &self, - workload: &QualificationWorkload, - resource_metrics: &[QualificationMetric], - ) -> Result { - if self.profile != workload.profile - || self.profile_digest != workload.profile_digest() - || self.seed != workload.seed - || self.cells != workload.cells - || self.operations != workload.operations - { - return Err(Error::Control("qualification run workload identity")); - } - validate_resource_metric_list(resource_metrics)?; - let mut metrics = self.metrics()?; - metrics.extend(resource_metrics.iter().cloned()); - validate_metrics(&metrics)?; - Ok(QualificationRunArtifact { - schema_version: QUALIFICATION_RUN_ARTIFACT_SCHEMA_VERSION, - workload: workload.clone(), - profile: self.profile.clone(), - profile_digest: *self.profile_digest.as_bytes(), - seed: self.seed, - cells: self.cells, - operations: self.operations, - elapsed_ms: self.elapsed_ms(), - primitive_counts: self.primitive_counts.clone(), - case_coverage: self.case_coverage.clone(), - outcome_digest: *qualification_run_outcome_digest( - workload, - &self.primitive_counts, - &self.case_coverage, - )? - .as_bytes(), - metrics, - }) - } - - fn elapsed_ms(&self) -> u64 { - self.elapsed.as_millis().max(1).min(u128::from(u64::MAX)) as u64 - } -} - -pub(in crate::qualification) fn qualification_run_outcome_digest( - workload: &QualificationWorkload, - counts: &[QualificationPrimitiveCounts], - case_coverage: &[u8], -) -> Result { - if case_coverage.len() != QUALIFICATION_CASE_COVERAGE_BYTES { - return Err(Error::Control("qualification run case coverage")); - } - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab-cell-runtime/qualification/run-v3"); - hasher.update(workload.outcome_digest().as_bytes()); - hasher.update(workload.profile_digest().as_bytes()); - hasher.update(&workload.seed.to_be_bytes()); - hasher.update(&workload.cells.to_be_bytes()); - hasher.update(&workload.operations.to_be_bytes()); - hasher.update(case_coverage); - for primitive in QUALIFICATION_PRIMITIVES { - let primitive_counts = counts - .iter() - .find(|counts| counts.primitive == *primitive) - .ok_or(Error::Control("qualification run primitive identity"))?; - hasher.update(&(primitive.len() as u64).to_be_bytes()); - hasher.update(primitive.as_bytes()); - for value in [ - primitive_counts.attempted, - primitive_counts.acknowledged, - primitive_counts.rejected, - primitive_counts.ambiguous, - primitive_counts.retried, - primitive_counts.verified, - ] { - hasher.update(&value.to_be_bytes()); - } - } - Ok(Digest::from_bytes(*hasher.finalize().as_bytes())) -} diff --git a/crates/crab-cell-runtime/src/read_policy.rs b/crates/crab-cell-runtime/src/read_policy.rs deleted file mode 100644 index c7f2a628b..000000000 --- a/crates/crab-cell-runtime/src/read_policy.rs +++ /dev/null @@ -1,244 +0,0 @@ -//! S3-backed desired reader count; never an owner or replica assignment. - -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::{ETag, StorageError}; -use serde::{Deserialize, Serialize}; - -use crate::identity::{CellId, IncarnationId, decode_hex, encode_hex}; -use crate::{Error, Result}; - -const MAX_POLICY_BYTES: u64 = 512; -/// Highest operator-requested reader count accepted by the policy codec. -pub const MAX_READERS: u16 = 10_000; - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct RawPolicy { - version: u8, - cell: String, - incarnation: String, - desired_readers: u16, - revision: u64, -} - -/// Advisory desired read-replica count for one Cell incarnation. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct ReadPolicy { - cell: CellId, - incarnation: IncarnationId, - desired_readers: u16, - revision: u64, -} - -impl ReadPolicy { - fn new( - cell: CellId, - incarnation: IncarnationId, - desired_readers: u16, - revision: u64, - ) -> Result { - if incarnation.as_bytes().iter().all(|byte| *byte == 0) - || cell.as_bytes().iter().all(|byte| *byte == 0) - || desired_readers > MAX_READERS - || revision == 0 - { - return Err(Error::Control("invalid Cell read policy")); - } - Ok(Self { - cell, - incarnation, - desired_readers, - revision, - }) - } - - /// Returns the Cell this policy configures. - #[must_use] - pub const fn cell(&self) -> CellId { - self.cell - } - - /// Returns the incarnation this policy configures. - #[must_use] - pub const fn incarnation(&self) -> IncarnationId { - self.incarnation - } - - /// Returns the requested number of distinct non-owner readers. - #[must_use] - pub const fn desired_readers(&self) -> u16 { - self.desired_readers - } - - /// Returns the policy CAS revision. - #[must_use] - pub const fn revision(&self) -> u64 { - self.revision - } - - fn encode(self) -> Result> { - let raw = RawPolicy { - version: 1, - cell: encode_hex(self.cell.as_bytes()), - incarnation: encode_hex(self.incarnation.as_bytes()), - desired_readers: self.desired_readers, - revision: self.revision, - }; - let bytes = serde_json::to_vec(&raw)?; - if bytes.len() as u64 > MAX_POLICY_BYTES { - return Err(Error::Control("Cell read policy exceeds byte limit")); - } - Ok(bytes) - } - - fn decode(bytes: &[u8]) -> Result { - if bytes.len() as u64 > MAX_POLICY_BYTES { - return Err(Error::Control("Cell read policy exceeds byte limit")); - } - let raw: RawPolicy = serde_json::from_slice(bytes)?; - if raw.version != 1 { - return Err(Error::Control("unsupported Cell read policy version")); - } - let policy = Self::new( - CellId::from_bytes(decode_hex(&raw.cell)?), - IncarnationId::from_bytes(decode_hex(&raw.incarnation)?), - raw.desired_readers, - raw.revision, - )?; - if policy.encode()? != bytes { - return Err(Error::Control("Cell read policy is not canonical")); - } - Ok(policy) - } -} - -/// A policy and the exact conditional-write token that read it. -pub struct VersionedReadPolicy { - value: ReadPolicy, - token: ETag, -} - -impl VersionedReadPolicy { - /// Returns the decoded desired-count policy. - #[must_use] - pub const fn value(&self) -> ReadPolicy { - self.value - } -} - -/// Loads and conditionally updates advisory read-replica targets in S3. -#[derive(Clone)] -pub struct ReadPolicyStore { - layout: CellStorageLayout, -} - -impl ReadPolicyStore { - /// Binds read policies to the same application root as Cell control. - #[must_use] - pub fn new(layout: CellStorageLayout) -> Self { - Self { layout } - } - - /// Loads one exact policy; absence means zero desired readers. - pub async fn load(&self, cell: CellId) -> Result> { - let path = self.layout.read_policy_path(cell.as_bytes()); - let (bytes, token) = match self - .layout - .store() - .get_with_etag_bounded(&path, MAX_POLICY_BYTES) - .await - { - Ok(found) => found, - Err(StorageError::NotFound { .. }) => return Ok(None), - Err(error) => return Err(error.into()), - }; - let value = ReadPolicy::decode(&bytes)?; - if value.cell != cell { - return Err(Error::Control("Cell read policy path changed Cell")); - } - Ok(Some(VersionedReadPolicy { value, token })) - } - - /// Strict-creates the first target count for this Cell incarnation. - pub async fn create( - &self, - cell: CellId, - incarnation: IncarnationId, - desired_readers: u16, - ) -> Result { - let value = ReadPolicy::new(cell, incarnation, desired_readers, 1)?; - let path = self.layout.read_policy_path(cell.as_bytes()); - match self - .layout - .store() - .create_strict_with_etag(&path, Bytes::from(value.encode()?)) - .await - { - Ok(token) => Ok(VersionedReadPolicy { value, token }), - Err(error) => match self.load(cell).await? { - Some(current) if current.value == value => Ok(current), - Some(_) | None => Err(error.into()), - }, - } - } - - /// CASes the next desired count, preserving the observed incarnation. - pub async fn update( - &self, - observed: &VersionedReadPolicy, - desired_readers: u16, - ) -> Result { - self.update_to(observed, observed.value.incarnation, desired_readers) - .await - } - - /// Rebinds an observed policy after the caller has authorized a new Cell incarnation. - /// - /// The caller must first compare the new incarnation with current Cell - /// authority. This policy is advisory and cannot authorize a replica read. - pub async fn replace_incarnation( - &self, - observed: &VersionedReadPolicy, - incarnation: IncarnationId, - desired_readers: u16, - ) -> Result { - if incarnation == observed.value.incarnation { - return Err(Error::Control( - "Cell read policy incarnation did not change", - )); - } - self.update_to(observed, incarnation, desired_readers).await - } - - async fn update_to( - &self, - observed: &VersionedReadPolicy, - incarnation: IncarnationId, - desired_readers: u16, - ) -> Result { - let value = ReadPolicy::new( - observed.value.cell, - incarnation, - desired_readers, - observed - .value - .revision - .checked_add(1) - .ok_or(Error::Control("Cell read policy revision overflow"))?, - )?; - let path = self.layout.read_policy_path(value.cell.as_bytes()); - match self - .layout - .store() - .update(&path, Bytes::from(value.encode()?), observed.token.clone()) - .await - { - Ok(token) => Ok(VersionedReadPolicy { value, token }), - Err(error) => match self.load(value.cell).await? { - Some(current) if current.value == value => Ok(current), - Some(_) | None => Err(error.into()), - }, - } - } -} diff --git a/crates/crab-cell-runtime/src/recovery.rs b/crates/crab-cell-runtime/src/recovery.rs deleted file mode 100644 index b78103ad1..000000000 --- a/crates/crab-cell-runtime/src/recovery.rs +++ /dev/null @@ -1,8 +0,0 @@ -//! Recovery artifacts, release control, backup pins, and retention. - -pub mod artifacts; -pub mod backup; -pub mod manifest; -pub mod release; -pub mod release_progress; -pub mod retention; diff --git a/crates/crab-cell-runtime/src/recovery/artifacts.rs b/crates/crab-cell-runtime/src/recovery/artifacts.rs deleted file mode 100644 index c84c14bf7..000000000 --- a/crates/crab-cell-runtime/src/recovery/artifacts.rs +++ /dev/null @@ -1,413 +0,0 @@ -//! Recovery artifact cache and its admission. -use std::{ - collections::BTreeMap, - fs::{self, File, OpenOptions}, - path::{Path, PathBuf}, - sync::{ - Arc, Mutex, - atomic::{AtomicU64, Ordering}, - }, -}; - -use crate::identity::encode_hex; -use crate::ltx::{DiskBudget, DiskReservation, Limits as ReplicaLimits}; -use crate::recovery::manifest::{RecoveryArtifact, RecoveryArtifactKey, RecoveryArtifactStore}; -use crate::{Error, Result}; - -const MAX_ARTIFACTS: usize = 256; - -/// One node-owned, digest-addressed recovery bundle cache. -/// -/// Entries are never trusted on startup. They are admitted only after the -/// immutable object and manifest have been published, and every hit reparses -/// and rehashes the bundle before returning it. -pub struct RecoveryArtifactRegistry { - root: PathBuf, - limits: ReplicaLimits, - budget: DiskBudget, - admission: Mutex<()>, - entries: Mutex>>, - clock: AtomicU64, -} - -struct ArtifactEntry { - path: PathBuf, - size: u64, - _reservation: DiskReservation, - last_used: AtomicU64, -} - -impl crab_ltx::bundle::BundleLease for ArtifactEntry {} - -impl Drop for ArtifactEntry { - fn drop(&mut self) { - let _ = fs::remove_file(&self.path); - } -} - -impl RecoveryArtifactRegistry { - /// Creates the on-disk cache, requiring an absolute root and a non-zero - /// disk budget. - pub fn new(root: PathBuf, limits: ReplicaLimits, budget: DiskBudget) -> Result { - if !root.is_absolute() || budget.capacity() == 0 { - return Err(Error::Node( - "recovery artifact registry requires an absolute path and disk budget", - )); - } - fs::create_dir_all(&root)?; - reclaim_stale_files(&root)?; - sync_directory(&root)?; - Ok(Self { - root, - limits, - budget, - admission: Mutex::new(()), - entries: Mutex::new(BTreeMap::new()), - clock: AtomicU64::new(0), - }) - } - - #[cfg(test)] - fn entry_count(&self) -> usize { - self.entries - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .len() - } - - fn next_access(&self) -> u64 { - self.clock.fetch_add(1, Ordering::Relaxed).saturating_add(1) - } - - fn evict_one(entries: &mut BTreeMap>) -> bool { - let Some(key) = entries - .iter() - .filter(|(_, entry)| Arc::strong_count(entry) == 1) - .min_by_key(|(_, entry)| entry.last_used.load(Ordering::Relaxed)) - .map(|(key, _)| *key) - else { - return false; - }; - entries.remove(&key).is_some() - } - - fn reserve(&self, size: u64) -> Result { - let mut entries = self - .entries - .lock() - .map_err(|_| Error::Node("recovery artifact lock poisoned"))?; - while entries.len() >= MAX_ARTIFACTS { - if !Self::evict_one(&mut entries) { - return Err(Error::Capacity("recovery artifact cache")); - } - } - drop(entries); - loop { - match self.budget.try_reserve(size) { - Ok(reservation) => return Ok(reservation), - Err(_) => { - let mut entries = self - .entries - .lock() - .map_err(|_| Error::Node("recovery artifact lock poisoned"))?; - if !Self::evict_one(&mut entries) { - return Err(Error::Capacity("recovery artifact cache")); - } - } - } - } - } - - fn invalidate(&self, key: &RecoveryArtifactKey, entry: &Arc) { - if let Ok(mut entries) = self.entries.lock() - && entries - .get(key) - .is_some_and(|current| Arc::ptr_eq(current, entry)) - { - entries.remove(key); - } - } - - fn destination(&self, key: &RecoveryArtifactKey) -> PathBuf { - self.root - .join(format!("{}.bundle", encode_hex(&key.cache_digest()))) - } -} - -impl RecoveryArtifactStore for RecoveryArtifactRegistry { - fn retain(&self, key: RecoveryArtifactKey, bundle: crab_ltx::bundle::Bundle) -> Result<()> { - let _admission = self - .admission - .lock() - .map_err(|_| Error::Node("recovery artifact lock poisoned"))?; - if bundle.is_empty() || bundle.len() > self.limits.max_plan_bytes { - return Err(Error::Capacity("recovery artifact size")); - } - let digest = key.bundle_digest(); - if bundle.digest() != digest { - return Err(Error::Ltx(crab_ltx::CrabError::ChecksumMismatch)); - } - let destination = self.destination(&key); - let size = bundle.len(); - if let Some(existing) = self - .entries - .lock() - .map_err(|_| Error::Node("recovery artifact lock poisoned"))? - .get(&key) - .cloned() - { - let valid = existing.size == size - && crab_ltx::bundle::Bundle::decode_file(&existing.path, self.limits) - .is_ok_and(|candidate| candidate.digest() == digest); - if valid { - existing - .last_used - .store(self.next_access(), Ordering::Relaxed); - return Ok(()); - } - self.invalidate(&key, &existing); - } - - let reservation = self.reserve(size)?; - let source = match bundle.detach_file() { - Ok(path) => path, - Err(error) => return Err(Error::Ltx(error)), - }; - if destination.exists() { - let _ = fs::remove_file(&source); - return Err(Error::Node("recovery artifact destination already exists")); - } - let temporary = self.root.join(format!( - ".{}.{}.tmp", - encode_hex(&digest), - self.next_access() - )); - if let Err(error) = relocate_file(&source, &temporary) { - let _ = fs::remove_file(&source); - let _ = fs::remove_file(&temporary); - return Err(error.into()); - } - if let Err(error) = sync_file(&temporary).and_then(|()| { - fs::rename(&temporary, &destination)?; - sync_directory(&self.root) - }) { - let _ = fs::remove_file(&temporary); - let _ = fs::remove_file(&destination); - return Err(error.into()); - } - - let entry = Arc::new(ArtifactEntry { - path: destination, - size, - _reservation: reservation, - last_used: AtomicU64::new(self.next_access()), - }); - let mut entries = self - .entries - .lock() - .map_err(|_| Error::Node("recovery artifact lock poisoned"))?; - if let Some(previous) = entries.insert(key, Arc::clone(&entry)) { - drop(previous); - } - Ok(()) - } - - fn load(&self, key: &RecoveryArtifactKey) -> Result> { - let entry = { - let entries = self - .entries - .lock() - .map_err(|_| Error::Node("recovery artifact lock poisoned"))?; - entries.get(key).cloned() - }; - let Some(entry) = entry else { - return Ok(None); - }; - entry.last_used.store(self.next_access(), Ordering::Relaxed); - let bundle = match crab_ltx::bundle::Bundle::decode_file(&entry.path, self.limits) { - Ok(bundle) if bundle.digest() == key.bundle_digest() => bundle, - Ok(_) | Err(_) => { - self.invalidate(key, &entry); - return Ok(None); - } - }; - Ok(Some(RecoveryArtifact::new(bundle, entry))) - } -} - -fn sync_file(path: &Path) -> std::io::Result<()> { - File::open(path)?.sync_all() -} - -fn relocate_file(source: &Path, destination: &Path) -> std::io::Result<()> { - match fs::rename(source, destination) { - Ok(()) => Ok(()), - Err(error) if error.kind() == std::io::ErrorKind::CrossesDevices => { - fs::copy(source, destination)?; - sync_file(destination)?; - fs::remove_file(source) - } - Err(error) => Err(error), - } -} - -fn sync_directory(path: &Path) -> std::io::Result<()> { - OpenOptions::new().read(true).open(path)?.sync_all() -} - -fn reclaim_stale_files(root: &Path) -> std::io::Result<()> { - for entry in fs::read_dir(root)? { - let entry = entry?; - if !entry.file_type()?.is_file() { - continue; - } - let name = entry.file_name(); - let Some(name) = name.to_str() else { - continue; - }; - if name.ends_with(".bundle") || name.ends_with(".tmp") { - fs::remove_file(entry.path())?; - } - } - Ok(()) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::control::RootRef; - use crate::identity::IncarnationId; - use crate::identity::{ApplicationId, CellId, Digest, SessionId}; - use crate::recovery::manifest::RecoveryArtifactKey; - use crab_ltx::{Limits, Position, bundle::BundleBuilder, bundle::BundleEntry}; - use tempfile::TempDir; - - fn bundle_fixture(directory: &Path, marker: u8) -> crab_ltx::bundle::Bundle { - let database_path = directory.join(format!("database-{marker}")); - let mut database = crab_ltx::Db::open(&database_path, Limits::default()).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - database.capture().unwrap(); - database - .transaction(|transaction| { - transaction.execute("INSERT INTO values_ VALUES (1)", [])?; - Ok(()) - }) - .unwrap(); - let capture = database.capture().unwrap(); - let segment = capture.segments.first().unwrap(); - let mut builder = BundleBuilder::new_temp(directory, Limits::default()).unwrap(); - builder - .push(BundleEntry::for_cell( - [marker; 32], - [marker; 16], - segment.info().clone(), - std::fs::read(segment.path()).unwrap(), - )) - .unwrap(); - let bundle = builder.finish().unwrap(); - database.close().unwrap(); - bundle - } - - fn key(bundle: &crab_ltx::bundle::Bundle, marker: u8) -> RecoveryArtifactKey { - RecoveryArtifactKey::new( - SessionId::from_bytes([1; 16]), - 2, - ApplicationId::from_bytes([2; 16]), - CellId::from_bytes([marker; 32]), - IncarnationId::from_bytes([marker; 16]), - 3, - 4, - 4, - RootRef { - digest: Digest::from_bytes([5; 32]), - txid: 1, - checksum: (1_u64 << 63) | 1, - commit_sequence: 1, - }, - Position { - txid: 2, - checksum: (1_u64 << 63) | 2, - }, - 2, - Digest::from_bytes(bundle.digest()), - ) - } - - #[test] - fn retains_verified_bundle_and_drops_corrupt_cache_entry() { - let temporary = TempDir::new().unwrap(); - let registry = RecoveryArtifactRegistry::new( - temporary.path().join("registry"), - Limits::default(), - DiskBudget::new(1 << 20), - ) - .unwrap(); - let bundle = bundle_fixture(temporary.path(), 7); - let key = key(&bundle, 7); - registry.retain(key, bundle).unwrap(); - assert_eq!(registry.entry_count(), 1); - assert!(registry.load(&key).unwrap().is_some()); - - let path = registry.destination(&key); - std::fs::write(&path, b"corrupt").unwrap(); - assert!(registry.load(&key).unwrap().is_none()); - assert_eq!(registry.entry_count(), 0); - assert!(!path.exists()); - } - - #[test] - fn reclaims_only_registry_artifacts_from_previous_process() { - let temporary = TempDir::new().unwrap(); - let root = temporary.path().join("registry"); - std::fs::create_dir_all(&root).unwrap(); - let stale = root.join("stale.bundle"); - let temporary_file = root.join("stale.tmp"); - let unrelated = root.join("operator-note"); - std::fs::write(&stale, b"stale").unwrap(); - std::fs::write(&temporary_file, b"stale").unwrap(); - std::fs::write(&unrelated, b"keep").unwrap(); - - let _registry = - RecoveryArtifactRegistry::new(root, Limits::default(), DiskBudget::new(1 << 20)) - .unwrap(); - assert!(!stale.exists()); - assert!(!temporary_file.exists()); - assert!(unrelated.exists()); - } - - #[test] - fn rejects_admission_when_all_artifact_slots_are_in_use() { - let temporary = TempDir::new().unwrap(); - let registry = RecoveryArtifactRegistry::new( - temporary.path().join("registry"), - Limits::default(), - DiskBudget::new((MAX_ARTIFACTS + 1) as u64), - ) - .unwrap(); - let bundle = bundle_fixture(temporary.path(), 9); - let mut held = Vec::with_capacity(MAX_ARTIFACTS); - { - let mut entries = registry.entries.lock().unwrap(); - for marker in 0..MAX_ARTIFACTS { - let reservation = registry.budget.try_reserve(1).unwrap(); - let entry = Arc::new(ArtifactEntry { - path: registry.root.join(format!("held-{marker}.bundle")), - size: 1, - _reservation: reservation, - last_used: AtomicU64::new(marker as u64), - }); - entries.insert(key(&bundle, marker as u8), Arc::clone(&entry)); - held.push(entry); - } - } - - assert!(matches!( - registry.reserve(1), - Err(Error::Capacity("recovery artifact cache")) - )); - drop(held); - } -} diff --git a/crates/crab-cell-runtime/src/recovery/backup.rs b/crates/crab-cell-runtime/src/recovery/backup.rs deleted file mode 100644 index 39b5fc87b..000000000 --- a/crates/crab-cell-runtime/src/recovery/backup.rs +++ /dev/null @@ -1,700 +0,0 @@ -//! Backup pins over revision-pinned catalog shards and pages. -use std::collections::BTreeMap; - -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::StorageError; -use serde::{Deserialize, Serialize}; - -use crate::cell::application::ApplicationIdentity; -use crate::cell::catalog::CellCatalog; -use crate::control::Control; -use crate::identity::Digest; -use crate::identity::RequestId; -use crate::identity::{decode_hex, encode_hex}; -use crate::ltx::{Host as ReplicaHost, Limits as ReplicaLimits}; -use crate::recovery::release::{ReleaseRecord, ReleaseState, ReleaseStore}; -use crate::{Error, Result}; - -mod restore; -pub use restore::BackupRestore; - -const MAX_PIN_BYTES: u64 = 32 * 1024; -const MAX_SHARD_BYTES: u64 = 256 * 1024; -const MAX_PAGE_BYTES: u64 = 1024 * 1024; -const MAX_RELEASE_SNAPSHOT_BYTES: u64 = 24 * 1024; -const MAX_PAGES_PER_SHARD: usize = 2048; - -/// Immutable backup boundary published after every referenced Cell root verifies. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct BackupPin { - application: crate::ApplicationId, - id: RequestId, - created_at_ms: i64, - control_count: u64, - catalog_revisions: Vec, - release: [u8; 32], - shards: Vec, -} - -impl BackupPin { - /// Returns the request that created the pin. - #[must_use] - pub const fn id(&self) -> RequestId { - self.id - } - - /// Returns the logical time the pin was created. - #[must_use] - pub const fn created_at_ms(&self) -> i64 { - self.created_at_ms - } - - /// Returns how many control records the pin covers. - #[must_use] - pub const fn control_count(&self) -> u64 { - self.control_count - } - - /// Returns the application the pin belongs to. - #[must_use] - pub const fn application(&self) -> crate::ApplicationId { - self.application - } - - /// Returns the catalog revision the pin covers for each shard. - #[must_use] - pub fn catalog_revisions(&self) -> &[u64] { - &self.catalog_revisions - } - - /// Returns the release digest the pin covers. - #[must_use] - pub const fn release_digest(&self) -> [u8; 32] { - self.release - } -} - -/// One revision-pinned catalog shard and its immutable page dependencies. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct PinnedCatalogShard { - /// Catalog shard the record pins. - pub shard: u8, - /// Revision the shard was pinned at. - pub revision: u64, - /// Immutable page digests the shard referenced. - pub pages: Vec, -} - -#[derive(Clone, Debug, PartialEq, Eq)] -struct ShardRef { - shard: u8, - digest: [u8; 32], -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct PinWire { - application: String, - catalog_revisions: Vec, - control_count: String, - created_at_ms: String, - pin: String, - release: String, - shards: Vec, - version: u8, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct ShardRefWire { - digest: String, - shard: u8, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct ShardWire { - catalog_pages: Vec, - catalog_revision: String, - control_count: String, - control_pages: Vec, - shard: u8, - version: u8, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct PageWire { - controls: Vec, - version: u8, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct ReleaseWire { - descriptors: Vec, - record: String, - version: u8, -} - -/// Creates and reopens application-scoped backup pins. -/// -/// Immutable pages and shard manifests may be left behind by interruption. The -/// pin pointer is strict-created only after every exact LTX graph verifies. -#[derive(Clone)] -pub struct BackupPinStore { - layout: CellStorageLayout, - identity: ApplicationIdentity, - limits: ReplicaLimits, - host: ReplicaHost, -} - -impl BackupPinStore { - /// Creates a backup-pin store after checking that the layout belongs to the - /// same application as the identity. - pub fn new( - layout: CellStorageLayout, - identity: ApplicationIdentity, - limits: ReplicaLimits, - host: ReplicaHost, - ) -> Result { - if layout.application_id() != identity.application().as_bytes() { - return Err(Error::Backup("layout and application identity differ")); - } - Ok(Self { - layout, - identity, - limits, - host, - }) - } - - /// Verifies exact roots, uploads immutable pages, and strict-creates one pin. - /// - /// The caller must remain enrolled in the application's maintenance drain - /// from before its final Ready check until this future returns. This method - /// validates Ready state but cannot atomically fence a separate collector. - pub async fn create( - &self, - id: RequestId, - created_at_ms: i64, - catalog: Vec, - mut controls: Vec, - ) -> Result { - if id.as_bytes().iter().all(|byte| *byte == 0) || created_at_ms < 0 { - return Err(Error::Backup("pin identity or timestamp is invalid")); - } - if catalog.len() != 256 - || catalog - .iter() - .enumerate() - .any(|(index, shard)| usize::from(shard.shard) != index) - { - return Err(Error::Backup("pin must name all 256 catalog shards")); - } - let catalog_store = CellCatalog::new(self.layout.clone(), self.identity.tenant()); - let mut catalog_cells = Vec::new(); - for shard in &catalog { - catalog_cells.extend( - catalog_store - .pinned_cells(shard.shard, shard.revision, &shard.pages) - .await?, - ); - } - let catalog_revisions = catalog.iter().map(|shard| shard.revision).collect(); - controls.sort_unstable_by_key(|control| *control.cell.as_bytes()); - if controls.windows(2).any(|pair| pair[0].cell == pair[1].cell) { - return Err(Error::Backup("pin contains duplicate Cell controls")); - } - if controls - .iter() - .map(|control| control.cell) - .ne(catalog_cells.iter().copied()) - { - return Err(Error::Backup("pin controls differ from the pinned catalog")); - } - for control in &controls { - control.encode()?; - self.verify_root(control).await?; - } - let release = self.snapshot_release().await?; - - let mut groups = BTreeMap::>::new(); - for control in controls { - groups - .entry(control.cell.as_bytes()[0]) - .or_default() - .push(control); - } - let mut shards = Vec::with_capacity(groups.len()); - let mut control_count = 0u64; - for catalog_shard in catalog { - let shard = catalog_shard.shard; - let controls = groups.remove(&shard).unwrap_or_default(); - let shard_control_count = controls.len() as u64; - control_count = control_count - .checked_add(shard_control_count) - .ok_or(Error::Backup("pin control count overflow"))?; - let mut pages = Vec::new(); - let mut page = Vec::new(); - for control in controls { - let encoded = hex(&control.encode()?); - page.push(encoded); - if page_bytes(&page)?.len() as u64 > MAX_PAGE_BYTES { - let last = page - .pop() - .ok_or(Error::Backup("pin page construction failed"))?; - if page.is_empty() { - return Err(Error::Backup("one control exceeds the pin page limit")); - } - pages.push(self.publish_page(&page).await?); - page.push(last); - } - } - if !page.is_empty() { - pages.push(self.publish_page(&page).await?); - } - if pages.len() > MAX_PAGES_PER_SHARD { - return Err(Error::Backup("pin shard page count is invalid")); - } - if pages.is_empty() && catalog_shard.pages.is_empty() { - continue; - } - let catalog_pages = catalog_shard - .pages - .iter() - .map(|digest| *digest.as_bytes()) - .collect::>(); - let body = encode_shard( - shard, - catalog_shard.revision, - &catalog_pages, - shard_control_count, - &pages, - )?; - let digest = publish_object(&self.layout, body).await?; - shards.push(ShardRef { shard, digest }); - } - let pin = BackupPin { - application: self.identity.application(), - id, - created_at_ms, - control_count, - catalog_revisions, - release, - shards, - }; - let body = pin.encode()?; - let path = self.layout.pin_path(id.as_bytes()); - match self - .layout - .store() - .create_strict(&path, Bytes::from(body.clone())) - .await - { - Ok(()) => Ok(pin), - Err(create_error) => match self.load(id).await? { - Some(existing) if existing.encode()? == body => Ok(existing), - Some(_) => Err(Error::Backup("pin ID already names different contents")), - None => Err(create_error.into()), - }, - } - } - - /// Loads one exact pin pointer without traversing its immutable pages. - pub async fn load(&self, id: RequestId) -> Result> { - let (body, _) = match self - .layout - .store() - .get_with_etag_bounded(&self.layout.pin_path(id.as_bytes()), MAX_PIN_BYTES) - .await - { - Ok(value) => value, - Err(StorageError::NotFound { .. }) => return Ok(None), - Err(error) => return Err(error.into()), - }; - let pin = BackupPin::decode(&body)?; - if pin.application != self.identity.application() || pin.id != id { - return Err(Error::Backup("pin pointer scope does not match its path")); - } - Ok(Some(pin)) - } - - /// Loads every canonical control captured by one pin. - pub async fn controls(&self, pin: &BackupPin) -> Result> { - Ok(self.manifest(pin).await?.controls) - } - - /// Reopens every immutable dependency captured by one pin. - pub async fn verify(&self, pin: &BackupPin) -> Result> { - let controls = self.controls(pin).await?; - for control in &controls { - self.verify_root(control).await?; - } - Ok(controls) - } - - async fn snapshot_release(&self) -> Result<[u8; 32]> { - let releases = ReleaseStore::new(self.layout.clone(), self.identity)?; - let record = releases - .load() - .await? - .ok_or(Error::Backup("release is absent from backup scope"))? - .record() - .clone(); - if record.state() != ReleaseState::Ready { - return Err(Error::Backup("backup release is not ready")); - } - let descriptors = release_descriptors(&record); - for digest in &descriptors { - releases.descriptor(*digest).await?; - } - publish_object(&self.layout, encode_release(&record, &descriptors)?).await - } - - async fn load_release(&self, digest: [u8; 32]) -> Result<(ReleaseRecord, Vec)> { - let body = self - .read_object(&digest, MAX_RELEASE_SNAPSHOT_BYTES) - .await?; - let (record, descriptors) = decode_release(&body, self.identity)?; - let releases = ReleaseStore::new(self.layout.clone(), self.identity)?; - for digest in &descriptors { - releases.descriptor(*digest).await?; - } - Ok((record, descriptors)) - } - - async fn verify_root(&self, control: &Control) -> Result<()> { - if let Some(root) = control.ltx_root() { - crab_ltx::CellReplica::new( - self.layout.clone(), - *control.cell.as_bytes(), - *control.incarnation.as_bytes(), - self.limits, - )? - .with_host(self.host.clone()) - .reachable_objects(&root) - .await?; - } - Ok(()) - } - - async fn publish_page(&self, controls: &[String]) -> Result<[u8; 32]> { - publish_object(&self.layout, encode_page(controls)?).await - } - - async fn read_object(&self, digest: &[u8; 32], limit: u64) -> Result> { - let (body, _) = self - .layout - .store() - .get_with_etag_bounded(&self.layout.pin_object_path(digest), limit) - .await?; - if blake3::hash(&body).as_bytes() != digest { - return Err(Error::Backup("pin object digest mismatch")); - } - Ok(body.to_vec()) - } -} - -impl BackupPin { - fn encode(&self) -> Result> { - if self.created_at_ms < 0 - || self.id.as_bytes().iter().all(|byte| *byte == 0) - || self.catalog_revisions.len() != 256 - || self.shards.len() > 256 - || self - .shards - .windows(2) - .any(|pair| pair[0].shard >= pair[1].shard) - || self - .catalog_revisions - .iter() - .enumerate() - .any(|(shard, revision)| { - *revision != 0 - && !self - .shards - .iter() - .any(|reference| usize::from(reference.shard) == shard) - }) - { - return Err(Error::Backup("pin pointer bounds are invalid")); - } - let body = serde_json::to_vec(&PinWire { - application: encode_hex(self.application.as_bytes()), - catalog_revisions: self.catalog_revisions.iter().map(u64::to_string).collect(), - control_count: self.control_count.to_string(), - created_at_ms: self.created_at_ms.to_string(), - pin: encode_hex(self.id.as_bytes()), - release: encode_hex(&self.release), - shards: self - .shards - .iter() - .map(|shard| ShardRefWire { - digest: encode_hex(&shard.digest), - shard: shard.shard, - }) - .collect(), - version: 1, - })?; - if body.len() as u64 > MAX_PIN_BYTES { - return Err(Error::Backup("pin pointer exceeds 32 KiB")); - } - Ok(body) - } - - fn decode(body: &[u8]) -> Result { - if body.len() as u64 > MAX_PIN_BYTES { - return Err(Error::Backup("pin pointer exceeds 32 KiB")); - } - let wire: PinWire = serde_json::from_slice(body)?; - if wire.version != 1 { - return Err(Error::Backup("unsupported pin pointer version")); - } - let pin = Self { - application: crate::ApplicationId::from_bytes(decode_hex(&wire.application)?), - catalog_revisions: wire - .catalog_revisions - .iter() - .map(|revision| canonical_u64(revision)) - .collect::>>()?, - id: RequestId::from_bytes(decode_hex(&wire.pin)?), - created_at_ms: canonical_i64(&wire.created_at_ms)?, - control_count: canonical_u64(&wire.control_count)?, - release: decode_hex(&wire.release)?, - shards: wire - .shards - .into_iter() - .map(|shard| { - Ok(ShardRef { - shard: shard.shard, - digest: decode_hex(&shard.digest)?, - }) - }) - .collect::>>()?, - }; - if pin.encode()?.as_slice() != body { - return Err(Error::Backup("pin pointer is not canonical")); - } - Ok(pin) - } -} - -struct ShardManifest { - catalog_revision: u64, - catalog_pages: Vec<[u8; 32]>, - control_count: u64, - control_pages: Vec<[u8; 32]>, -} - -fn encode_page(controls: &[String]) -> Result> { - if controls.is_empty() { - return Err(Error::Backup("pin page is empty")); - } - let body = page_bytes(controls)?; - if body.len() as u64 > MAX_PAGE_BYTES { - return Err(Error::Backup("pin page exceeds 1 MiB")); - } - Ok(body) -} - -fn page_bytes(controls: &[String]) -> Result> { - Ok(serde_json::to_vec(&PageWire { - controls: controls.to_vec(), - version: 1, - })?) -} - -fn decode_page(body: &[u8]) -> Result> { - if body.len() as u64 > MAX_PAGE_BYTES { - return Err(Error::Backup("pin page exceeds 1 MiB")); - } - let page: PageWire = serde_json::from_slice(body)?; - if page.version != 1 || page.controls.is_empty() || encode_page(&page.controls)? != body { - return Err(Error::Backup("pin page is not canonical")); - } - Ok(page.controls) -} - -fn encode_shard( - shard: u8, - catalog_revision: u64, - catalog_pages: &[[u8; 32]], - control_count: u64, - control_pages: &[[u8; 32]], -) -> Result> { - if (catalog_revision == 0) != catalog_pages.is_empty() - || catalog_pages.len() > 256 - || (control_count == 0) != control_pages.is_empty() - || control_pages.len() > MAX_PAGES_PER_SHARD - || (catalog_pages.is_empty() && control_pages.is_empty()) - { - return Err(Error::Backup("pin shard bounds are invalid")); - } - let body = serde_json::to_vec(&ShardWire { - catalog_pages: catalog_pages - .iter() - .map(|digest| encode_hex(digest)) - .collect(), - catalog_revision: catalog_revision.to_string(), - control_count: control_count.to_string(), - control_pages: control_pages - .iter() - .map(|digest| encode_hex(digest)) - .collect(), - shard, - version: 1, - })?; - if body.len() as u64 > MAX_SHARD_BYTES { - return Err(Error::Backup("pin shard exceeds 256 KiB")); - } - Ok(body) -} - -fn decode_shard(body: &[u8], expected_shard: u8) -> Result { - if body.len() as u64 > MAX_SHARD_BYTES { - return Err(Error::Backup("pin shard exceeds 256 KiB")); - } - let wire: ShardWire = serde_json::from_slice(body)?; - if wire.version != 1 || wire.shard != expected_shard { - return Err(Error::Backup("pin shard identity differs")); - } - let manifest = ShardManifest { - catalog_revision: canonical_u64(&wire.catalog_revision)?, - catalog_pages: wire - .catalog_pages - .iter() - .map(|digest| decode_hex(digest)) - .collect::>>()?, - control_count: canonical_u64(&wire.control_count)?, - control_pages: wire - .control_pages - .iter() - .map(|digest| decode_hex(digest)) - .collect::>>()?, - }; - if encode_shard( - wire.shard, - manifest.catalog_revision, - &manifest.catalog_pages, - manifest.control_count, - &manifest.control_pages, - )? != body - { - return Err(Error::Backup("pin shard is not canonical")); - } - Ok(manifest) -} - -fn release_descriptors(record: &ReleaseRecord) -> Vec { - let mut descriptors = [record.current(), record.desired()] - .into_iter() - .flatten() - .collect::>(); - descriptors.sort_unstable_by_key(|digest| *digest.as_bytes()); - descriptors.dedup(); - descriptors -} - -fn encode_release(record: &ReleaseRecord, descriptors: &[Digest]) -> Result> { - if descriptors != release_descriptors(record) { - return Err(Error::Backup( - "release descriptor set differs from its record", - )); - } - let body = serde_json::to_vec(&ReleaseWire { - descriptors: descriptors - .iter() - .map(|digest| encode_hex(digest.as_bytes())) - .collect(), - record: encode_hex(&record.encode()?), - version: 1, - })?; - if body.len() as u64 > MAX_RELEASE_SNAPSHOT_BYTES { - return Err(Error::Backup("release snapshot exceeds 24 KiB")); - } - Ok(body) -} - -fn decode_release( - body: &[u8], - identity: ApplicationIdentity, -) -> Result<(ReleaseRecord, Vec)> { - if body.len() as u64 > MAX_RELEASE_SNAPSHOT_BYTES { - return Err(Error::Backup("release snapshot exceeds 24 KiB")); - } - let wire: ReleaseWire = serde_json::from_slice(body)?; - if wire.version != 1 { - return Err(Error::Backup("unsupported release snapshot version")); - } - let record = ReleaseRecord::decode(&unhex(&wire.record)?, identity)?; - let descriptors = wire - .descriptors - .iter() - .map(|digest| decode_hex(digest).map(Digest::from_bytes)) - .collect::>>()?; - if encode_release(&record, &descriptors)? != body { - return Err(Error::Backup("release snapshot is not canonical")); - } - Ok((record, descriptors)) -} - -async fn publish_object(layout: &CellStorageLayout, body: Vec) -> Result<[u8; 32]> { - let digest = *blake3::hash(&body).as_bytes(); - layout - .store() - .put(&layout.pin_object_path(&digest), Bytes::from(body)) - .await?; - Ok(digest) -} - -fn canonical_u64(value: &str) -> Result { - let parsed = value - .parse::() - .map_err(|_| Error::Backup("pin number is invalid"))?; - if parsed.to_string() != value { - return Err(Error::Backup("pin number is not canonical")); - } - Ok(parsed) -} - -fn canonical_i64(value: &str) -> Result { - let parsed = value - .parse::() - .map_err(|_| Error::Backup("pin timestamp is invalid"))?; - if parsed < 0 || parsed.to_string() != value { - return Err(Error::Backup("pin timestamp is not canonical")); - } - Ok(parsed) -} - -fn hex(bytes: &[u8]) -> String { - encode_hex(bytes) -} - -fn unhex(value: &str) -> Result> { - if !value.len().is_multiple_of(2) - || value - .bytes() - .any(|byte| !byte.is_ascii_digit() && !(b'a'..=b'f').contains(&byte)) - { - return Err(Error::Backup("pin control encoding is invalid")); - } - (0..value.len()) - .step_by(2) - .map(|offset| { - let high = nibble(value.as_bytes()[offset])?; - let low = nibble(value.as_bytes()[offset + 1])?; - Ok((high << 4) | low) - }) - .collect() -} - -fn nibble(value: u8) -> Result { - match value { - b'0'..=b'9' => Ok(value - b'0'), - b'a'..=b'f' => Ok(value - b'a' + 10), - _ => Err(Error::Backup("pin control encoding is invalid")), - } -} diff --git a/crates/crab-cell-runtime/src/recovery/backup/restore.rs b/crates/crab-cell-runtime/src/recovery/backup/restore.rs deleted file mode 100644 index f30426b6b..000000000 --- a/crates/crab-cell-runtime/src/recovery/backup/restore.rs +++ /dev/null @@ -1,350 +0,0 @@ -use std::collections::BTreeSet; - -use crab_ltx::CellStorageLayout; -use crab_storage::StorageError; -use object_store::path::Path; - -use super::{ - BackupPin, BackupPinStore, MAX_PAGE_BYTES, MAX_SHARD_BYTES, PinnedCatalogShard, decode_page, - decode_shard, -}; -use crate::cell::application::ApplicationIdentityStore; -use crate::cell::catalog::CellCatalog; -use crate::control::authority::CellAuthority; -use crate::control::{Control, ControlState}; -use crate::identity::Digest; -use crate::recovery::release::{ReleaseRecord, ReleaseState, ReleaseStore}; -use crate::{Error, Result}; - -/// Verified result of installing one pin into a separate storage prefix. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct BackupRestore { - application: crate::ApplicationId, - pin: crate::identity::RequestId, - control_count: u64, - immutable_object_count: u64, - nonempty_catalog_shards: u16, -} - -impl BackupRestore { - /// Returns the application the restore belongs to. - #[must_use] - pub const fn application(&self) -> crate::ApplicationId { - self.application - } - - /// Returns the backup pin the restore verified. - #[must_use] - pub const fn pin(&self) -> crate::identity::RequestId { - self.pin - } - - /// Returns how many control records the restore verified. - #[must_use] - pub const fn control_count(&self) -> u64 { - self.control_count - } - - /// Returns how many immutable objects the restore verified. - #[must_use] - pub const fn immutable_object_count(&self) -> u64 { - self.immutable_object_count - } - - /// Returns how many catalog shards held entries. - #[must_use] - pub const fn nonempty_catalog_shards(&self) -> u16 { - self.nonempty_catalog_shards - } -} - -pub(crate) struct BackupManifest { - pub(crate) release: ReleaseRecord, - pub(crate) release_descriptors: Vec, - pub(crate) catalog: Vec, - pub(crate) controls: Vec, - pub(crate) pin_objects: BTreeSet<[u8; 32]>, -} - -impl BackupPinStore { - /// Installs one verified pin into a separate root on the same object store. - /// - /// Immutable objects are copied conditionally before catalog, control, and - /// release pointers become visible. Repeating an interrupted restore adopts - /// only byte-identical state. Existing divergent state fails closed. - pub async fn restore(&self, pin: &BackupPin, destination_root: Path) -> Result { - if pin.application != self.identity.application() { - return Err(Error::Backup("pin belongs to another application")); - } - if CellStorageLayout::root_identity_path(&destination_root) == self.layout.identity_path() { - return Err(Error::Backup( - "restore destination must be an isolated prefix", - )); - } - let manifest = self.manifest(pin).await?; - if manifest.release.state() != ReleaseState::Ready { - return Err(Error::Backup("backup release is not ready")); - } - let mut roots = Vec::new(); - for control in &manifest.controls { - if control.state != ControlState::Tombstoned && control.root.is_none() { - return Err(Error::Backup("backup contains an unpublished Cell")); - } - if let Some(root) = control.ltx_root() { - let replica = crab_ltx::CellReplica::new( - self.layout.clone(), - *control.cell.as_bytes(), - *control.incarnation.as_bytes(), - self.limits, - )? - .with_host(self.host.clone()); - roots.push(( - control.cell, - control.incarnation, - root, - replica.reachable_objects(&root).await?, - )); - } - } - - let identities = - ApplicationIdentityStore::new(self.layout.store().clone(), destination_root); - identities.initialize(self.identity).await?; - let destination = identities.layout(self.identity).await?; - if destination.immutable_cache_identity() != self.layout.immutable_cache_identity() { - return Err(Error::Backup( - "restore destination must use the source object store", - )); - } - - for digest in &manifest.release_descriptors { - self.copy_if_absent( - self.layout.release_descriptor_path(digest.as_bytes()), - destination.release_descriptor_path(digest.as_bytes()), - ) - .await?; - } - for shard in &manifest.catalog { - for digest in &shard.pages { - self.copy_if_absent( - self.layout.catalog_object_path(digest.as_bytes()), - destination.catalog_object_path(digest.as_bytes()), - ) - .await?; - } - } - for digest in &manifest.pin_objects { - self.copy_if_absent( - self.layout.pin_object_path(digest), - destination.pin_object_path(digest), - ) - .await?; - } - let mut immutable_object_count = manifest - .release_descriptors - .len() - .checked_add(manifest.catalog.iter().map(|shard| shard.pages.len()).sum()) - .and_then(|count| count.checked_add(manifest.pin_objects.len())) - .ok_or(Error::Backup("restore object count overflow"))?; - for (cell, incarnation, _, objects) in &roots { - immutable_object_count = immutable_object_count - .checked_add(objects.len()) - .ok_or(Error::Backup("restore object count overflow"))?; - for object in objects { - self.copy_if_absent( - self.layout.incarnation_object_path( - cell.as_bytes(), - incarnation.as_bytes(), - &object.digest, - object.kind, - ), - destination.incarnation_object_path( - cell.as_bytes(), - incarnation.as_bytes(), - &object.digest, - object.kind, - ), - ) - .await?; - } - } - - let destination_releases = ReleaseStore::new(destination.clone(), self.identity)?; - for digest in &manifest.release_descriptors { - destination_releases.descriptor(*digest).await?; - } - let destination_catalog = CellCatalog::new(destination.clone(), self.identity.tenant()); - for shard in &manifest.catalog { - destination_catalog - .pinned_cells(shard.shard, shard.revision, &shard.pages) - .await?; - } - for (cell, incarnation, root, _) in &roots { - crab_ltx::CellReplica::new( - destination.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - self.limits, - )? - .with_host(self.host.clone()) - .reachable_objects(root) - .await?; - } - - let destination_authority = CellAuthority::new(destination.clone()); - for control in &manifest.controls { - let mut restored = control.clone(); - restored.owner = None; - if restored.state != ControlState::Tombstoned { - restored.state = ControlState::Idle; - } - destination_authority.install_restored(restored).await?; - } - for shard in &manifest.catalog { - destination_catalog - .install_pinned_shard(shard.shard, shard.revision, &shard.pages) - .await?; - } - destination_releases - .install_restored(manifest.release.clone()) - .await?; - self.copy_if_absent( - self.layout.pin_path(pin.id.as_bytes()), - destination.pin_path(pin.id.as_bytes()), - ) - .await?; - - let restored_pins = BackupPinStore::new( - destination.clone(), - self.identity, - self.limits, - self.host.clone(), - )?; - let restored_pin = restored_pins - .load(pin.id) - .await? - .ok_or(Error::Backup("restored pin pointer is absent"))?; - if restored_pin != *pin { - return Err(Error::Backup("restored pin pointer differs")); - } - restored_pins.verify(&restored_pin).await?; - for control in &manifest.controls { - let current = destination_authority - .load(control.cell) - .await? - .ok_or(Error::Backup("restored control is absent"))?; - if current.value().root != control.root - || current.value().code != control.code - || current.value().schema != control.schema - || current.value().next_due_ms != control.next_due_ms - || current.value().owner.is_some() - || (current.value().state != ControlState::Idle - && current.value().state != ControlState::Tombstoned) - { - return Err(Error::Backup("restored control differs from its pin")); - } - } - - Ok(BackupRestore { - application: self.identity.application(), - pin: pin.id, - control_count: pin.control_count, - immutable_object_count: immutable_object_count as u64, - nonempty_catalog_shards: manifest - .catalog - .iter() - .filter(|shard| shard.revision != 0) - .count() as u16, - }) - } - - pub(crate) async fn manifest(&self, pin: &BackupPin) -> Result { - if pin.application != self.identity.application() { - return Err(Error::Backup("pin belongs to another application")); - } - let (release, release_descriptors) = self.load_release(pin.release).await?; - let catalog_store = CellCatalog::new(self.layout.clone(), self.identity.tenant()); - let mut catalog = pin - .catalog_revisions - .iter() - .enumerate() - .map(|(shard, revision)| PinnedCatalogShard { - shard: shard as u8, - revision: *revision, - pages: Vec::new(), - }) - .collect::>(); - let mut catalog_cells = Vec::new(); - let mut controls = Vec::new(); - let mut pin_objects = BTreeSet::from([pin.release]); - for shard in &pin.shards { - pin_objects.insert(shard.digest); - let body = self.read_object(&shard.digest, MAX_SHARD_BYTES).await?; - let manifest = decode_shard(&body, shard.shard)?; - if manifest.catalog_revision != pin.catalog_revisions[usize::from(shard.shard)] { - return Err(Error::Backup("pin catalog revision differs from its shard")); - } - let catalog_pages = manifest - .catalog_pages - .iter() - .copied() - .map(Digest::from_bytes) - .collect::>(); - catalog[usize::from(shard.shard)].pages = catalog_pages.clone(); - catalog_cells.extend( - catalog_store - .pinned_cells(shard.shard, manifest.catalog_revision, &catalog_pages) - .await?, - ); - let before = controls.len(); - for digest in manifest.control_pages { - pin_objects.insert(digest); - let body = self.read_object(&digest, MAX_PAGE_BYTES).await?; - let page = decode_page(&body)?; - for encoded in page { - let control = Control::decode(&super::unhex(&encoded)?)?; - if control.cell.as_bytes()[0] != shard.shard { - return Err(Error::Backup("pin page crossed a catalog shard")); - } - controls.push(control); - } - } - if controls.len() - before != manifest.control_count as usize { - return Err(Error::Backup("pin shard control count differs")); - } - } - if controls.len() as u64 != pin.control_count - || controls - .windows(2) - .any(|pair| pair[0].cell.as_bytes() >= pair[1].cell.as_bytes()) - { - return Err(Error::Backup("pin controls are not globally ordered")); - } - if controls - .iter() - .map(|control| control.cell) - .ne(catalog_cells.iter().copied()) - { - return Err(Error::Backup("pin controls differ from the pinned catalog")); - } - Ok(BackupManifest { - release, - release_descriptors, - catalog, - controls, - pin_objects, - }) - } - - async fn copy_if_absent(&self, source: Path, destination: Path) -> Result<()> { - match self - .layout - .store() - .copy_if_not_exists(&source, &destination) - .await - { - Ok(()) | Err(StorageError::StateConflict { .. }) => Ok(()), - Err(error) => Err(error.into()), - } - } -} diff --git a/crates/crab-cell-runtime/src/recovery/manifest.rs b/crates/crab-cell-runtime/src/recovery/manifest.rs deleted file mode 100644 index 3b3cc90ba..000000000 --- a/crates/crab-cell-runtime/src/recovery/manifest.rs +++ /dev/null @@ -1,819 +0,0 @@ -//! Recovery manifests and the exact-root stores that serve them. -use std::sync::Arc; - -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::StorageError; -use serde::{Deserialize, Serialize}; - -use crate::control::{RecoveryOverlayRef, RootRef}; -use crate::identity::IncarnationId; -use crate::identity::{ApplicationId, CellId, Digest, SessionId, encode_hex}; -use crate::node::log::RecoveredCellTail; -use crate::{Error, Result}; - -const MAX_MANIFEST_BYTES: u64 = 2 << 20; -const MULTIPART_BYTES: usize = 8 << 20; - -/// One control-ready pointer returned after bundle and manifest publication. -pub struct PinnedRecoveryCell { - /// Application the recovered Cell belongs to. - pub application: ApplicationId, - /// Cell the pointer addresses. - pub cell: CellId, - /// Incarnation the overlay belongs to. - pub incarnation: IncarnationId, - /// Cell epoch the overlay was published under. - pub cell_epoch: u64, - /// Control-ready overlay reference. - pub recovery: RecoveryOverlayRef, -} - -/// Bounded object publication counters for one recovery manifest. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct RecoveryPublicationSummary { - /// Bundle bytes transferred. - pub bundle_bytes: u64, - /// Object-store reads. - pub object_reads: u64, - /// Object-store writes. - pub object_writes: u64, -} - -/// Control-ready recovery pointers and their publication work summary. -pub struct PinnedRecoveryCells { - /// Control-ready pointers, one per recovered Cell. - pub cells: Vec, - /// Publication work the pinning performed. - pub summary: RecoveryPublicationSummary, -} - -/// Exact immutable identity used for a local recovery-artifact cache entry. -#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] -pub struct RecoveryArtifactKey { - leader_session: [u8; 16], - log_epoch: u64, - application: [u8; 16], - cell: [u8; 32], - incarnation: [u8; 16], - cell_epoch: u64, - first_node_sequence: u64, - last_node_sequence: u64, - predecessor_digest: [u8; 32], - predecessor_txid: u64, - predecessor_checksum: u64, - predecessor_commit_sequence: u64, - final_txid: u64, - final_checksum: u64, - final_commit_sequence: u64, - bundle_digest: [u8; 32], -} - -impl RecoveryArtifactKey { - /// Builds the exact scope and content identity used for cache admission. - #[must_use] - #[expect( - clippy::too_many_arguments, - reason = "the persisted recovery scope is explicit" - )] - pub fn new( - leader_session: SessionId, - log_epoch: u64, - application: ApplicationId, - cell: CellId, - incarnation: IncarnationId, - cell_epoch: u64, - first_node_sequence: u64, - last_node_sequence: u64, - predecessor: RootRef, - final_position: crab_ltx::Position, - final_commit_sequence: u64, - bundle_digest: Digest, - ) -> Self { - Self { - leader_session: *leader_session.as_bytes(), - log_epoch, - application: *application.as_bytes(), - cell: *cell.as_bytes(), - incarnation: *incarnation.as_bytes(), - cell_epoch, - first_node_sequence, - last_node_sequence, - predecessor_digest: *predecessor.digest.as_bytes(), - predecessor_txid: predecessor.txid, - predecessor_checksum: predecessor.checksum, - predecessor_commit_sequence: predecessor.commit_sequence, - final_txid: final_position.txid, - final_checksum: final_position.checksum, - final_commit_sequence, - bundle_digest: *bundle_digest.as_bytes(), - } - } - - /// Returns the digest of the pinned bundle. - #[must_use] - pub const fn bundle_digest(&self) -> [u8; 32] { - self.bundle_digest - } - - /// Derives a collision-resistant local filename identity from the full key. - #[must_use] - pub fn cache_digest(&self) -> [u8; 32] { - let mut hasher = blake3::Hasher::new(); - hasher.update(b"crab.recovery-artifact.v1\0"); - hasher.update(&self.leader_session); - hasher.update(&self.log_epoch.to_be_bytes()); - hasher.update(&self.application); - hasher.update(&self.cell); - hasher.update(&self.incarnation); - hasher.update(&self.cell_epoch.to_be_bytes()); - hasher.update(&self.first_node_sequence.to_be_bytes()); - hasher.update(&self.last_node_sequence.to_be_bytes()); - hasher.update(&self.predecessor_digest); - hasher.update(&self.predecessor_txid.to_be_bytes()); - hasher.update(&self.predecessor_checksum.to_be_bytes()); - hasher.update(&self.predecessor_commit_sequence.to_be_bytes()); - hasher.update(&self.final_txid.to_be_bytes()); - hasher.update(&self.final_checksum.to_be_bytes()); - hasher.update(&self.final_commit_sequence.to_be_bytes()); - hasher.update(&self.bundle_digest); - *hasher.finalize().as_bytes() - } -} - -/// Keeps a retained local artifact alive while a recovery overlay reads it. -pub struct RecoveryArtifact { - bundle: crab_ltx::bundle::Bundle, - lease: Arc, -} - -impl RecoveryArtifact { - /// Wraps a verified bundle with the lease that keeps its backing artifact alive. - #[must_use] - pub fn new( - bundle: crab_ltx::bundle::Bundle, - lease: Arc, - ) -> Self { - Self { bundle, lease } - } - - fn into_parts( - self, - ) -> ( - crab_ltx::bundle::Bundle, - Arc, - ) { - (self.bundle, self.lease) - } -} - -/// Server-owned cache boundary for verified recovery bundles. -pub trait RecoveryArtifactStore: Send + Sync { - /// Retains a bundle after its immutable object has been published. - fn retain(&self, key: RecoveryArtifactKey, bundle: crab_ltx::bundle::Bundle) -> Result<()>; - - /// Loads and revalidates an exact retained artifact, or reports a miss. - fn load(&self, key: &RecoveryArtifactKey) -> Result>; -} - -/// Immutable object-store owner for recovered follower bundles and manifests. -#[derive(Clone)] -pub struct RecoveryManifestStore { - layout: CellStorageLayout, - limits: crab_ltx::Limits, - recovery_disk: crab_ltx::DiskBudget, - recovery_scratch: Option, - artifact_store: Option>, -} - -impl RecoveryManifestStore { - /// Creates a manifest store over one layout, with recovery disk bounded by - /// the plan byte limit. - #[must_use] - pub fn new(layout: CellStorageLayout, limits: crab_ltx::Limits) -> Self { - let recovery_disk = crab_ltx::DiskBudget::new(limits.max_plan_bytes); - Self { - layout, - limits, - recovery_disk, - recovery_scratch: None, - artifact_store: None, - } - } - - pub(crate) const fn limits(&self) -> crab_ltx::Limits { - self.limits - } - - /// Shares byte admission with node-wide follower-tail recovery work. - #[must_use] - pub fn with_recovery_disk(mut self, recovery_disk: crab_ltx::DiskBudget) -> Self { - self.recovery_disk = recovery_disk; - self - } - - /// Places streamed recovery bundles on the runtime-owned session volume. - /// - /// The directory must already exist, be private to this runtime session, - /// and remain on the same local volume accounted by its disk budget. - /// Library callers that do not provide one use the operating-system - /// temporary directory. - #[must_use] - pub fn with_recovery_scratch(mut self, directory: std::path::PathBuf) -> Self { - self.recovery_scratch = Some(directory); - self - } - - /// Shares one server-owned verified artifact cache with pinning and load. - /// Cache admission is opportunistic; object-store publication remains authoritative. - #[must_use] - pub fn with_recovery_artifacts(mut self, store: Arc) -> Self { - self.artifact_store = Some(store); - self - } - - pub(crate) fn recovery_scratch_directory(&self) -> std::path::PathBuf { - self.recovery_scratch - .clone() - .unwrap_or_else(std::env::temp_dir) - } - - /// Publishes every verified bundle before one content-addressed manifest. - pub async fn pin( - &self, - leader_session: SessionId, - log_epoch: u64, - tails: Vec, - ) -> Result> { - Ok(self - .pin_with_summary(leader_session, log_epoch, tails) - .await? - .cells) - } - - /// Publishes bundles and manifests for the recovered tails and returns the - /// pointers together with their publication work summary. - pub async fn pin_with_summary( - &self, - leader_session: SessionId, - log_epoch: u64, - tails: Vec, - ) -> Result { - if leader_session.as_bytes().iter().all(|byte| *byte == 0) - || log_epoch == 0 - || tails.is_empty() - { - return Err(Error::Node("invalid recovery manifest scope")); - } - let mut rows = Vec::with_capacity(tails.len()); - let mut artifacts = Vec::with_capacity(tails.len()); - let mut summary = RecoveryPublicationSummary::default(); - for tail in tails { - let predecessor = tail.overlay.predecessor(); - let cell = CellId::from_bytes(predecessor.cell); - let incarnation = IncarnationId::from_bytes(predecessor.incarnation); - let runtime_predecessor = RootRef::from_ltx(cell, incarnation, predecessor)?; - let final_position = tail.overlay.final_position(); - let final_commit_sequence = tail.overlay.final_commit_sequence(); - let bundle = tail.overlay.into_bundle()?; - summary.bundle_bytes = summary - .bundle_bytes - .checked_add(bundle.len()) - .ok_or(Error::Capacity("recovery bundle byte count"))?; - let bundle_digest = bundle.digest(); - let path = self.layout.node_log_bundle_path( - leader_session.as_bytes(), - log_epoch, - &bundle_digest, - ); - publish_bundle_immutable(&self.layout, &path, &bundle, &mut summary).await?; - if self.artifact_store.is_some() { - let key = RecoveryArtifactKey::new( - leader_session, - log_epoch, - ApplicationId::from_bytes(tail.application), - cell, - incarnation, - tail.cell_epoch, - tail.first_node_sequence, - tail.last_node_sequence, - runtime_predecessor, - final_position, - final_commit_sequence, - Digest::from_bytes(bundle_digest), - ); - artifacts.push((key, bundle)); - } - rows.push(ManifestCell { - application: tail.application, - cell: predecessor.cell, - incarnation: predecessor.incarnation, - cell_epoch: tail.cell_epoch, - first_node_sequence: tail.first_node_sequence, - last_node_sequence: tail.last_node_sequence, - predecessor, - final_position, - final_commit_sequence, - bundle_digest, - }); - } - rows.sort_unstable_by(|left, right| { - ( - left.application, - left.cell, - left.incarnation, - left.cell_epoch, - ) - .cmp(&( - right.application, - right.cell, - right.incarnation, - right.cell_epoch, - )) - }); - if rows.windows(2).any(|pair| { - pair[0].application == pair[1].application - && pair[0].cell == pair[1].cell - && pair[0].incarnation == pair[1].incarnation - && pair[0].cell_epoch == pair[1].cell_epoch - }) { - return Err(Error::Node( - "recovery manifest contains duplicate Cell scope", - )); - } - let manifest = RecoveryManifest { - leader_session, - log_epoch, - cells: rows, - }; - let body = manifest.encode()?; - let manifest_digest = *blake3::hash(&body).as_bytes(); - let path = self.layout.node_log_recovery_path( - leader_session.as_bytes(), - log_epoch, - &manifest_digest, - ); - publish_immutable(&self.layout, &path, &body, MAX_MANIFEST_BYTES, &mut summary).await?; - if let Some(store) = &self.artifact_store { - for (key, bundle) in artifacts { - let store = Arc::clone(store); - // Both immutable objects are the correctness boundary; a local - // cache admission failure must not block control progress. - let _ = tokio::task::spawn_blocking(move || store.retain(key, bundle)).await; - } - } - Ok(PinnedRecoveryCells { - cells: manifest - .cells - .into_iter() - .map(|cell| PinnedRecoveryCell { - application: ApplicationId::from_bytes(cell.application), - cell: CellId::from_bytes(cell.cell), - incarnation: IncarnationId::from_bytes(cell.incarnation), - cell_epoch: cell.cell_epoch, - recovery: RecoveryOverlayRef { - leader_session, - log_epoch, - manifest_digest: Digest::from_bytes(manifest_digest), - first_node_sequence: cell.first_node_sequence, - last_node_sequence: cell.last_node_sequence, - predecessor: runtime_root(cell.predecessor), - final_txid: cell.final_position.txid, - final_checksum: cell.final_position.checksum, - final_commit_sequence: cell.final_commit_sequence, - }, - }) - .collect(), - summary, - }) - } - - /// Reopens the exact bundle named by a control-pinned recovery reference. - pub async fn load_overlay( - &self, - cell: CellId, - incarnation: IncarnationId, - recovery: &RecoveryOverlayRef, - ) -> Result { - let path = self.layout.node_log_recovery_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - recovery.manifest_digest.as_bytes(), - ); - let (body, _) = self - .layout - .store() - .get_with_etag_bounded(&path, MAX_MANIFEST_BYTES) - .await?; - if *blake3::hash(&body).as_bytes() != *recovery.manifest_digest.as_bytes() { - return Err(Error::Node("recovery manifest digest differs")); - } - let manifest = RecoveryManifest::decode(&body)?; - if manifest.leader_session != recovery.leader_session - || manifest.log_epoch != recovery.log_epoch - { - return Err(Error::Node("recovery manifest path scope differs")); - } - let row = manifest - .cells - .into_iter() - .find(|row| { - row.application == *self.layout.application_id() - && row.cell == *cell.as_bytes() - && row.incarnation == *incarnation.as_bytes() - }) - .ok_or(Error::Node("recovery manifest does not contain Cell"))?; - let expected = RecoveryOverlayRef { - leader_session: recovery.leader_session, - log_epoch: recovery.log_epoch, - manifest_digest: recovery.manifest_digest, - first_node_sequence: row.first_node_sequence, - last_node_sequence: row.last_node_sequence, - predecessor: runtime_root(row.predecessor), - final_txid: row.final_position.txid, - final_checksum: row.final_position.checksum, - final_commit_sequence: row.final_commit_sequence, - }; - if &expected != recovery { - return Err(Error::Node( - "recovery control pointer differs from manifest", - )); - } - if let Some(artifacts) = &self.artifact_store { - let runtime_predecessor = RootRef::from_ltx(cell, incarnation, row.predecessor)?; - let key = RecoveryArtifactKey::new( - recovery.leader_session, - recovery.log_epoch, - ApplicationId::from_bytes(*self.layout.application_id()), - cell, - incarnation, - row.cell_epoch, - row.first_node_sequence, - row.last_node_sequence, - runtime_predecessor, - row.final_position, - row.final_commit_sequence, - Digest::from_bytes(row.bundle_digest), - ); - let artifacts = Arc::clone(artifacts); - let cached = tokio::task::spawn_blocking(move || artifacts.load(&key)) - .await - .map_err(|_| Error::Node("recovery artifact worker failed"))?; - if let Ok(Some(artifact)) = cached { - let (bundle, lease) = artifact.into_parts(); - return Ok(crab_ltx::RecoveryOverlay::new( - row.predecessor, - bundle, - row.final_position, - row.final_commit_sequence, - ) - .with_bundle_lease(lease)); - } - } - let bundle_path = self.layout.node_log_bundle_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - &row.bundle_digest, - ); - let metadata = self.layout.store().head(&bundle_path).await?; - if metadata.size > self.limits.max_plan_bytes { - return Err(Error::Node("recovery bundle exceeds limit")); - } - let disk_reservation = self.recovery_disk.try_reserve(metadata.size)?; - let temporary = match &self.recovery_scratch { - Some(directory) => tempfile::Builder::new() - .prefix(".crab-recovery-") - .tempfile_in(directory)?, - None => tempfile::NamedTempFile::new()?, - }; - let temporary = temporary.into_temp_path(); - self.layout - .store() - .download_to_path_bounded(&bundle_path, temporary.as_ref(), self.limits.max_plan_bytes) - .await?; - let limits = self.limits; - let decoded = tokio::task::spawn_blocking(move || { - crab_ltx::bundle::Bundle::decode_temp_file_with_digest( - temporary, - row.bundle_digest, - limits, - ) - }) - .await - .map_err(crab_ltx::CrabError::from)?; - let bundle = match decoded { - Ok(bundle) => bundle, - Err(crab_ltx::CrabError::ChecksumMismatch) => { - return Err(Error::Node("recovery bundle digest differs")); - } - Err(error) => return Err(error.into()), - }; - Ok(crab_ltx::RecoveryOverlay::new( - row.predecessor, - bundle, - row.final_position, - row.final_commit_sequence, - ) - .with_disk_reservation(disk_reservation)) - } -} - -struct RecoveryManifest { - leader_session: SessionId, - log_epoch: u64, - cells: Vec, -} - -struct ManifestCell { - application: [u8; 16], - cell: [u8; 32], - incarnation: [u8; 16], - cell_epoch: u64, - first_node_sequence: u64, - last_node_sequence: u64, - predecessor: crab_ltx::RootRef, - final_position: crab_ltx::Position, - final_commit_sequence: u64, - bundle_digest: [u8; 32], -} - -impl RecoveryManifest { - fn encode(&self) -> Result> { - let body = serde_json::to_vec(&RawManifest::from(self))?; - if body.len() as u64 > MAX_MANIFEST_BYTES { - return Err(Error::Node("recovery manifest exceeds limit")); - } - Ok(body) - } - - fn decode(body: &[u8]) -> Result { - if body.len() as u64 > MAX_MANIFEST_BYTES { - return Err(Error::Node("recovery manifest exceeds limit")); - } - let raw: RawManifest = serde_json::from_slice(body)?; - let manifest = Self::try_from(raw)?; - if manifest.encode()? != body { - return Err(Error::Node("recovery manifest is not canonical")); - } - Ok(manifest) - } -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct RawManifest { - version: u32, - leader_session: String, - log_epoch: String, - cells: Vec, -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct RawManifestCell { - application: String, - cell: String, - incarnation: String, - cell_epoch: String, - first_node_sequence: String, - last_node_sequence: String, - predecessor_digest: String, - predecessor_txid: String, - predecessor_checksum: String, - predecessor_commit_sequence: String, - final_txid: String, - final_checksum: String, - final_commit_sequence: String, - bundle_digest: String, -} - -impl From<&RecoveryManifest> for RawManifest { - fn from(manifest: &RecoveryManifest) -> Self { - Self { - version: 1, - leader_session: encode_hex(manifest.leader_session.as_bytes()), - log_epoch: manifest.log_epoch.to_string(), - cells: manifest - .cells - .iter() - .map(|cell| RawManifestCell { - application: encode_hex(&cell.application), - cell: encode_hex(&cell.cell), - incarnation: encode_hex(&cell.incarnation), - cell_epoch: cell.cell_epoch.to_string(), - first_node_sequence: cell.first_node_sequence.to_string(), - last_node_sequence: cell.last_node_sequence.to_string(), - predecessor_digest: encode_hex(&cell.predecessor.digest), - predecessor_txid: cell.predecessor.position.txid.to_string(), - predecessor_checksum: encode_hex( - &cell.predecessor.position.checksum.to_be_bytes(), - ), - predecessor_commit_sequence: cell.predecessor.commit_sequence.to_string(), - final_txid: cell.final_position.txid.to_string(), - final_checksum: encode_hex(&cell.final_position.checksum.to_be_bytes()), - final_commit_sequence: cell.final_commit_sequence.to_string(), - bundle_digest: encode_hex(&cell.bundle_digest), - }) - .collect(), - } - } -} - -impl TryFrom for RecoveryManifest { - type Error = Error; - - fn try_from(raw: RawManifest) -> Result { - if raw.version != 1 || raw.cells.is_empty() || raw.cells.len() > 1_024 { - return Err(Error::Node("invalid recovery manifest shape")); - } - let leader_session = SessionId::from_bytes(unhex(&raw.leader_session)?); - let log_epoch = decimal(&raw.log_epoch)?; - let mut cells = Vec::with_capacity(raw.cells.len()); - for cell in raw.cells { - let cell_id = unhex(&cell.cell)?; - let incarnation = unhex(&cell.incarnation)?; - cells.push(ManifestCell { - application: unhex(&cell.application)?, - cell: cell_id, - incarnation, - cell_epoch: decimal(&cell.cell_epoch)?, - first_node_sequence: decimal(&cell.first_node_sequence)?, - last_node_sequence: decimal(&cell.last_node_sequence)?, - predecessor: crab_ltx::RootRef { - cell: cell_id, - incarnation, - digest: unhex(&cell.predecessor_digest)?, - position: crab_ltx::Position { - txid: decimal(&cell.predecessor_txid)?, - checksum: u64::from_be_bytes(unhex(&cell.predecessor_checksum)?), - }, - commit_sequence: decimal(&cell.predecessor_commit_sequence)?, - }, - final_position: crab_ltx::Position { - txid: decimal(&cell.final_txid)?, - checksum: u64::from_be_bytes(unhex(&cell.final_checksum)?), - }, - final_commit_sequence: decimal(&cell.final_commit_sequence)?, - bundle_digest: unhex(&cell.bundle_digest)?, - }); - } - let manifest = Self { - leader_session, - log_epoch, - cells, - }; - if leader_session.as_bytes().iter().all(|byte| *byte == 0) - || log_epoch == 0 - || manifest.cells.iter().any(|cell| { - cell.cell_epoch == 0 - || cell.first_node_sequence == 0 - || cell.first_node_sequence > cell.last_node_sequence - || cell.final_position.txid <= cell.predecessor.position.txid - || cell.final_commit_sequence <= cell.predecessor.commit_sequence - }) - { - return Err(Error::Node("invalid recovery manifest values")); - } - Ok(manifest) - } -} - -async fn publish_immutable( - layout: &CellStorageLayout, - path: &object_store::path::Path, - body: &[u8], - limit: u64, - summary: &mut RecoveryPublicationSummary, -) -> Result<()> { - if body.len() as u64 > limit { - return Err(Error::Node("recovery object exceeds limit")); - } - summary.object_writes = summary - .object_writes - .checked_add(1) - .ok_or(Error::Capacity("recovery object write count"))?; - match layout - .store() - .create_strict(path, Bytes::copy_from_slice(body)) - .await - { - Ok(()) => Ok(()), - Err(create_error) => { - summary.object_reads = summary - .object_reads - .checked_add(1) - .ok_or(Error::Capacity("recovery object read count"))?; - match layout.store().get_with_etag_bounded(path, limit).await { - Ok((existing, _)) if existing.as_ref() == body => Ok(()), - Ok(_) => Err(Error::Node("recovery digest path contains different bytes")), - Err(StorageError::NotFound { .. }) => Err(create_error.into()), - Err(error) => Err(error.into()), - } - } - } -} - -async fn publish_bundle_immutable( - layout: &CellStorageLayout, - path: &object_store::path::Path, - bundle: &crab_ltx::bundle::Bundle, - summary: &mut RecoveryPublicationSummary, -) -> Result<()> { - let store = layout.store(); - let digest = bundle.digest(); - let size = bundle.len(); - summary.object_reads = summary - .object_reads - .checked_add(1) - .ok_or(Error::Capacity("recovery object read count"))?; - match store.verify_size_and_hash(path, size, &digest).await { - Ok(()) => return Ok(()), - Err(StorageError::NotFound { .. }) => {} - Err(StorageError::CorruptObject { .. }) => { - return Err(Error::Node("recovery bundle path contains different bytes")); - } - Err(error) => return Err(error.into()), - } - - let staged = object_store::path::Path::from(format!("{path}.staging")); - let cancel = tokio_util::sync::CancellationToken::new(); - let upload = store - .put_multipart_source_retry( - &staged, - bundle.upload_source(), - size, - digest, - MULTIPART_BYTES, - &cancel, - None, - ) - .await; - summary.object_writes = summary - .object_writes - .checked_add(2) - .ok_or(Error::Capacity("recovery object write count"))?; - if let Err(error) = upload { - return match cleanup_staged(store, &staged).await { - Ok(()) => Err(error.into()), - Err(cleanup_error) => Err(cleanup_error), - }; - } - let promotion = store - .promote_staged_content_addressed_object(&staged, path, digest, size) - .await; - match cleanup_staged(store, &staged).await { - Err(error) => Err(error), - Ok(()) => promotion.map(|_| ()).map_err(Into::into), - } -} - -async fn cleanup_staged( - store: &crab_storage::Store, - path: &object_store::path::Path, -) -> Result<()> { - match store.delete(path).await { - Ok(()) | Err(StorageError::NotFound { .. }) => Ok(()), - Err(error) => Err(error.into()), - } -} - -fn runtime_root(root: crab_ltx::RootRef) -> RootRef { - RootRef { - digest: Digest::from_bytes(root.digest), - txid: root.position.txid, - checksum: root.position.checksum, - commit_sequence: root.commit_sequence, - } -} - -fn decimal(value: &str) -> Result { - let parsed = value - .parse::() - .map_err(|_| Error::Node("invalid recovery manifest decimal"))?; - if parsed.to_string() != value { - return Err(Error::Node("noncanonical recovery manifest decimal")); - } - Ok(parsed) -} - -fn unhex(value: &str) -> Result<[u8; N]> { - if value.len() != N * 2 { - return Err(Error::Node("invalid recovery manifest hex length")); - } - let mut decoded = [0_u8; N]; - for (index, pair) in value.as_bytes().as_chunks::<2>().0.iter().enumerate() { - decoded[index] = (nibble(pair[0])? << 4) | nibble(pair[1])?; - } - Ok(decoded) -} - -fn nibble(value: u8) -> Result { - match value { - b'0'..=b'9' => Ok(value - b'0'), - b'a'..=b'f' => Ok(value - b'a' + 10), - _ => Err(Error::Node("invalid recovery manifest hex")), - } -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/recovery/manifest/tests.rs b/crates/crab-cell-runtime/src/recovery/manifest/tests.rs deleted file mode 100644 index bdbdbede4..000000000 --- a/crates/crab-cell-runtime/src/recovery/manifest/tests.rs +++ /dev/null @@ -1,405 +0,0 @@ -use std::{ - collections::BTreeMap, - sync::{ - Arc, Mutex, - atomic::{AtomicU64, Ordering}, - }, -}; - -use object_store::{ObjectStoreExt, memory::InMemory, path::Path}; - -use super::*; -use crate::node::log::{RecoveryBase, build_recovery_overlays}; - -struct RecoveryFixture { - inner: Arc, - layout: CellStorageLayout, - replica: crab_ltx::CellReplica, - manifests: RecoveryManifestStore, - pinned: PinnedRecoveryCell, - publication: RecoveryPublicationSummary, - base: crab_ltx::RootRef, - final_position: crab_ltx::Position, -} - -async fn recovery_fixture() -> RecoveryFixture { - recovery_fixture_with_store(None).await -} - -struct MemoryArtifactStore { - limits: crab_ltx::Limits, - reject_retain: bool, - bundles: Mutex>>, - loads: AtomicU64, -} - -struct MemoryArtifactLease; - -impl crab_ltx::bundle::BundleLease for MemoryArtifactLease {} - -impl RecoveryArtifactStore for MemoryArtifactStore { - fn retain(&self, key: RecoveryArtifactKey, bundle: crab_ltx::bundle::Bundle) -> Result<()> { - if self.reject_retain { - return Err(Error::Capacity("artifact test store")); - } - let bytes = bundle.read_all()?; - self.bundles - .lock() - .map_err(|_| Error::Node("artifact test store lock poisoned"))? - .insert(key, bytes.to_vec()); - Ok(()) - } - - fn load(&self, key: &RecoveryArtifactKey) -> Result> { - let bytes = self - .bundles - .lock() - .map_err(|_| Error::Node("artifact test store lock poisoned"))? - .get(key) - .cloned(); - let Some(bytes) = bytes else { - return Ok(None); - }; - self.loads.fetch_add(1, Ordering::Relaxed); - let bundle = crab_ltx::bundle::Bundle::decode(bytes, self.limits)?; - Ok(Some(RecoveryArtifact::new( - bundle, - Arc::new(MemoryArtifactLease), - ))) - } -} - -async fn recovery_fixture_with_store( - artifacts: Option>, -) -> RecoveryFixture { - let limits = crab_ltx::Limits::default(); - let directory = tempfile::TempDir::new().unwrap(); - let mut database = crab_ltx::Db::open(&directory.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let first = database.capture().unwrap(); - let inner = Arc::new(InMemory::new()); - let store = crab_storage::Store::new(inner.clone()); - let application = [3; 16]; - let cell = [4; 32]; - let incarnation = [5; 16]; - let layout = CellStorageLayout::new(store, Path::from("root"), application); - let replica = crab_ltx::CellReplica::new(layout.clone(), cell, incarnation, limits).unwrap(); - let base = replica.prepare(None, &first, 1, 1).await.unwrap().root(); - database - .transaction(|transaction| { - transaction.execute("INSERT INTO values_ VALUES (1)", [])?; - Ok(()) - }) - .unwrap(); - let tail = database.capture().unwrap(); - let segment = tail.segments.first().unwrap(); - let frame = crab_ltx::encode_node_frame( - crab_ltx::NodeFrameScope { - leader_session: [1; 16], - log_epoch: 2, - node_sequence: 1, - application, - cell, - incarnation, - cell_epoch: 6, - commit_sequence: 2, - }, - segment.info().clone(), - Bytes::from(std::fs::read(segment.path()).unwrap()), - limits, - ) - .unwrap(); - let recovered = build_recovery_overlays( - vec![frame], - &[RecoveryBase { - application, - cell_epoch: 6, - root: base, - }], - limits, - ) - .unwrap(); - let manifests = RecoveryManifestStore::new(layout.clone(), limits); - let manifests = artifacts.map_or(manifests.clone(), |store| { - manifests.with_recovery_artifacts(store) - }); - let pinned = manifests - .pin_with_summary(SessionId::from_bytes([1; 16]), 2, recovered) - .await - .unwrap(); - let publication = pinned.summary; - let mut pinned = pinned.cells; - database.close().unwrap(); - RecoveryFixture { - inner, - layout, - replica, - manifests, - pinned: pinned.pop().unwrap(), - publication, - base, - final_position: tail.position, - } -} - -async fn load_error(fixture: &RecoveryFixture, recovery: &RecoveryOverlayRef) -> Error { - match fixture - .manifests - .load_overlay(fixture.pinned.cell, fixture.pinned.incarnation, recovery) - .await - { - Ok(_) => panic!("corrupt recovery input must not load"), - Err(error) => error, - } -} - -#[tokio::test] -async fn pinned_manifest_reopens_exact_overlay_and_prepares_successor() { - let fixture = recovery_fixture().await; - assert!(fixture.publication.bundle_bytes > 0); - assert!(fixture.publication.object_reads > 0); - assert!(fixture.publication.object_writes > 0); - let overlay = fixture - .manifests - .load_overlay( - fixture.pinned.cell, - fixture.pinned.incarnation, - &fixture.pinned.recovery, - ) - .await - .unwrap(); - let prepared = fixture - .replica - .prepare_recovered_overlay(&overlay, 1) - .await - .unwrap(); - assert_eq!(prepared.predecessor(), Some(fixture.base)); - assert_eq!(prepared.root().position, fixture.final_position); - let mut control = crate::control::Control::initial( - fixture.pinned.cell, - fixture.pinned.incarnation, - crate::control::Owner { - session: SessionId::from_bytes([1; 16]), - endpoint: "https://dead.internal:8081".into(), - }, - Digest::from_bytes([12; 32]), - 1, - ) - .unwrap(); - control.state = crate::control::ControlState::Serving; - control.root = Some(runtime_root(fixture.base)); - let attached = control - .attach_recovery(fixture.pinned.recovery.clone()) - .unwrap(); - let takeover = attached - .takeover(crate::control::Owner { - session: SessionId::from_bytes([13; 16]), - endpoint: "https://successor.internal:8081".into(), - }) - .unwrap(); - let published = takeover.publish_recovery(&prepared, None).unwrap(); - assert_eq!(published.state, crate::control::ControlState::Recovering); - assert!(published.recovery.is_none()); - assert_eq!(published.root.unwrap().txid, fixture.final_position.txid); -} - -#[tokio::test] -async fn verified_artifact_store_hit_reuses_the_pinned_bundle() { - let limits = crab_ltx::Limits::default(); - let artifacts = Arc::new(MemoryArtifactStore { - limits, - reject_retain: false, - bundles: Mutex::new(BTreeMap::new()), - loads: AtomicU64::new(0), - }); - let fixture = recovery_fixture_with_store(Some( - Arc::clone(&artifacts) as Arc - )) - .await; - let overlay = fixture - .manifests - .load_overlay( - fixture.pinned.cell, - fixture.pinned.incarnation, - &fixture.pinned.recovery, - ) - .await - .unwrap(); - assert_eq!(artifacts.loads.load(Ordering::Relaxed), 1); - let prepared = fixture - .replica - .prepare_recovered_overlay(&overlay, 1) - .await - .unwrap(); - assert_eq!(prepared.root().position, fixture.final_position); -} - -#[tokio::test] -async fn artifact_cache_failure_keeps_object_store_recovery_available() { - let artifacts = Arc::new(MemoryArtifactStore { - limits: crab_ltx::Limits::default(), - reject_retain: true, - bundles: Mutex::new(BTreeMap::new()), - loads: AtomicU64::new(0), - }); - let fixture = recovery_fixture_with_store(Some( - Arc::clone(&artifacts) as Arc - )) - .await; - let overlay = fixture - .manifests - .load_overlay( - fixture.pinned.cell, - fixture.pinned.incarnation, - &fixture.pinned.recovery, - ) - .await - .unwrap(); - assert_eq!(artifacts.loads.load(Ordering::Relaxed), 0); - assert_eq!(overlay.final_position(), fixture.final_position); -} - -#[tokio::test] -async fn loaded_overlay_holds_bundle_disk_reservation_until_drop() { - let fixture = recovery_fixture().await; - let scratch = tempfile::TempDir::new().unwrap(); - let recovery = &fixture.pinned.recovery; - let manifest_path = fixture.layout.node_log_recovery_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - recovery.manifest_digest.as_bytes(), - ); - let (body, _) = fixture - .layout - .store() - .get_with_etag_bounded(&manifest_path, MAX_MANIFEST_BYTES) - .await - .unwrap(); - let manifest = RecoveryManifest::decode(&body).unwrap(); - let bundle_path = fixture.layout.node_log_bundle_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - &manifest.cells[0].bundle_digest, - ); - let size = fixture - .layout - .store() - .head(&bundle_path) - .await - .unwrap() - .size; - let budget = crab_ltx::DiskBudget::new(size); - let manifests = RecoveryManifestStore::new(fixture.layout.clone(), crab_ltx::Limits::default()) - .with_recovery_disk(budget.clone()) - .with_recovery_scratch(scratch.path().to_owned()); - let overlay = manifests - .load_overlay(fixture.pinned.cell, fixture.pinned.incarnation, recovery) - .await - .unwrap(); - assert_eq!(overlay.bundle().len(), size); - assert_eq!(budget.used(), size); - assert_eq!(std::fs::read_dir(scratch.path()).unwrap().count(), 1); - drop(overlay); - assert_eq!(budget.used(), 0); - assert_eq!(std::fs::read_dir(scratch.path()).unwrap().count(), 0); -} - -#[tokio::test] -async fn load_overlay_rejects_corrupt_manifest_bytes() { - let fixture = recovery_fixture().await; - let recovery = &fixture.pinned.recovery; - let path = fixture.layout.node_log_recovery_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - recovery.manifest_digest.as_bytes(), - ); - fixture - .inner - .put(&path, Bytes::from_static(b"corrupt manifest").into()) - .await - .unwrap(); - - let error = load_error(&fixture, recovery).await; - - assert!(matches!( - error, - Error::Node("recovery manifest digest differs") - )); -} - -#[tokio::test] -async fn load_overlay_rejects_self_consistent_manifest_metadata_change() { - let fixture = recovery_fixture().await; - let mut recovery = fixture.pinned.recovery.clone(); - let original_path = fixture.layout.node_log_recovery_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - recovery.manifest_digest.as_bytes(), - ); - let (body, _) = fixture - .layout - .store() - .get_with_etag_bounded(&original_path, MAX_MANIFEST_BYTES) - .await - .unwrap(); - let mut raw: RawManifest = serde_json::from_slice(&body).unwrap(); - raw.cells[0].final_commit_sequence = "3".into(); - let changed = serde_json::to_vec(&raw).unwrap(); - let digest = *blake3::hash(&changed).as_bytes(); - let changed_path = fixture.layout.node_log_recovery_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - &digest, - ); - fixture - .layout - .store() - .put(&changed_path, Bytes::from(changed)) - .await - .unwrap(); - recovery.manifest_digest = Digest::from_bytes(digest); - - let error = load_error(&fixture, &recovery).await; - - assert!(matches!( - error, - Error::Node("recovery control pointer differs from manifest") - )); -} - -#[tokio::test] -async fn load_overlay_rejects_corrupt_bundle_bytes() { - let fixture = recovery_fixture().await; - let recovery = &fixture.pinned.recovery; - let manifest_path = fixture.layout.node_log_recovery_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - recovery.manifest_digest.as_bytes(), - ); - let (body, _) = fixture - .layout - .store() - .get_with_etag_bounded(&manifest_path, MAX_MANIFEST_BYTES) - .await - .unwrap(); - let manifest = RecoveryManifest::decode(&body).unwrap(); - let bundle_path = fixture.layout.node_log_bundle_path( - recovery.leader_session.as_bytes(), - recovery.log_epoch, - &manifest.cells[0].bundle_digest, - ); - fixture - .inner - .put(&bundle_path, Bytes::from_static(b"corrupt bundle").into()) - .await - .unwrap(); - - let error = load_error(&fixture, recovery).await; - - assert!(matches!( - error, - Error::Node("recovery bundle digest differs") - )); -} diff --git a/crates/crab-cell-runtime/src/recovery/release.rs b/crates/crab-cell-runtime/src/recovery/release.rs deleted file mode 100644 index 66caf5602..000000000 --- a/crates/crab-cell-runtime/src/recovery/release.rs +++ /dev/null @@ -1,723 +0,0 @@ -//! Release records for one application's operator transitions. -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::{ETag, StorageError}; -use serde::{Deserialize, Serialize}; - -use crate::cell::application::ApplicationIdentity; -use crate::cell::catalog::{CatalogEntry, CatalogProof, CellCatalog}; -use crate::identity::Digest; -use crate::identity::RequestId; -use crate::identity::encode_hex; -use crate::registry::Registry; -use crate::{Error, Result}; - -const MAX_RELEASE_BYTES: u64 = 8 * 1024; -const MAX_DESCRIPTOR_BYTES: u64 = 256 * 1024; - -/// Durable rollout phase for one compiled application release. -#[derive(Clone, Copy, Debug, Deserialize, Serialize, PartialEq, Eq)] -#[serde(rename_all = "snake_case")] -pub enum ReleaseState { - /// No release is being rolled out. - Ready, - /// A release is published but not yet selected. - Prepared, - /// The application is held for release maintenance. - Maintenance, - /// Cells are migrating to the desired release. - Activating, - /// The rollout failed and needs operator action. - Failed, -} - -/// Canonical mutable release selection for one application. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct ReleaseRecord { - application: crate::ApplicationId, - revision: u64, - current: Option, - desired: Option, - desired_image: String, - operation: RequestId, - state: ReleaseState, -} - -impl ReleaseRecord { - /// Returns the record revision. - #[must_use] - pub const fn revision(&self) -> u64 { - self.revision - } - - /// Returns the release currently selected for the application. - #[must_use] - pub const fn current(&self) -> Option { - self.current - } - - /// Returns the release being rolled out, when one is. - #[must_use] - pub const fn desired(&self) -> Option { - self.desired - } - - /// Returns the desired release's image name. - #[must_use] - pub fn desired_image(&self) -> &str { - &self.desired_image - } - - /// Returns the request that started the rollout. - #[must_use] - pub const fn operation(&self) -> RequestId { - self.operation - } - - /// Returns the durable rollout phase. - #[must_use] - pub const fn state(&self) -> ReleaseState { - self.state - } - - /// Encodes the exact canonical JSON body stored and printed by administration. - pub fn encode(&self) -> Result> { - let bytes = serde_json::to_vec(&RawRelease { - application: encode_hex(self.application.as_bytes()), - current: self.current.map(|value| encode_hex(value.as_bytes())), - desired: self.desired.map(|value| encode_hex(value.as_bytes())), - desired_image: self.desired_image.clone(), - operation: encode_hex(self.operation.as_bytes()), - revision: self.revision.to_string(), - state: self.state, - version: 1, - })?; - if bytes.len() as u64 > MAX_RELEASE_BYTES { - return Err(Error::Release("release record exceeds 8 KiB")); - } - Ok(bytes) - } - - pub(crate) fn decode(bytes: &[u8], identity: ApplicationIdentity) -> Result { - let raw: RawRelease = serde_json::from_slice(bytes)?; - if raw.version != 1 { - return Err(Error::Release("unsupported release version")); - } - let record = Self { - application: crate::ApplicationId::from_bytes(parse_hex(&raw.application)?), - revision: parse_revision(&raw.revision)?, - current: raw.current.as_deref().map(parse_digest).transpose()?, - desired: raw.desired.as_deref().map(parse_digest).transpose()?, - desired_image: raw.desired_image, - operation: RequestId::from_bytes(parse_hex(&raw.operation)?), - state: raw.state, - }; - record.validate(identity)?; - if record.encode()?.as_slice() != bytes { - return Err(Error::Release("release record is not canonical")); - } - Ok(record) - } - - fn validate(&self, identity: ApplicationIdentity) -> Result<()> { - if self.application != identity.application() || self.revision == 0 { - return Err(Error::Release("release identity or revision is invalid")); - } - if self.operation.as_bytes().iter().all(|byte| *byte == 0) - || self - .current - .is_some_and(|digest| digest.as_bytes().iter().all(|byte| *byte == 0)) - || self - .desired - .is_some_and(|digest| digest.as_bytes().iter().all(|byte| *byte == 0)) - { - return Err(Error::Release("release IDs and digests must be nonzero")); - } - validate_image(&self.desired_image)?; - if matches!( - self.state, - ReleaseState::Prepared | ReleaseState::Maintenance | ReleaseState::Activating - ) && self.desired.is_none() - { - return Err(Error::Release( - "release phase requires a desired descriptor", - )); - } - if self.state == ReleaseState::Ready - && (self.current.is_none() || self.current != self.desired) - { - return Err(Error::Release( - "ready release requires one current desired descriptor", - )); - } - Ok(()) - } -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct RawRelease { - application: String, - current: Option, - desired: Option, - desired_image: String, - operation: String, - revision: String, - state: ReleaseState, - version: u8, -} - -/// One exact release observation and its conditional-write token. -pub struct VersionedRelease { - record: ReleaseRecord, - token: ETag, -} - -impl VersionedRelease { - /// Returns the release record this version carries. - #[must_use] - pub const fn record(&self) -> &ReleaseRecord { - &self.record - } -} - -/// Publishes immutable descriptors before selecting them through release CAS. -#[derive(Clone)] -pub struct ReleaseStore { - layout: CellStorageLayout, - identity: ApplicationIdentity, -} - -impl ReleaseStore { - /// Creates a release store after checking that the layout belongs to the - /// same application as the identity. - pub fn new(layout: CellStorageLayout, identity: ApplicationIdentity) -> Result { - if layout.application_id() != identity.application().as_bytes() { - return Err(Error::Release("layout and application identity differ")); - } - Ok(Self { layout, identity }) - } - - /// Loads and verifies the current release selection. - pub async fn load(&self) -> Result> { - let (bytes, token) = match self - .layout - .store() - .get_with_etag_bounded(&self.layout.release_path(), MAX_RELEASE_BYTES) - .await - { - Ok(value) => value, - Err(StorageError::NotFound { .. }) => return Ok(None), - Err(error) => return Err(error.into()), - }; - Ok(Some(VersionedRelease { - record: ReleaseRecord::decode(&bytes, self.identity)?, - token, - })) - } - - pub(crate) async fn install_restored(&self, record: ReleaseRecord) -> Result { - record.validate(self.identity)?; - let path = self.layout.release_path(); - match self - .layout - .store() - .create_strict_with_etag(&path, Bytes::from(record.encode()?)) - .await - { - Ok(_) => Ok(record), - Err(create_error) => match self.load().await? { - Some(current) if current.record == record => Ok(current.record), - Some(_) => Err(Error::Release( - "restored release conflicts with existing selection", - )), - None => Err(create_error.into()), - }, - } - } - - /// Uploads one compiled descriptor and conditionally selects it as prepared. - pub async fn prepare( - &self, - descriptor: &[u8], - digest: Digest, - expected_revision: u64, - image: &str, - operation: RequestId, - ) -> Result { - if descriptor.is_empty() - || descriptor.len() as u64 > MAX_DESCRIPTOR_BYTES - || blake3::hash(descriptor).as_bytes() != digest.as_bytes() - { - return Err(Error::Release( - "compiled descriptor digest or size is invalid", - )); - } - validate_image(image)?; - self.publish_descriptor(descriptor, digest).await?; - - let observed = self.load().await?; - if let Some(observed) = &observed - && prepared_matches( - &observed.record, - digest, - expected_revision, - image, - operation, - ) - { - return Ok(observed.record.clone()); - } - if observed.as_ref().map_or(0, |value| value.record.revision) != expected_revision { - return Err(Error::Release("release revision changed concurrently")); - } - if observed.as_ref().is_some_and(|value| { - matches!( - value.record.state, - ReleaseState::Activating | ReleaseState::Maintenance - ) - }) { - return Err(Error::Release("release activation is already in progress")); - } - let next = ReleaseRecord { - application: self.identity.application(), - revision: expected_revision - .checked_add(1) - .ok_or(Error::Release("release revision overflow"))?, - current: observed.as_ref().and_then(|value| value.record.current), - desired: Some(digest), - desired_image: image.to_owned(), - operation, - state: ReleaseState::Prepared, - }; - next.validate(self.identity)?; - let write = match observed { - Some(observed) => { - self.layout - .store() - .update( - &self.layout.release_path(), - Bytes::from(next.encode()?), - observed.token, - ) - .await - } - None => { - self.layout - .store() - .create_strict_with_etag( - &self.layout.release_path(), - Bytes::from(next.encode()?), - ) - .await - } - }; - match write { - Ok(_) => Ok(next), - Err(write_error) => match self.load().await? { - Some(current) - if current.record.revision == next.revision - && current.record.current == next.current - && current.record.desired == next.desired - && current.record.desired_image == next.desired_image - && current.record.operation == next.operation - && current.record.state == next.state => - { - Ok(current.record) - } - Some(_) => Err(Error::Release("release changed concurrently")), - None => Err(write_error.into()), - }, - } - } - - /// Loads and verifies one immutable descriptor selected by release state. - pub async fn descriptor(&self, digest: Digest) -> Result> { - let path = self.layout.release_descriptor_path(digest.as_bytes()); - let (descriptor, _) = self - .layout - .store() - .get_with_etag_bounded(&path, MAX_DESCRIPTOR_BYTES) - .await?; - if descriptor.is_empty() || blake3::hash(&descriptor).as_bytes() != digest.as_bytes() { - return Err(Error::Release( - "stored descriptor digest or size is invalid", - )); - } - Ok(descriptor.to_vec()) - } - - /// Publishes one catalog entry under an exact ready or activating release. - /// - /// A failed post-publication recheck leaves the immutable catalog entry - /// visible, so a later activation must admit it before becoming ready. - pub async fn provision( - &self, - catalog: &CellCatalog, - registry: &Registry, - entry: CatalogEntry, - ) -> Result { - if !catalog.matches_identity(self.identity) { - return Err(Error::Release( - "catalog and release application identities differ", - )); - } - let before = self - .load() - .await? - .ok_or(Error::Release("release is not ready for provisioning"))? - .record; - let selected = selected_provision_release(&before)?; - if selected != registry.release_digest() - || self.descriptor(selected).await? != registry.release_bytes() - || !registry.is_current_cell( - entry.namespace(), - entry.role(), - entry.initial_code(), - entry.initial_schema(), - ) - { - return Err(Error::Release( - "catalog entry does not use the current release code and schema", - )); - } - - let proof = catalog.provision(entry).await?; - let after = self - .load() - .await? - .ok_or(Error::Release("release disappeared during provisioning"))? - .record; - if !provision_release_continues(&before, &after, selected) { - return Err(Error::Release("release changed during provisioning")); - } - Ok(proof) - } - - /// CASes one prepared release into its resumable activation phase. - /// - /// The caller must complete fleet and Cell compatibility admission before - /// calling `complete_activation`; this storage owner only serializes phases. - pub async fn start_activation( - &self, - expected_revision: u64, - operation: RequestId, - ) -> Result { - let observed = self - .load() - .await? - .ok_or(Error::Release("release is not prepared"))?; - if activation_retry(&observed.record, expected_revision, operation) { - let desired = observed - .record - .desired - .ok_or(Error::Release("release activation has no descriptor"))?; - self.descriptor(desired).await?; - return Ok(observed.record); - } - if observed.record.revision != expected_revision - || observed.record.state != ReleaseState::Prepared - || observed.record.operation != operation - { - return Err(Error::Release("prepared release changed concurrently")); - } - let desired = observed - .record - .desired - .ok_or(Error::Release("prepared release has no descriptor"))?; - self.descriptor(desired).await?; - let mut next = observed.record.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Release("release revision overflow"))?; - next.state = ReleaseState::Activating; - self.update_exact(observed, next).await - } - - /// Enters the operation-bound offline maintenance phase. - /// - /// This transition only closes release admission. The caller must drain every - /// writer and prove the advertised fleet empty before changing Cell data. - pub async fn start_maintenance( - &self, - expected_revision: u64, - operation: RequestId, - ) -> Result { - let observed = self - .load() - .await? - .ok_or(Error::Release("release is not prepared"))?; - if maintenance_retry(&observed.record, expected_revision, operation) { - let desired = observed - .record - .desired - .ok_or(Error::Release("maintenance release has no descriptor"))?; - self.descriptor(desired).await?; - return Ok(observed.record); - } - if observed.record.revision != expected_revision - || observed.record.state != ReleaseState::Prepared - || observed.record.operation != operation - { - return Err(Error::Release("prepared release changed concurrently")); - } - let desired = observed - .record - .desired - .ok_or(Error::Release("prepared release has no descriptor"))?; - self.descriptor(desired).await?; - let mut next = observed.record.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Release("release revision overflow"))?; - next.state = ReleaseState::Maintenance; - self.update_exact(observed, next).await - } - - /// Publishes the desired descriptor as current after caller-side admission. - pub async fn complete_activation( - &self, - expected_revision: u64, - operation: RequestId, - ) -> Result { - self.complete( - expected_revision, - operation, - ReleaseState::Activating, - "release is not activating", - "activating release changed concurrently", - ) - .await - } - - /// Publishes the desired maintenance release after offline admission succeeds. - pub async fn complete_maintenance( - &self, - expected_revision: u64, - operation: RequestId, - ) -> Result { - self.complete( - expected_revision, - operation, - ReleaseState::Maintenance, - "release is not in maintenance", - "maintenance release changed concurrently", - ) - .await - } - - async fn complete( - &self, - expected_revision: u64, - operation: RequestId, - required_state: ReleaseState, - unavailable: &'static str, - changed: &'static str, - ) -> Result { - let observed = self.load().await?.ok_or(Error::Release(unavailable))?; - if completion_retry(&observed.record, expected_revision, operation) { - let desired = observed - .record - .desired - .ok_or(Error::Release("ready release has no descriptor"))?; - self.descriptor(desired).await?; - return Ok(observed.record); - } - if observed.record.revision != expected_revision - || observed.record.state != required_state - || observed.record.operation != operation - { - return Err(Error::Release(changed)); - } - let desired = observed - .record - .desired - .ok_or(Error::Release("activating release has no descriptor"))?; - self.descriptor(desired).await?; - let mut next = observed.record.clone(); - next.revision = next - .revision - .checked_add(1) - .ok_or(Error::Release("release revision overflow"))?; - next.current = Some(desired); - next.state = ReleaseState::Ready; - self.update_exact(observed, next).await - } - - async fn update_exact( - &self, - observed: VersionedRelease, - next: ReleaseRecord, - ) -> Result { - next.validate(self.identity)?; - let write = self - .layout - .store() - .update( - &self.layout.release_path(), - Bytes::from(next.encode()?), - observed.token, - ) - .await; - match write { - Ok(_) => Ok(next), - Err(write_error) => match self.load().await? { - Some(current) if current.record == next => Ok(current.record), - Some(_) => Err(Error::Release("release changed concurrently")), - None => Err(write_error.into()), - }, - } - } - - async fn publish_descriptor(&self, descriptor: &[u8], digest: Digest) -> Result<()> { - let path = self.layout.release_descriptor_path(digest.as_bytes()); - match self - .layout - .store() - .create_strict(&path, Bytes::copy_from_slice(descriptor)) - .await - { - Ok(()) => Ok(()), - Err(create_error) => { - let (current, _) = self - .layout - .store() - .get_with_etag_bounded(&path, MAX_DESCRIPTOR_BYTES) - .await?; - if current.as_ref() == descriptor { - Ok(()) - } else { - let _ = create_error; - Err(Error::Release( - "descriptor digest path contains different bytes", - )) - } - } - } - } -} - -fn prepared_matches( - record: &ReleaseRecord, - digest: Digest, - expected_revision: u64, - image: &str, - operation: RequestId, -) -> bool { - let Some(next_revision) = expected_revision.checked_add(1) else { - return false; - }; - record.revision == next_revision - && record.desired == Some(digest) - && record.desired_image == image - && record.state == ReleaseState::Prepared - && record.operation == operation -} - -fn activation_retry(record: &ReleaseRecord, expected_revision: u64, operation: RequestId) -> bool { - if record.operation != operation || record.desired.is_none() { - return false; - } - match record.state { - ReleaseState::Activating => expected_revision - .checked_add(1) - .is_some_and(|revision| record.revision == revision), - ReleaseState::Ready => { - expected_revision - .checked_add(2) - .is_some_and(|revision| record.revision == revision) - && record.current == record.desired - } - _ => false, - } -} - -fn maintenance_retry(record: &ReleaseRecord, expected_revision: u64, operation: RequestId) -> bool { - record.operation == operation - && record.state == ReleaseState::Maintenance - && record.desired.is_some() - && expected_revision - .checked_add(1) - .is_some_and(|revision| record.revision == revision) -} - -fn completion_retry(record: &ReleaseRecord, expected_revision: u64, operation: RequestId) -> bool { - record.operation == operation - && record.state == ReleaseState::Ready - && record.current == record.desired - && expected_revision - .checked_add(1) - .is_some_and(|revision| record.revision == revision) -} - -fn selected_provision_release(record: &ReleaseRecord) -> Result { - match record.state { - ReleaseState::Ready => record - .current - .filter(|current| Some(*current) == record.desired) - .ok_or(Error::Release("ready release has no current descriptor")), - ReleaseState::Activating => record.desired.ok_or(Error::Release( - "activating release has no desired descriptor", - )), - _ => Err(Error::Release("release is not ready for provisioning")), - } -} - -fn provision_release_continues( - before: &ReleaseRecord, - after: &ReleaseRecord, - selected: Digest, -) -> bool { - if before == after { - return true; - } - before.state == ReleaseState::Activating - && after.state == ReleaseState::Ready - && before - .revision - .checked_add(1) - .is_some_and(|revision| after.revision == revision) - && after.operation == before.operation - && after.current == Some(selected) - && after.desired == Some(selected) - && after.desired_image == before.desired_image -} - -fn validate_image(image: &str) -> Result<()> { - let Some(digest) = image.strip_prefix("sha256:") else { - return Err(Error::Release("image must be a sha256 digest")); - }; - if digest.len() != 64 - || !digest - .bytes() - .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) - { - return Err(Error::Release("image must be a lowercase sha256 digest")); - } - Ok(()) -} - -fn parse_hex(value: &str) -> Result<[u8; N]> { - crate::identity::decode_hex(value).map_err(|_| Error::Release("invalid lowercase hex field")) -} - -fn parse_digest(value: &str) -> Result { - Ok(Digest::from_bytes(parse_hex(value)?)) -} - -fn parse_revision(value: &str) -> Result { - if value.is_empty() - || (value.len() > 1 && value.starts_with('0')) - || !value.bytes().all(|byte| byte.is_ascii_digit()) - { - return Err(Error::Release("invalid release revision")); - } - value - .parse() - .map_err(|_| Error::Release("invalid release revision")) -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/recovery/release/tests.rs b/crates/crab-cell-runtime/src/recovery/release/tests.rs deleted file mode 100644 index dcc9a74e5..000000000 --- a/crates/crab-cell-runtime/src/recovery/release/tests.rs +++ /dev/null @@ -1,340 +0,0 @@ -use std::sync::Arc; - -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use super::*; -use crate::identity::{ApplicationId, TenantId}; - -fn fixture() -> (ReleaseStore, Vec, Digest) { - let identity = ApplicationIdentity::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - ); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("root"), - *identity.application().as_bytes(), - ); - let descriptor = br#"{"runtime":"crab-http-server","version":1}"#.to_vec(); - let digest = Digest::from_bytes(*blake3::hash(&descriptor).as_bytes()); - ( - ReleaseStore::new(layout, identity).unwrap(), - descriptor, - digest, - ) -} - -#[tokio::test] -async fn prepare_uploads_descriptor_before_cas_and_reconciles_retry() { - let (releases, descriptor, digest) = fixture(); - let image = format!("sha256:{}", "a".repeat(64)); - let first = releases - .prepare( - &descriptor, - digest, - 0, - &image, - RequestId::from_bytes([3; 16]), - ) - .await - .unwrap(); - assert_eq!(first.revision(), 1); - assert_eq!(first.desired(), Some(digest)); - assert_eq!(first.state(), ReleaseState::Prepared); - - let retry = releases - .prepare( - &descriptor, - digest, - 0, - &image, - RequestId::from_bytes([3; 16]), - ) - .await - .unwrap(); - assert_eq!(retry, first); - assert!( - releases - .prepare( - &descriptor, - digest, - 1, - &format!("sha256:{}", "b".repeat(64)), - RequestId::from_bytes([5; 16]), - ) - .await - .is_ok() - ); -} - -#[tokio::test] -async fn prepare_rejects_digest_image_and_revision_drift() { - let (releases, descriptor, digest) = fixture(); - let image = format!("sha256:{}", "a".repeat(64)); - assert!(matches!( - releases - .prepare( - &descriptor, - Digest::from_bytes([9; 32]), - 0, - &image, - RequestId::from_bytes([3; 16]), - ) - .await, - Err(Error::Release(_)) - )); - assert!(matches!( - releases - .prepare( - &descriptor, - digest, - 0, - "latest", - RequestId::from_bytes([3; 16]), - ) - .await, - Err(Error::Release(_)) - )); - releases - .prepare( - &descriptor, - digest, - 0, - &image, - RequestId::from_bytes([3; 16]), - ) - .await - .unwrap(); - assert!(matches!( - releases - .prepare( - &descriptor, - digest, - 9, - &image, - RequestId::from_bytes([3; 16]), - ) - .await, - Err(Error::Release(_)) - )); -} - -#[tokio::test] -async fn activation_is_operation_bound_resumable_and_publishes_current() { - let (releases, descriptor, digest) = fixture(); - let operation = RequestId::from_bytes([3; 16]); - let prepared = releases - .prepare( - &descriptor, - digest, - 0, - &format!("sha256:{}", "a".repeat(64)), - operation, - ) - .await - .unwrap(); - - let activating = releases - .start_activation(prepared.revision(), operation) - .await - .unwrap(); - assert_eq!(activating.revision(), 2); - assert_eq!(activating.state(), ReleaseState::Activating); - assert_eq!(activating.current(), None); - assert_eq!( - releases - .start_activation(prepared.revision(), operation) - .await - .unwrap(), - activating - ); - - let ready = releases - .complete_activation(activating.revision(), operation) - .await - .unwrap(); - assert_eq!(ready.revision(), 3); - assert_eq!(ready.state(), ReleaseState::Ready); - assert_eq!(ready.current(), Some(digest)); - assert_eq!(ready.current(), ready.desired()); - assert_eq!( - releases - .start_activation(prepared.revision(), operation) - .await - .unwrap(), - ready - ); - assert_eq!( - releases - .complete_activation(activating.revision(), operation) - .await - .unwrap(), - ready - ); -} - -#[tokio::test] -async fn activation_rejects_operation_revision_and_descriptor_drift() { - let (releases, descriptor, digest) = fixture(); - let operation = RequestId::from_bytes([3; 16]); - let prepared = releases - .prepare( - &descriptor, - digest, - 0, - &format!("sha256:{}", "a".repeat(64)), - operation, - ) - .await - .unwrap(); - assert!( - releases - .start_activation(prepared.revision(), RequestId::from_bytes([4; 16])) - .await - .is_err() - ); - assert!( - releases - .start_activation(prepared.revision() + 1, operation) - .await - .is_err() - ); - - let path = releases.layout.release_descriptor_path(digest.as_bytes()); - releases - .layout - .store() - .put_overwrite(&path, Bytes::from_static(b"corrupt")) - .await - .unwrap(); - assert!( - releases - .start_activation(prepared.revision(), operation) - .await - .is_err() - ); -} - -#[tokio::test] -async fn maintenance_is_operation_bound_and_resumable() { - let (releases, descriptor, digest) = fixture(); - let operation = RequestId::from_bytes([3; 16]); - let prepared = releases - .prepare( - &descriptor, - digest, - 0, - &format!("sha256:{}", "a".repeat(64)), - operation, - ) - .await - .unwrap(); - - let maintenance = releases - .start_maintenance(prepared.revision(), operation) - .await - .unwrap(); - assert_eq!(maintenance.revision(), 2); - assert_eq!(maintenance.state(), ReleaseState::Maintenance); - assert_eq!(maintenance.current(), None); - assert_eq!(maintenance.desired(), Some(digest)); - assert_eq!( - releases - .start_maintenance(prepared.revision(), operation) - .await - .unwrap(), - maintenance - ); - assert!( - releases - .start_activation(prepared.revision(), operation) - .await - .is_err() - ); - assert!( - releases - .prepare( - &descriptor, - digest, - maintenance.revision(), - &format!("sha256:{}", "b".repeat(64)), - RequestId::from_bytes([4; 16]), - ) - .await - .is_err() - ); - let ready = releases - .complete_maintenance(maintenance.revision(), operation) - .await - .unwrap(); - assert_eq!(ready.revision(), 3); - assert_eq!(ready.state(), ReleaseState::Ready); - assert_eq!(ready.current(), Some(digest)); - assert_eq!( - releases - .complete_maintenance(maintenance.revision(), operation) - .await - .unwrap(), - ready - ); -} - -#[tokio::test] -async fn prepare_cannot_replace_an_activation_in_progress() { - let (releases, descriptor, digest) = fixture(); - let operation = RequestId::from_bytes([3; 16]); - let prepared = releases - .prepare( - &descriptor, - digest, - 0, - &format!("sha256:{}", "a".repeat(64)), - operation, - ) - .await - .unwrap(); - let activating = releases - .start_activation(prepared.revision(), operation) - .await - .unwrap(); - - assert!( - releases - .prepare( - &descriptor, - digest, - activating.revision(), - &format!("sha256:{}", "b".repeat(64)), - RequestId::from_bytes([5; 16]), - ) - .await - .is_err() - ); -} - -#[test] -fn provisioning_recheck_accepts_only_the_exact_activation_completion() { - let (_, _, digest) = fixture(); - let identity = ApplicationIdentity::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - ); - let before = ReleaseRecord { - application: identity.application(), - revision: 2, - current: None, - desired: Some(digest), - desired_image: format!("sha256:{}", "a".repeat(64)), - operation: RequestId::from_bytes([3; 16]), - state: ReleaseState::Activating, - }; - let mut completed = before.clone(); - completed.revision = 3; - completed.current = Some(digest); - completed.state = ReleaseState::Ready; - assert!(provision_release_continues(&before, &completed, digest)); - - completed.operation = RequestId::from_bytes([4; 16]); - assert!(!provision_release_continues(&before, &completed, digest)); -} diff --git a/crates/crab-cell-runtime/src/recovery/release_progress.rs b/crates/crab-cell-runtime/src/recovery/release_progress.rs deleted file mode 100644 index 92b90acd4..000000000 --- a/crates/crab-cell-runtime/src/recovery/release_progress.rs +++ /dev/null @@ -1,479 +0,0 @@ -//! Per-operation release progress and migration failure records. -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_storage::{ETag, StorageError}; -use serde::{Deserialize, Serialize}; - -use crate::cell::application::ApplicationIdentity; -use crate::identity::{CellId, Digest, RequestId, SessionId}; -use crate::identity::{decode_hex, encode_hex}; -use crate::retry::retryable_storage_error; -use crate::{Error, Result}; - -const MAX_PROGRESS_BYTES: u64 = 4 * 1024; -const MAX_WRITE_ATTEMPTS: usize = 4; - -/// Stable failure class exposed by release migration progress. -#[derive(Clone, Copy, Debug, Deserialize, Serialize, PartialEq, Eq)] -#[serde(rename_all = "snake_case")] -pub enum MigrationFailure { - /// The destination release was unavailable. - Unavailable, - /// The Cell ran out of local capacity. - Capacity, - /// The migration exceeded its deadline. - Deadline, - /// The migration cannot apply to this Cell. - Incompatible, - /// The migration failed for an internal reason. - Internal, -} - -/// Terminal state of one attempted release migration. -#[derive(Clone, Copy, Debug, Deserialize, Serialize, PartialEq, Eq)] -#[serde(rename_all = "snake_case")] -pub enum MigrationProgressState { - /// The migration finished successfully. - Completed, - /// The migration ended in failure. - Failed, -} - -/// Immutable scope of one Cell's migration toward an activating release. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct MigrationProgressAttempt { - operation: RequestId, - release: Digest, - session: SessionId, - cell: CellId, - from_code: Digest, - from_schema: u32, - to_code: Digest, - to_schema: u32, -} - -impl MigrationProgressAttempt { - /// Binds progress to one release operation and exact source/target versions. - pub fn new( - operation: RequestId, - release: Digest, - session: SessionId, - cell: CellId, - from: (Digest, u32), - to: (Digest, u32), - ) -> Result { - let attempt = Self { - operation, - release, - session, - cell, - from_code: from.0, - from_schema: from.1, - to_code: to.0, - to_schema: to.1, - }; - attempt.validate()?; - Ok(attempt) - } - - /// Returns the request that started the attempt. - #[must_use] - pub const fn operation(&self) -> RequestId { - self.operation - } - - /// Returns the release the Cell migrates to. - #[must_use] - pub const fn release(&self) -> Digest { - self.release - } - - /// Returns the session that owns the attempt. - #[must_use] - pub const fn session(&self) -> SessionId { - self.session - } - - /// Returns the Cell the attempt migrates. - #[must_use] - pub const fn cell(&self) -> CellId { - self.cell - } - - /// Returns the code digest and schema the Cell migrates from. - #[must_use] - pub const fn from(&self) -> (Digest, u32) { - (self.from_code, self.from_schema) - } - - /// Returns the code digest and schema the Cell migrates to. - #[must_use] - pub const fn to(&self) -> (Digest, u32) { - (self.to_code, self.to_schema) - } - - fn validate(&self) -> Result<()> { - if self.operation.as_bytes().iter().all(|byte| *byte == 0) - || self.release.as_bytes().iter().all(|byte| *byte == 0) - || self.session.as_bytes().iter().all(|byte| *byte == 0) - || self.cell.as_bytes().iter().all(|byte| *byte == 0) - || self.from_code.as_bytes().iter().all(|byte| *byte == 0) - || self.to_code.as_bytes().iter().all(|byte| *byte == 0) - || self.from_schema == 0 - || self.to_schema == 0 - || self.from_code == self.to_code && self.from_schema == self.to_schema - { - return Err(Error::Release("invalid release migration progress scope")); - } - Ok(()) - } -} - -/// Canonical latest terminal result for one release operation and Cell. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct MigrationProgress { - application: crate::ApplicationId, - attempt: MigrationProgressAttempt, - revision: u64, - attempts: u32, - state: MigrationProgressState, - failure: Option, - updated_at_ms: i64, -} - -impl MigrationProgress { - /// Returns the immutable attempt scope. - #[must_use] - pub const fn attempt(&self) -> MigrationProgressAttempt { - self.attempt - } - - /// Returns the progress revision. - #[must_use] - pub const fn revision(&self) -> u64 { - self.revision - } - - /// Returns how many attempts were recorded. - #[must_use] - pub const fn attempts(&self) -> u32 { - self.attempts - } - - /// Returns the terminal state, once the attempt finished. - #[must_use] - pub const fn state(&self) -> MigrationProgressState { - self.state - } - - /// Returns the failure class, when the attempt failed. - #[must_use] - pub const fn failure(&self) -> Option { - self.failure - } - - /// Returns the logical time the progress was last written. - #[must_use] - pub const fn updated_at_ms(&self) -> i64 { - self.updated_at_ms - } - - fn encode(&self) -> Result> { - let raw = RawMigrationProgress { - application: encode_hex(self.application.as_bytes()), - attempts: self.attempts, - cell: encode_hex(self.attempt.cell.as_bytes()), - failure: self.failure, - from_code: encode_hex(self.attempt.from_code.as_bytes()), - from_schema: self.attempt.from_schema, - operation: encode_hex(self.attempt.operation.as_bytes()), - release: encode_hex(self.attempt.release.as_bytes()), - revision: self.revision.to_string(), - session: encode_hex(self.attempt.session.as_bytes()), - state: self.state, - to_code: encode_hex(self.attempt.to_code.as_bytes()), - to_schema: self.attempt.to_schema, - updated_at_ms: self.updated_at_ms.to_string(), - version: 1, - }; - let encoded = serde_json::to_vec(&raw)?; - if encoded.len() as u64 > MAX_PROGRESS_BYTES { - return Err(Error::Release("release migration progress exceeds 4 KiB")); - } - Ok(encoded) - } - - fn decode(bytes: &[u8], identity: ApplicationIdentity) -> Result { - let raw: RawMigrationProgress = serde_json::from_slice(bytes)?; - if raw.version != 1 { - return Err(Error::Release( - "unsupported release migration progress version", - )); - } - let progress = Self { - application: crate::ApplicationId::from_bytes(parse_hex(&raw.application)?), - attempt: MigrationProgressAttempt { - operation: RequestId::from_bytes(parse_hex(&raw.operation)?), - release: Digest::from_bytes(parse_hex(&raw.release)?), - session: SessionId::from_bytes(parse_hex(&raw.session)?), - cell: CellId::from_bytes(parse_hex(&raw.cell)?), - from_code: Digest::from_bytes(parse_hex(&raw.from_code)?), - from_schema: raw.from_schema, - to_code: Digest::from_bytes(parse_hex(&raw.to_code)?), - to_schema: raw.to_schema, - }, - revision: parse_u64(&raw.revision)?, - attempts: raw.attempts, - state: raw.state, - failure: raw.failure, - updated_at_ms: parse_i64(&raw.updated_at_ms)?, - }; - progress.validate(identity)?; - if progress.encode()?.as_slice() != bytes { - return Err(Error::Release( - "release migration progress is not canonical", - )); - } - Ok(progress) - } - - fn validate(&self, identity: ApplicationIdentity) -> Result<()> { - self.attempt.validate()?; - if self.application != identity.application() - || self.revision == 0 - || self.attempts == 0 - || self.updated_at_ms < 0 - || matches!(self.state, MigrationProgressState::Completed) && self.failure.is_some() - || matches!(self.state, MigrationProgressState::Failed) && self.failure.is_none() - { - return Err(Error::Release("invalid release migration progress")); - } - Ok(()) - } -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct RawMigrationProgress { - application: String, - attempts: u32, - cell: String, - failure: Option, - from_code: String, - from_schema: u32, - operation: String, - release: String, - revision: String, - session: String, - state: MigrationProgressState, - to_code: String, - to_schema: u32, - updated_at_ms: String, - version: u8, -} - -struct VersionedProgress { - progress: MigrationProgress, - token: ETag, -} - -/// CAS-backed terminal progress records for a release migration operation. -#[derive(Clone)] -pub struct MigrationProgressStore { - layout: CellStorageLayout, - identity: ApplicationIdentity, -} - -impl MigrationProgressStore { - /// Creates a progress store after checking that the layout belongs to the - /// same application as the identity. - pub fn new(layout: CellStorageLayout, identity: ApplicationIdentity) -> Result { - if layout.application_id() != identity.application().as_bytes() { - return Err(Error::Release( - "migration progress layout and application differ", - )); - } - Ok(Self { layout, identity }) - } - - /// Loads and verifies the latest terminal result for one operation and Cell. - pub async fn load( - &self, - cell: CellId, - operation: RequestId, - ) -> Result> { - Ok(self - .load_versioned(cell, operation) - .await? - .map(|versioned| versioned.progress)) - } - - /// Records a migration whose target code/schema is now authoritative. - pub async fn completed( - &self, - attempt: MigrationProgressAttempt, - now_ms: i64, - ) -> Result { - self.record(attempt, MigrationProgressState::Completed, None, now_ms) - .await - } - - /// Records a bounded failure class without persisting source error text. - pub async fn failed( - &self, - attempt: MigrationProgressAttempt, - failure: MigrationFailure, - now_ms: i64, - ) -> Result { - self.record( - attempt, - MigrationProgressState::Failed, - Some(failure), - now_ms, - ) - .await - } - - async fn record( - &self, - attempt: MigrationProgressAttempt, - state: MigrationProgressState, - failure: Option, - now_ms: i64, - ) -> Result { - attempt.validate()?; - if now_ms < 0 { - return Err(Error::Release("negative migration progress time")); - } - let mut last_write_error = None; - for _ in 0..MAX_WRITE_ATTEMPTS { - let observed = self.load_versioned(attempt.cell, attempt.operation).await?; - if let Some(observed) = &observed - && !same_operation(&observed.progress.attempt, &attempt) - { - return Err(Error::Release("release migration progress scope changed")); - } - if let Some(observed) = &observed - && observed.progress.state == MigrationProgressState::Completed - { - return Ok(observed.progress.clone()); - } - let next = MigrationProgress { - application: self.identity.application(), - attempt, - revision: observed.as_ref().map_or(Ok(1), |value| { - value - .progress - .revision - .checked_add(1) - .ok_or(Error::Release("migration progress revision overflow")) - })?, - attempts: observed.as_ref().map_or(Ok(1), |value| { - value - .progress - .attempts - .checked_add(1) - .ok_or(Error::Release("migration progress attempt overflow")) - })?, - state, - failure, - updated_at_ms: now_ms, - }; - next.validate(self.identity)?; - let bytes = Bytes::from(next.encode()?); - let path = self.path(attempt.cell, attempt.operation); - let write = match observed { - Some(observed) => { - self.layout - .store() - .update(&path, bytes, observed.token) - .await - } - None => { - self.layout - .store() - .create_strict_with_etag(&path, bytes) - .await - } - }; - match write { - Ok(_) => return Ok(next), - Err(error) => { - if let Some(current) = - self.load_versioned(attempt.cell, attempt.operation).await? - && current.progress == next - { - return Ok(current.progress); - } - if !retryable_storage_error(&error) { - return Err(error.into()); - } - last_write_error = Some(error); - } - } - } - Err(last_write_error.map_or( - Error::Release("release migration progress changed repeatedly"), - Error::Storage, - )) - } - - async fn load_versioned( - &self, - cell: CellId, - operation: RequestId, - ) -> Result> { - let path = self.path(cell, operation); - let (bytes, token) = match self - .layout - .store() - .get_with_etag_bounded(&path, MAX_PROGRESS_BYTES) - .await - { - Ok(value) => value, - Err(StorageError::NotFound { .. }) => return Ok(None), - Err(error) => return Err(error.into()), - }; - let progress = MigrationProgress::decode(&bytes, self.identity)?; - if progress.attempt.cell != cell || progress.attempt.operation != operation { - return Err(Error::Release( - "release migration progress is stored at the wrong path", - )); - } - Ok(Some(VersionedProgress { progress, token })) - } - - fn path(&self, cell: CellId, operation: RequestId) -> object_store::path::Path { - self.layout - .migration_path(cell.as_bytes(), operation.as_bytes(), "release.json") - } -} - -fn same_operation(left: &MigrationProgressAttempt, right: &MigrationProgressAttempt) -> bool { - left.operation == right.operation - && left.release == right.release - && left.cell == right.cell - && left.to_code == right.to_code - && left.to_schema == right.to_schema -} - -fn parse_hex(value: &str) -> Result<[u8; N]> { - decode_hex(value).map_err(|_| Error::Release("invalid migration progress hex field")) -} - -fn parse_u64(value: &str) -> Result { - if value.is_empty() - || (value.len() > 1 && value.starts_with('0')) - || !value.bytes().all(|byte| byte.is_ascii_digit()) - { - return Err(Error::Release("invalid migration progress integer")); - } - value - .parse() - .map_err(|_| Error::Release("invalid migration progress integer")) -} - -fn parse_i64(value: &str) -> Result { - let parsed = parse_u64(value)?; - i64::try_from(parsed).map_err(|_| Error::Release("migration progress time overflow")) -} diff --git a/crates/crab-cell-runtime/src/recovery/retention.rs b/crates/crab-cell-runtime/src/recovery/retention.rs deleted file mode 100644 index 0b46f38d3..000000000 --- a/crates/crab-cell-runtime/src/recovery/retention.rs +++ /dev/null @@ -1,598 +0,0 @@ -//! Offline retention of immutable Cell objects. -use std::path::Path as FilePath; -use std::sync::{Arc, Mutex}; - -use crab_ltx::CellStorageLayout; -use crab_storage::StorageError; -use futures_util::StreamExt as _; -use object_store::{ObjectMeta, path::Path}; -use rusqlite::{Connection, OptionalExtension, params}; - -use crate::cell::application::ApplicationIdentity; -use crate::cell::catalog::CellCatalog; -use crate::control::authority::CellAuthority; -use crate::control::{Control, ControlState}; -use crate::identity::RequestId; -use crate::identity::decode_hex; -use crate::ltx::{Host as ReplicaHost, Limits as ReplicaLimits}; -use crate::recovery::backup::BackupPinStore; -use crate::recovery::release::{ReleaseRecord, ReleaseState, ReleaseStore}; -use crate::{Error, Result}; - -const CANDIDATE_BATCH: usize = 256; -const MAX_DELETES_PER_RUN: u64 = 100_000; - -/// Explicit bounds for one offline immutable-object collection pass. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct GarbageCollectionPolicy { - now_ms: i64, - grace_ms: u64, - max_deletes: u64, -} - -impl GarbageCollectionPolicy { - /// Creates a bounded collection policy: logical time, grace window, and a - /// per-run deletion limit. - pub fn new(now_ms: i64, grace_ms: u64, max_deletes: u64) -> Result { - if now_ms < 0 || grace_ms == 0 || !(1..=MAX_DELETES_PER_RUN).contains(&max_deletes) { - return Err(Error::Retention( - "time, grace, or deletion limit is invalid", - )); - } - Ok(Self { - now_ms, - grace_ms, - max_deletes, - }) - } - - fn cutoff_ms(self) -> i64 { - self.now_ms - .saturating_sub(i64::try_from(self.grace_ms).unwrap_or(i64::MAX)) - } -} - -/// Auditable outcome of one complete mark scan and bounded sweep. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct GarbageCollectionReport { - listed_objects: u64, - immutable_candidates: u64, - reachable_objects: u64, - retained_objects: u64, - grace_objects: u64, - eligible_objects: u64, - deleted_objects: u64, - current_controls: u64, - retained_pins: u64, -} - -impl GarbageCollectionReport { - /// Returns the objects listed in the pass. - #[must_use] - pub const fn listed_objects(&self) -> u64 { - self.listed_objects - } - - /// Returns the listed objects that are immutable candidates. - #[must_use] - pub const fn immutable_candidates(&self) -> u64 { - self.immutable_candidates - } - - /// Returns the candidates still reachable from a control record or pin. - #[must_use] - pub const fn reachable_objects(&self) -> u64 { - self.reachable_objects - } - - /// Returns the candidates a backup pin retains. - #[must_use] - pub const fn retained_objects(&self) -> u64 { - self.retained_objects - } - - /// Returns the candidates inside the grace window. - #[must_use] - pub const fn grace_objects(&self) -> u64 { - self.grace_objects - } - - /// Returns the candidates eligible for deletion. - #[must_use] - pub const fn eligible_objects(&self) -> u64 { - self.eligible_objects - } - - /// Returns the objects deleted in this pass. - #[must_use] - pub const fn deleted_objects(&self) -> u64 { - self.deleted_objects - } - - /// Returns the control records observed as current. - #[must_use] - pub const fn current_controls(&self) -> u64 { - self.current_controls - } - - /// Returns the backup pins that retained objects. - #[must_use] - pub const fn retained_pins(&self) -> u64 { - self.retained_pins - } - - /// Reports whether every eligible object was deleted. - #[must_use] - pub const fn complete(&self) -> bool { - self.eligible_objects == self.deleted_objects - } -} - -/// Reclaims immutable Cell objects only while an application is fenced offline. -/// -/// The caller must own the exclusive maintenance advertisement associated with -/// `maintenance` until this future returns, and that drain must include every -/// read replica, in-flight snapshot query, and backup-pin publisher. -/// The collector independently checks the exact release -/// record and rejects every owned control before sweeping. -#[derive(Clone)] -pub struct CellGarbageCollector { - layout: CellStorageLayout, - identity: ApplicationIdentity, - limits: ReplicaLimits, - host: ReplicaHost, -} - -impl CellGarbageCollector { - /// Creates the collector after checking that the layout belongs to the same - /// application as the identity. - pub fn new( - layout: CellStorageLayout, - identity: ApplicationIdentity, - limits: ReplicaLimits, - host: ReplicaHost, - ) -> Result { - if layout.application_id() != identity.application().as_bytes() { - return Err(Error::Retention("layout and application identity differ")); - } - Ok(Self { - layout, - identity, - limits, - host, - }) - } - - /// Marks current controls and every retained pin, then deletes old unreachable objects. - /// - /// `scratch_dir` receives a private SQLite mark set and must be on bounded - /// operator-owned local storage. A corrupt or missing reachable object fails - /// before any deletion. The provider listing is streamed, so remote inventory - /// size does not become resident memory. - pub async fn collect( - &self, - maintenance: &ReleaseRecord, - scratch_dir: &FilePath, - policy: GarbageCollectionPolicy, - ) -> Result { - self.require_maintenance(maintenance).await?; - self.require_single_tenant_catalog().await?; - let marks = MarkStore::open(scratch_dir).await?; - let releases = ReleaseStore::new(self.layout.clone(), self.identity)?; - let mut current_controls = 0u64; - let mut retained_pins = 0u64; - - self.mark_release(&marks, maintenance, &releases).await?; - self.mark_current_catalog(&marks, &mut current_controls) - .await?; - self.mark_pins(&marks, &mut retained_pins).await?; - - // Every graph has verified before this point. Recheck the global write - // fence immediately before the first destructive operation. - self.require_maintenance(maintenance).await?; - let reachable_objects = marks.count().await?; - let mut report = GarbageCollectionReport { - listed_objects: 0, - immutable_candidates: 0, - reachable_objects, - retained_objects: 0, - grace_objects: 0, - eligible_objects: 0, - deleted_objects: 0, - current_controls, - retained_pins, - }; - let prefix = self.layout.application_prefix(); - let mut stream = self.layout.store().list_stream(&prefix); - let mut candidates = Vec::with_capacity(CANDIDATE_BATCH); - while let Some(item) = stream.next().await { - let meta = item?; - report.listed_objects = checked_increment(report.listed_objects)?; - if immutable_candidate(&prefix, &meta.location) { - report.immutable_candidates = checked_increment(report.immutable_candidates)?; - candidates.push(meta); - if candidates.len() == CANDIDATE_BATCH { - self.sweep_batch(&marks, policy, &mut report, &mut candidates) - .await?; - } - } - } - self.sweep_batch(&marks, policy, &mut report, &mut candidates) - .await?; - self.require_maintenance(maintenance).await?; - Ok(report) - } - - async fn require_single_tenant_catalog(&self) -> Result<()> { - // Sweep spans the application, but this collector marks one tenant. - // Reject shared roots before deleting any unmarked tenant's objects. - let prefix = self.layout.catalog_tenants_prefix(); - let expected = crate::identity::encode_hex(self.identity.tenant().as_bytes()); - let mut heads = self.layout.store().list_stream(&prefix); - while let Some(head) = heads.next().await { - let head = head?; - if relative_path(&prefix, &head.location).and_then(|path| path.split('/').next()) - != Some(expected.as_str()) - { - return Err(Error::Retention( - "maintenance collection requires a single-tenant catalog", - )); - } - } - Ok(()) - } - - async fn require_maintenance(&self, expected: &ReleaseRecord) -> Result<()> { - if expected.state() != ReleaseState::Maintenance { - return Err(Error::Retention("release is not fenced for maintenance")); - } - let current = ReleaseStore::new(self.layout.clone(), self.identity)? - .load() - .await? - .ok_or(Error::Retention("maintenance release is absent"))?; - if current.record() != expected { - return Err(Error::Retention("maintenance release changed")); - } - Ok(()) - } - - async fn mark_release( - &self, - marks: &MarkStore, - release: &ReleaseRecord, - releases: &ReleaseStore, - ) -> Result<()> { - let mut digests = [release.current(), release.desired()] - .into_iter() - .flatten() - .collect::>(); - digests.sort_unstable_by_key(|digest| *digest.as_bytes()); - digests.dedup(); - let mut paths = Vec::with_capacity(digests.len()); - for digest in digests { - releases.descriptor(digest).await?; - paths.push(self.layout.release_descriptor_path(digest.as_bytes())); - } - marks.insert(paths).await - } - - async fn mark_current_catalog( - &self, - marks: &MarkStore, - current_controls: &mut u64, - ) -> Result<()> { - let catalog = CellCatalog::new(self.layout.clone(), self.identity.tenant()); - let authority = CellAuthority::new(self.layout.clone()); - for shard in 0_u8..=u8::MAX { - let mut scan = catalog.scan_shard(shard).await?; - marks - .insert( - scan.page_digests() - .iter() - .map(|digest| self.layout.catalog_object_path(digest.as_bytes())) - .collect(), - ) - .await?; - while let Some(page) = scan.next_page().await? { - for proof in page.entries() { - let Some(observed) = authority.load(proof.entry().cell()).await? else { - continue; - }; - *current_controls = checked_increment(*current_controls)?; - let control = observed.value(); - if control.owner.is_some() { - return Err(Error::Retention( - "a current Cell still has an owner during maintenance", - )); - } - if control.state != ControlState::Tombstoned { - self.mark_control_root(marks, control).await?; - } - } - } - } - Ok(()) - } - - async fn mark_pins(&self, marks: &MarkStore, retained_pins: &mut u64) -> Result<()> { - let pins = BackupPinStore::new( - self.layout.clone(), - self.identity, - self.limits, - self.host.clone(), - )?; - let prefix = self.layout.pin_prefix(); - let mut stream = self.layout.store().list_stream(&prefix); - while let Some(item) = stream.next().await { - let meta = item?; - let Some(id) = pin_id(&prefix, &meta.location)? else { - continue; - }; - *retained_pins = checked_increment(*retained_pins)?; - let pin = pins - .load(id) - .await? - .ok_or(Error::Retention("retained pin disappeared during mark"))?; - let manifest = pins.manifest(&pin).await?; - let mut paths = manifest - .release_descriptors - .iter() - .map(|digest| self.layout.release_descriptor_path(digest.as_bytes())) - .collect::>(); - for shard in &manifest.catalog { - paths.extend( - shard - .pages - .iter() - .map(|digest| self.layout.catalog_object_path(digest.as_bytes())), - ); - } - paths.extend( - manifest - .pin_objects - .iter() - .map(|digest| self.layout.pin_object_path(digest)), - ); - marks.insert(paths).await?; - for control in &manifest.controls { - self.mark_control_root(marks, control).await?; - } - } - Ok(()) - } - - async fn mark_control_root(&self, marks: &MarkStore, control: &Control) -> Result<()> { - let Some(root) = control.ltx_root() else { - return Ok(()); - }; - let objects = crab_ltx::CellReplica::new( - self.layout.clone(), - *control.cell.as_bytes(), - *control.incarnation.as_bytes(), - self.limits, - )? - .with_host(self.host.clone()) - .reachable_objects(&root) - .await?; - marks - .insert( - objects - .into_iter() - .map(|object| { - self.layout.incarnation_object_path( - control.cell.as_bytes(), - control.incarnation.as_bytes(), - &object.digest, - object.kind, - ) - }) - .collect(), - ) - .await - } - - async fn sweep_batch( - &self, - marks: &MarkStore, - policy: GarbageCollectionPolicy, - report: &mut GarbageCollectionReport, - batch: &mut Vec, - ) -> Result<()> { - if batch.is_empty() { - return Ok(()); - } - let paths = batch - .iter() - .map(|meta| meta.location.clone()) - .collect::>(); - let retained = marks.contains(paths).await?; - for (meta, retained) in batch.drain(..).zip(retained) { - if retained { - report.retained_objects = checked_increment(report.retained_objects)?; - continue; - } - if meta.last_modified.timestamp_millis() >= policy.cutoff_ms() { - report.grace_objects = checked_increment(report.grace_objects)?; - continue; - } - report.eligible_objects = checked_increment(report.eligible_objects)?; - if report.deleted_objects == policy.max_deletes { - continue; - } - match self.layout.store().delete(&meta.location).await { - Ok(()) | Err(StorageError::NotFound { .. }) => { - report.deleted_objects = checked_increment(report.deleted_objects)?; - } - Err(error) => return Err(error.into()), - } - } - Ok(()) - } -} - -struct MarkStore { - connection: Arc>, - _path: tempfile::TempPath, -} - -impl MarkStore { - async fn open(directory: &FilePath) -> Result { - let file = tempfile::Builder::new() - .prefix("crab-cell-retention-") - .suffix(".sqlite3") - .tempfile_in(directory) - .map_err(Error::RetentionIo)?; - let path = file.into_temp_path(); - let open_path = path.to_path_buf(); - let connection = tokio::task::spawn_blocking(move || -> Result { - let connection = Connection::open(open_path)?; - connection.execute_batch( - "PRAGMA journal_mode=OFF; PRAGMA synchronous=OFF;\ - CREATE TABLE marks(path TEXT PRIMARY KEY) WITHOUT ROWID;", - )?; - Ok(connection) - }) - .await - .map_err(Error::RetentionWorkerJoin)??; - Ok(Self { - connection: Arc::new(Mutex::new(connection)), - _path: path, - }) - } - - async fn insert(&self, paths: Vec) -> Result<()> { - if paths.is_empty() { - return Ok(()); - } - let paths = paths - .into_iter() - .map(|path| path.to_string()) - .collect::>(); - let connection = Arc::clone(&self.connection); - tokio::task::spawn_blocking(move || -> Result<()> { - let mut connection = connection - .lock() - .map_err(|_| Error::Retention("retention scratch lock is poisoned"))?; - let transaction = connection.transaction()?; - { - let mut insert = transaction.prepare_cached( - "INSERT INTO marks(path) VALUES(?1) ON CONFLICT(path) DO NOTHING", - )?; - for path in paths { - insert.execute(params![path])?; - } - } - transaction.commit()?; - Ok(()) - }) - .await - .map_err(Error::RetentionWorkerJoin)??; - Ok(()) - } - - async fn contains(&self, paths: Vec) -> Result> { - let paths = paths - .into_iter() - .map(|path| path.to_string()) - .collect::>(); - let connection = Arc::clone(&self.connection); - tokio::task::spawn_blocking(move || -> Result> { - let connection = connection - .lock() - .map_err(|_| Error::Retention("retention scratch lock is poisoned"))?; - let mut lookup = connection.prepare_cached("SELECT 1 FROM marks WHERE path = ?1")?; - paths - .into_iter() - .map(|path| { - lookup - .query_row(params![path], |_| Ok(())) - .optional() - .map(|value| value.is_some()) - .map_err(Error::from) - }) - .collect() - }) - .await - .map_err(Error::RetentionWorkerJoin)? - } - - async fn count(&self) -> Result { - let connection = Arc::clone(&self.connection); - tokio::task::spawn_blocking(move || -> Result { - let connection = connection - .lock() - .map_err(|_| Error::Retention("retention scratch lock is poisoned"))?; - let count = connection - .query_row("SELECT COUNT(*) FROM marks", [], |row| row.get::<_, i64>(0))?; - u64::try_from(count).map_err(|_| Error::Retention("reachable object count overflow")) - }) - .await - .map_err(Error::RetentionWorkerJoin)? - } -} - -fn immutable_candidate(application_prefix: &Path, location: &Path) -> bool { - let Some(relative) = relative_path(application_prefix, location) else { - return false; - }; - let components = relative.split('/').collect::>(); - match components.as_slice() { - ["releases", object] | ["catalog", "objects", object] | ["pins", "objects", object] => { - json_digest(object) - } - ["cells", cell, "inc", incarnation, "objects", object] => { - lower_hex(cell, 64) - && lower_hex(incarnation, 32) - && object.rsplit_once('.').is_some_and(|(digest, extension)| { - lower_hex(digest, 64) - && matches!(extension, "ltx" | "index" | "dir" | "root" | "bundle") - }) - } - _ => false, - } -} - -fn pin_id(prefix: &Path, location: &Path) -> Result> { - let Some(relative) = relative_path(prefix, location) else { - return Ok(None); - }; - if relative.contains('/') { - return Ok(None); - } - let Some(encoded) = relative.strip_suffix(".json") else { - return Ok(None); - }; - if !lower_hex(encoded, 32) { - return Err(Error::Retention("pin pointer path is not canonical")); - } - Ok(Some(RequestId::from_bytes(decode_hex(encoded)?))) -} - -fn relative_path<'a>(prefix: &Path, location: &'a Path) -> Option<&'a str> { - location - .as_ref() - .strip_prefix(prefix.as_ref())? - .strip_prefix('/') -} - -fn json_digest(value: &str) -> bool { - value - .strip_suffix(".json") - .is_some_and(|digest| lower_hex(digest, 64)) -} - -fn lower_hex(value: &str, length: usize) -> bool { - value.len() == length - && value - .bytes() - .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) -} - -fn checked_increment(value: u64) -> Result { - value - .checked_add(1) - .ok_or(Error::Retention("retention object count overflow")) -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-cell-runtime/src/recovery/retention/tests.rs b/crates/crab-cell-runtime/src/recovery/retention/tests.rs deleted file mode 100644 index 7ff180ede..000000000 --- a/crates/crab-cell-runtime/src/recovery/retention/tests.rs +++ /dev/null @@ -1,620 +0,0 @@ -use std::{sync::Arc, time::UNIX_EPOCH}; - -use bytes::Bytes; -use crab_ltx::CellObjectKind; -use crab_storage::{ObjectStoreCredentials, Store, build_explicit_store}; -use object_store::{memory::InMemory, path::Path}; - -use super::*; -use crate::cell::catalog::CatalogEntry; -use crate::cell::catalog::CatalogRole; -use crate::control::{Owner, RootRef}; -use crate::identity::IncarnationId; -use crate::identity::{ - ApplicationId, CellId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; - -fn identity() -> ApplicationIdentity { - ApplicationIdentity::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - ) -} - -fn target(identity: ApplicationIdentity, partition: &[u8]) -> CellTarget { - CellTarget::new( - identity.tenant(), - identity.application(), - NamespaceId::from_bytes([3; 16]), - partition, - ) - .unwrap() -} - -async fn root( - layout: &CellStorageLayout, - target: &CellTarget, - incarnation: IncarnationId, - payload: usize, -) -> (crab_ltx::RootRef, Vec) { - let limits = ReplicaLimits::default(); - let replica = crab_ltx::CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *incarnation.as_bytes(), - limits, - ) - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let mut database = crab_ltx::Db::open(&directory.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute( - "CREATE TABLE values_(value BLOB NOT NULL)", - rusqlite::params![], - )?; - transaction.execute( - "INSERT INTO values_ VALUES(zeroblob(?1))", - rusqlite::params![payload], - )?; - Ok(()) - }) - .unwrap(); - let prepared = replica - .prepare(None, &database.capture().unwrap(), 1, 1) - .await - .unwrap(); - database.close().unwrap(); - let root = prepared.root(); - let paths = replica - .reachable_objects(&root) - .await - .unwrap() - .into_iter() - .map(|object| { - layout.incarnation_object_path( - target.cell_id().as_bytes(), - incarnation.as_bytes(), - &object.digest, - object.kind, - ) - }) - .collect(); - (root, paths) -} - -fn idle_control( - cell: CellId, - incarnation: IncarnationId, - root: crab_ltx::RootRef, - code: Digest, -) -> Control { - let control = Control { - cell, - incarnation, - epoch: 1, - revision: 3, - progress: 3, - state: ControlState::Idle, - owner: None, - root: Some(RootRef::from_ltx(cell, incarnation, root).unwrap()), - recovery: None, - code, - schema: 1, - next_due_ms: None, - }; - control.encode().unwrap(); - control -} - -async fn pinned_catalog(catalog: &CellCatalog) -> Vec { - let mut pinned = Vec::with_capacity(256); - for shard in 0_u8..=u8::MAX { - let mut scan = catalog.scan_shard(shard).await.unwrap(); - let revision = scan.revision(); - let pages = scan.page_digests().to_vec(); - while scan.next_page().await.unwrap().is_some() {} - pinned.push(crate::recovery::backup::PinnedCatalogShard { - shard, - revision, - pages, - }); - } - pinned -} - -async fn put_content_addressed(layout: &CellStorageLayout, path: Path, body: &[u8]) -> Path { - layout - .store() - .put(&path, Bytes::copy_from_slice(body)) - .await - .unwrap(); - path -} - -#[tokio::test(flavor = "multi_thread")] -async fn maintenance_collection_preserves_live_and_pinned_graphs() { - collection_preserves_live_and_pinned_graphs( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - ) - .await; -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires an isolated pre-created RustFS bucket, prefix and explicit test credentials"] -async fn rustfs_maintenance_collection_preserves_live_and_pinned_graphs() { - let required = |name| std::env::var(name).unwrap_or_else(|_| panic!("missing {name}")); - let store = build_explicit_store( - &required("CRAB_CELL_TEST_BUCKET"), - ObjectStoreCredentials::Aws { - access_key_id: required("AWS_ACCESS_KEY_ID"), - secret_access_key: required("AWS_SECRET_ACCESS_KEY"), - session_token: None, - region: "us-east-1".into(), - }, - Some(&required("CRAB_CELL_TEST_ENDPOINT")), - true, - ) - .unwrap(); - collection_preserves_live_and_pinned_graphs( - store, - Path::from(format!("{}/retention", required("CRAB_CELL_TEST_PREFIX"))), - ) - .await; -} - -async fn collection_preserves_live_and_pinned_graphs(store: Store, prefix: Path) { - let identity = identity(); - let layout = CellStorageLayout::new(store, prefix, *identity.application().as_bytes()); - let catalog = CellCatalog::new(layout.clone(), identity.tenant()); - let code = Digest::from_bytes([4; 32]); - let current_target = target(identity, b"current"); - let pinned_target = target(identity, b"pinned"); - for target in [¤t_target, &pinned_target] { - catalog - .provision(CatalogEntry::new(target, CatalogRole::Repository, code, 1).unwrap()) - .await - .unwrap(); - } - - let current_incarnation = IncarnationId::from_bytes([5; 16]); - let pinned_incarnation = IncarnationId::from_bytes([6; 16]); - let orphan_incarnation = IncarnationId::from_bytes([7; 16]); - let (current_root, current_paths) = - root(&layout, ¤t_target, current_incarnation, 128_000).await; - let current_replica = crab_ltx::CellReplica::new( - layout.clone(), - *current_target.cell_id().as_bytes(), - *current_incarnation.as_bytes(), - ReplicaLimits::default(), - ) - .unwrap(); - let compaction_scratch = tempfile::TempDir::new().unwrap(); - let compacted_root = current_replica - .prepare_compaction(¤t_root, 0..1, 9, compaction_scratch.path()) - .await - .unwrap() - .root(); - assert_eq!(compacted_root.position, current_root.position); - assert_eq!(compacted_root.commit_sequence, current_root.commit_sequence); - assert_ne!(compacted_root.digest, current_root.digest); - let compacted_paths = current_replica - .reachable_objects(&compacted_root) - .await - .unwrap() - .into_iter() - .map(|object| { - layout.incarnation_object_path( - current_target.cell_id().as_bytes(), - current_incarnation.as_bytes(), - &object.digest, - object.kind, - ) - }) - .collect::>(); - let (pinned_root, pinned_paths) = - root(&layout, &pinned_target, pinned_incarnation, 96_000).await; - let (_, unpublished_paths) = root( - &layout, - ¤t_target, - IncarnationId::from_bytes([13; 16]), - 48_000, - ) - .await; - let orphan_target = target(identity, b"orphan"); - let (_, orphan_paths) = root(&layout, &orphan_target, orphan_incarnation, 64_000).await; - let current = idle_control( - current_target.cell_id(), - current_incarnation, - current_root, - code, - ); - let compacted_current = idle_control( - current_target.cell_id(), - current_incarnation, - compacted_root, - code, - ); - let pinned = idle_control( - pinned_target.cell_id(), - pinned_incarnation, - pinned_root, - code, - ); - - let descriptor = br#"{"runtime":"retention-test","version":1}"#; - let descriptor_digest = Digest::from_bytes(*blake3::hash(descriptor).as_bytes()); - let releases = ReleaseStore::new(layout.clone(), identity).unwrap(); - let initial_operation = RequestId::from_bytes([8; 16]); - let prepared = releases - .prepare( - descriptor, - descriptor_digest, - 0, - &format!("sha256:{}", "a".repeat(64)), - initial_operation, - ) - .await - .unwrap(); - let activating = releases - .start_activation(prepared.revision(), initial_operation) - .await - .unwrap(); - let ready = releases - .complete_activation(activating.revision(), initial_operation) - .await - .unwrap(); - - let pins = BackupPinStore::new( - layout.clone(), - identity, - ReplicaLimits::default(), - ReplicaHost::default(), - ) - .unwrap(); - pins.create( - RequestId::from_bytes([9; 16]), - 1_000, - pinned_catalog(&catalog).await, - vec![current.clone(), pinned.clone()], - ) - .await - .unwrap(); - - layout - .store() - .create_strict( - &layout.control_path(current.cell.as_bytes()), - Bytes::from(compacted_current.encode().unwrap()), - ) - .await - .unwrap(); - let mut tombstoned = pinned; - tombstoned.epoch += 1; - tombstoned.revision += 1; - tombstoned.progress += 1; - tombstoned.state = ControlState::Tombstoned; - tombstoned.encode().unwrap(); - layout - .store() - .create_strict( - &layout.control_path(tombstoned.cell.as_bytes()), - Bytes::from(tombstoned.encode().unwrap()), - ) - .await - .unwrap(); - - let orphan_release_body = b"orphan release descriptor"; - let orphan_release_digest = blake3::hash(orphan_release_body); - let orphan_release = put_content_addressed( - &layout, - layout.release_descriptor_path(orphan_release_digest.as_bytes()), - orphan_release_body, - ) - .await; - let orphan_catalog_body = b"orphan catalog page"; - let orphan_catalog_digest = blake3::hash(orphan_catalog_body); - let orphan_catalog = put_content_addressed( - &layout, - layout.catalog_object_path(orphan_catalog_digest.as_bytes()), - orphan_catalog_body, - ) - .await; - let orphan_pin_body = b"orphan pin object"; - let orphan_pin_digest = blake3::hash(orphan_pin_body); - let orphan_pin = put_content_addressed( - &layout, - layout.pin_object_path(orphan_pin_digest.as_bytes()), - orphan_pin_body, - ) - .await; - let unknown = put_content_addressed( - &layout, - Path::from(format!( - "{}/releases/future-format.bin", - layout.application_prefix() - )), - b"future", - ) - .await; - - let maintenance_operation = RequestId::from_bytes([10; 16]); - let prepared = releases - .prepare( - descriptor, - descriptor_digest, - ready.revision(), - &format!("sha256:{}", "a".repeat(64)), - maintenance_operation, - ) - .await - .unwrap(); - let maintenance = releases - .start_maintenance(prepared.revision(), maintenance_operation) - .await - .unwrap(); - assert!( - pins.create( - RequestId::from_bytes([11; 16]), - 2_000, - pinned_catalog(&catalog).await, - vec![current.clone(), tombstoned.clone()], - ) - .await - .is_err() - ); - - let collector = CellGarbageCollector::new( - layout.clone(), - identity, - ReplicaLimits::default(), - ReplicaHost::default(), - ) - .unwrap(); - let scratch = tempfile::TempDir::new().unwrap(); - let grace = collector - .collect( - &maintenance, - scratch.path(), - GarbageCollectionPolicy::new(0, 1, 100_000).unwrap(), - ) - .await - .unwrap(); - assert_eq!(grace.deleted_objects(), 0); - assert!(grace.grace_objects() >= (orphan_paths.len() + unpublished_paths.len()) as u64 + 3); - - let now_ms = i64::try_from( - std::time::SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap(); - let limited = collector - .collect( - &maintenance, - scratch.path(), - GarbageCollectionPolicy::new(now_ms.saturating_add(60_000), 1, 1).unwrap(), - ) - .await - .unwrap(); - assert_eq!(limited.deleted_objects(), 1); - assert!(limited.eligible_objects() > limited.deleted_objects()); - assert!(!limited.complete()); - - let report = collector - .collect( - &maintenance, - scratch.path(), - GarbageCollectionPolicy::new(now_ms.saturating_add(60_000), 1, 100_000).unwrap(), - ) - .await - .unwrap(); - assert!(report.complete()); - assert_eq!(report.current_controls(), 2); - assert_eq!(report.retained_pins(), 1); - assert!( - limited.deleted_objects() + report.deleted_objects() - >= (orphan_paths.len() + unpublished_paths.len()) as u64 + 3 - ); - assert!(report.reachable_objects() >= current_paths.len() as u64 + pinned_paths.len() as u64); - - for (target, incarnation, root, payload, name) in [ - ( - ¤t_target, - current_incarnation, - compacted_root, - 128_000, - "compacted-current", - ), - ( - ¤t_target, - current_incarnation, - current_root, - 128_000, - "pinned-previous-representation", - ), - ( - &pinned_target, - pinned_incarnation, - pinned_root, - 96_000, - "pinned", - ), - ] { - let replica = crab_ltx::CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *incarnation.as_bytes(), - ReplicaLimits::default(), - ) - .unwrap(); - let verified = replica.open_root(&root).await.unwrap(); - let restored = scratch.path().join(format!("{name}-reopened.sqlite")); - assert_eq!(verified.restore(&restored).await.unwrap(), root.position); - let connection = rusqlite::Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT length(value) FROM values_", [], |row| row - .get::<_, usize>(0)) - .unwrap(), - payload - ); - } - - for path in current_paths - .into_iter() - .chain(compacted_paths) - .chain(pinned_paths) - { - layout.store().head(&path).await.unwrap(); - } - for path in orphan_paths.into_iter().chain(unpublished_paths).chain([ - orphan_release, - orphan_catalog, - orphan_pin, - ]) { - assert!(matches!( - layout.store().head(&path).await, - Err(StorageError::NotFound { .. }) - )); - } - layout.store().head(&unknown).await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn maintenance_collection_rejects_an_owned_current_cell_before_deleting() { - collection_rejects_unscanned_or_owned_cells(false).await; -} - -#[tokio::test(flavor = "multi_thread")] -async fn maintenance_collection_rejects_another_tenants_catalog_before_deleting() { - collection_rejects_unscanned_or_owned_cells(true).await; -} - -async fn collection_rejects_unscanned_or_owned_cells(foreign_tenant: bool) { - let identity = identity(); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - *identity.application().as_bytes(), - ); - let code = Digest::from_bytes([12; 32]); - let cell_identity = if foreign_tenant { - ApplicationIdentity::new(TenantId::from_bytes([99; 16]), identity.application()) - } else { - identity - }; - let target = target(cell_identity, b"owned"); - CellCatalog::new(layout.clone(), cell_identity.tenant()) - .provision(CatalogEntry::new(&target, CatalogRole::Repository, code, 1).unwrap()) - .await - .unwrap(); - let control = Control::initial( - target.cell_id(), - IncarnationId::from_bytes([13; 16]), - Owner { - session: SessionId::from_bytes([14; 16]), - endpoint: "https://owner.internal:8789".into(), - }, - code, - 1, - ) - .unwrap(); - layout - .store() - .create_strict( - &layout.control_path(control.cell.as_bytes()), - Bytes::from(control.encode().unwrap()), - ) - .await - .unwrap(); - - let orphan_body = b"old unreachable descriptor"; - let orphan_digest = blake3::hash(orphan_body); - let orphan = put_content_addressed( - &layout, - layout.release_descriptor_path(orphan_digest.as_bytes()), - orphan_body, - ) - .await; - let descriptor = br#"{"runtime":"retention-owner-test","version":1}"#; - let descriptor_digest = Digest::from_bytes(*blake3::hash(descriptor).as_bytes()); - let releases = ReleaseStore::new(layout.clone(), identity).unwrap(); - let operation = RequestId::from_bytes([15; 16]); - let prepared = releases - .prepare( - descriptor, - descriptor_digest, - 0, - &format!("sha256:{}", "b".repeat(64)), - operation, - ) - .await - .unwrap(); - let maintenance = releases - .start_maintenance(prepared.revision(), operation) - .await - .unwrap(); - let now_ms = i64::try_from( - std::time::SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap(); - let scratch = tempfile::TempDir::new().unwrap(); - let error = CellGarbageCollector::new( - layout.clone(), - identity, - ReplicaLimits::default(), - ReplicaHost::default(), - ) - .unwrap() - .collect( - &maintenance, - scratch.path(), - GarbageCollectionPolicy::new(now_ms.saturating_add(60_000), 1, 100).unwrap(), - ) - .await - .unwrap_err(); - - let expected = if foreign_tenant { - "maintenance collection requires a single-tenant catalog" - } else { - "a current Cell still has an owner during maintenance" - }; - assert!(matches!(error, Error::Retention(message) if message == expected)); - layout.store().head(&orphan).await.unwrap(); -} - -#[test] -fn immutable_path_classifier_accepts_only_version_one_layouts() { - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - [2; 16], - ); - let prefix = layout.application_prefix(); - assert!(immutable_candidate( - &prefix, - &layout.incarnation_object_path(&[3; 32], &[4; 16], &[5; 32], CellObjectKind::Ltx), - )); - assert!(immutable_candidate( - &prefix, - &layout.catalog_object_path(&[6; 32]), - )); - assert!(!immutable_candidate( - &prefix, - &layout.control_path(&[3; 32]), - )); - assert!(!immutable_candidate( - &prefix, - &Path::from(format!("{prefix}/releases/future-format.bin")), - )); - - assert!(GarbageCollectionPolicy::new(0, 0, 1).is_err()); - assert!(GarbageCollectionPolicy::new(0, 1, 0).is_err()); - assert!(GarbageCollectionPolicy::new(0, 1, 100_001).is_err()); -} diff --git a/crates/crab-cell-runtime/src/registry.rs b/crates/crab-cell-runtime/src/registry.rs deleted file mode 100644 index f1533a9b3..000000000 --- a/crates/crab-cell-runtime/src/registry.rs +++ /dev/null @@ -1,39 +0,0 @@ -//! Compiled primitive registry: descriptors, schemas, handlers, and builder. -mod descriptor; -pub use builder::{Registry, RegistryBuilder}; -pub use handlers::{ - CellModule, Command, CommandContext, CommandInvocation, CommandResult, Query, QueryContext, - QueryInvocation, -}; -pub use schemas::{ - BuildDescriptor, MigrationDescriptor, MigrationPlan, ModuleDescriptor, NamespaceDescriptor, - OperationDescriptor, RegistryError, RetainedCodeDescriptor, -}; -mod builder; -mod handlers; -mod schemas; - -use descriptor::{encode_release, requires_persisted_work_inventory, verify_rolling_compatibility}; - -use crate::cell::catalog::CatalogRole; -use crate::cell::executor::{HandlerOutcome, MutationIdentity}; -use crate::client::{CellClient, Committed, InvocationError}; -use crate::codec::WireValue; -use crate::codec::{decode_wire, encode_wire}; -use crate::identity::{ApplicationId, CellId, CellTarget, Digest, NamespaceId, TenantId}; -use crate::peer::EffectPeerClient; -use crate::primitives::activity_pool::BlockingActivityReservation; -use crate::primitives::effects::EffectBatch; -use crate::primitives::effects::{ - EffectCommandIntent, EffectModule, EffectRunOutcome, EffectSupervisor, EffectSupervisorError, -}; -use crate::primitives::maintenance::{ - MaintenanceModule, MaintenanceTickCommand, MaintenanceTickOutcome, MaintenanceTickRequest, -}; -use crate::primitives::sql::{SqlBatch, SqlResultSet, sql_batch, sql_query_batch}; -use crate::primitives::workflow::{ - ActivityContext, ActivityExecution, ActivityHandler, ActivityRunOutcome, ActivitySupervisor, - ActivitySupervisorError, ActivitySupport, BlockingActivityHandler, WorkflowActivities, - WorkflowActivityModule, WorkflowDefinition, -}; -use crate::{Error, Result}; diff --git a/crates/crab-cell-runtime/src/registry/builder.rs b/crates/crab-cell-runtime/src/registry/builder.rs deleted file mode 100644 index 666fb1975..000000000 --- a/crates/crab-cell-runtime/src/registry/builder.rs +++ /dev/null @@ -1,1074 +0,0 @@ -//! Registry construction and the compiled module registry. - -use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet}; - -use rusqlite::{Connection, Transaction}; - -use super::*; -use crate::registry::handlers::{ - ActivityFunction, ActivityFuture, ActivityRunner, CommandHandler, EffectRunner, - MaintenanceRunner, QueryHandler, -}; -use crate::registry::schemas::{ - MAX_DESCRIPTOR_BYTES, MAX_MODULES, operation_descriptors, typed_activity, - typed_activity_runner, typed_blocking_activity, typed_command, typed_effect, typed_maintenance, - typed_query, valid_name, validate_build, validate_cron_bindings, validate_invocation, - validate_maintenance_bindings, validate_module, validate_namespaces, - validate_primitive_bindings, validate_queue_bindings, validate_workflow_effect_targets, -}; - -/// Mutable startup-only collector for descriptors and compiled bindings. -pub struct RegistryBuilder { - pub(super) build: BuildDescriptor, - pub(super) modules: Vec<&'static ModuleDescriptor>, - pub(super) commands: BTreeMap, - pub(super) queries: BTreeMap, - pub(super) workflow_definitions: HashMap<(String, [u8; 32]), Vec>, - pub(super) activities: BTreeMap, - pub(super) activity_claims: BTreeSet, - pub(super) activity_runners: HashMap, - pub(super) queue_bindings: Vec, - pub(super) blob_bindings: Vec, - pub(super) cron_bindings: Vec, - pub(super) maintenance_bindings: - BTreeMap<&'static str, Option>, - pub(super) maintenance_runners: BTreeMap<&'static str, MaintenanceRunner>, - pub(super) effect_runners: BTreeMap<&'static str, EffectRunner>, - pub(super) maintenance_operations: BTreeMap<&'static str, (u32, u32)>, - pub(super) effect_operations: BTreeMap<&'static str, (u32, u32, u32, u32, u32)>, - pub(super) activity_operations: HashMap, -} - -impl RegistryBuilder { - /// Starts an empty builder for one build descriptor. - #[must_use] - pub fn new(build: BuildDescriptor) -> Self { - Self { - build, - modules: Vec::new(), - commands: BTreeMap::new(), - queries: BTreeMap::new(), - workflow_definitions: HashMap::new(), - activities: BTreeMap::new(), - activity_claims: BTreeSet::new(), - activity_runners: HashMap::new(), - queue_bindings: Vec::new(), - blob_bindings: Vec::new(), - cron_bindings: Vec::new(), - maintenance_bindings: BTreeMap::new(), - maintenance_runners: BTreeMap::new(), - effect_runners: BTreeMap::new(), - maintenance_operations: BTreeMap::new(), - effect_operations: BTreeMap::new(), - activity_operations: HashMap::new(), - } - } - - /// Registers one static module and all bindings it contributes. - pub fn register(&mut self, module: M) -> std::result::Result<(), RegistryError> { - let descriptor = module.descriptor(); - if M::NAME != descriptor.name { - return Err(Error::Registry( - "module constant and descriptor name differ", - )); - } - module.register(self)?; - self.modules.push(descriptor); - Ok(()) - } - - /// Binds one descriptor command key to its monomorphized typed function. - pub fn bind_command(&mut self) -> std::result::Result<(), RegistryError> { - let key = BindingKey::new(C::MODULE, C::ID, C::CODEC_VERSION)?; - if self.commands.insert(key, typed_command::).is_some() { - return Err(Error::Registry("duplicate command binding")); - } - Ok(()) - } - - /// Binds one descriptor query key to its monomorphized typed function. - pub fn bind_query(&mut self) -> std::result::Result<(), RegistryError> { - let key = BindingKey::new(Q::MODULE, Q::ID, Q::CODEC_VERSION)?; - if self.queries.insert(key, typed_query::).is_some() { - return Err(Error::Registry("duplicate query binding")); - } - Ok(()) - } - - pub(crate) fn bind_queue_module( - &mut self, - module: &'static str, - namespace: NamespaceId, - send_command_id: u32, - codec_version: u32, - dead_letter: Option, - ) -> Result<()> { - if self - .queue_bindings - .iter() - .any(|binding| binding.namespace == namespace) - { - return Err(Error::Registry("duplicate Queue module binding")); - } - self.queue_bindings.push(QueueBinding { - module, - namespace, - send_command_id, - codec_version, - dead_letter, - }); - Ok(()) - } - - pub(crate) fn bind_cron_module( - &mut self, - module: &'static str, - namespace: NamespaceId, - targets: &'static [crate::primitives::cron::CronTarget], - ) -> Result<()> { - if self - .cron_bindings - .iter() - .any(|binding| binding.namespace == namespace) - { - return Err(Error::Registry("duplicate Cron module binding")); - } - self.cron_bindings.push(CronBinding { - module, - namespace, - targets, - }); - Ok(()) - } - - pub(crate) fn bind_blob_module( - &mut self, - module: &'static str, - namespace: NamespaceId, - ) -> Result<()> { - if self - .blob_bindings - .iter() - .any(|binding| binding.namespace == namespace) - { - return Err(Error::Registry("duplicate Blob module binding")); - } - self.blob_bindings - .push(PrimitiveBinding { module, namespace }); - Ok(()) - } - - pub(crate) fn bind_maintenance_module( - &mut self, - module: &'static str, - queue_dead_letter: Option, - ) -> Result<()> { - if self - .maintenance_bindings - .insert(module, queue_dead_letter) - .is_some() - { - return Err(Error::Registry("duplicate maintenance module binding")); - } - Ok(()) - } - - pub(crate) fn bind_maintenance_runner(&mut self) -> Result<()> { - if self - .maintenance_runners - .insert(M::MODULE, typed_maintenance::) - .is_some() - { - return Err(Error::Registry("duplicate maintenance runner binding")); - } - self.maintenance_operations - .insert(M::MODULE, (M::TICK_COMMAND_ID, M::CODEC_VERSION)); - Ok(()) - } - - pub(crate) fn bind_effect_runner(&mut self) -> Result<()> { - if self - .effect_runners - .insert(M::MODULE, typed_effect::) - .is_some() - { - return Err(Error::Registry("duplicate effect runner binding")); - } - self.effect_operations.insert( - M::MODULE, - ( - M::CLAIM_COMMAND_ID, - M::LEASE_COMMAND_ID, - M::VALIDATE_QUERY_ID, - M::STATUS_QUERY_ID, - M::CODEC_VERSION, - ), - ); - Ok(()) - } - - pub(crate) fn bind_activity_runner(&mut self) -> Result<()> { - if self - .activity_runners - .insert(M::NAMESPACE, typed_activity_runner::) - .is_some() - { - return Err(Error::Registry("duplicate activity runner binding")); - } - self.activity_operations.insert( - M::NAMESPACE, - ( - M::ACTIVITY_CLAIM_COMMAND_ID, - M::ACTIVITY_COMPLETE_COMMAND_ID, - M::ACTIVITY_EXTEND_COMMAND_ID, - M::ACTIVITY_VALIDATE_QUERY_ID, - M::CODEC_VERSION, - ), - ); - Ok(()) - } - - /// Binds one descriptor digest to its statically linked transition function. - pub fn bind_workflow_definition( - &mut self, - module: &'static str, - definition: &'static dyn WorkflowDefinition, - ) -> std::result::Result<(), RegistryError> { - let digest = *definition.digest().as_bytes(); - let targets = definition.effect_targets(); - let unique_targets = targets.iter().copied().collect::>(); - if !valid_name(module) - || digest.iter().all(|byte| *byte == 0) - || unique_targets.len() != targets.len() - || self - .workflow_definitions - .insert((module.to_owned(), digest), targets.to_vec()) - .is_some() - { - return Err(Error::Registry("invalid workflow definition binding")); - } - Ok(()) - } - - /// Binds one declared activity type to its statically linked Rust future. - pub fn bind_activity( - &mut self, - module: &'static str, - definition: Digest, - ) -> std::result::Result<(), RegistryError> { - let key = ActivityKey::new(module, definition, A::TYPE)?; - if self - .activities - .insert(key, ActivityFunction::Async(typed_activity::)) - .is_some() - { - return Err(Error::Registry("duplicate activity binding")); - } - Ok(()) - } - - /// Binds one descriptor activity to a node-owned blocking callback. - pub fn bind_blocking_activity( - &mut self, - module: &'static str, - definition: Digest, - ) -> std::result::Result<(), RegistryError> { - let key = ActivityKey::new(module, definition, A::TYPE)?; - if self - .activities - .insert( - key, - ActivityFunction::Blocking(typed_blocking_activity::), - ) - .is_some() - { - return Err(Error::Registry("duplicate activity binding")); - } - Ok(()) - } - - /// Binds the claim-time activity inventory to the same compiled definitions. - pub fn bind_activity_inventory( - &mut self, - module: &'static str, - definition: Digest, - activity_types: &'static [&'static str], - ) -> std::result::Result<(), RegistryError> { - if activity_types.is_empty() { - return Err(Error::Registry("activity claim inventory is empty")); - } - for activity in activity_types { - if !self - .activity_claims - .insert(ActivityKey::new(module, definition, activity)?) - { - return Err(Error::Registry("duplicate activity claim binding")); - } - } - Ok(()) - } - - /// Freezes registration after validating inventory and canonical bytes. - pub fn finish(mut self) -> std::result::Result { - validate_build(&self.build)?; - if self.modules.is_empty() || self.modules.len() > MAX_MODULES { - return Err(Error::Registry("module count must be in 1..=128")); - } - self.modules.sort_by_key(|module| module.name); - let mut module_names = HashSet::with_capacity(self.modules.len()); - let mut expected_commands = BTreeSet::new(); - let mut expected_queries = BTreeSet::new(); - let mut namespace_owners = HashMap::new(); - for module in &self.modules { - if !module_names.insert(module.name) { - return Err(Error::Registry("duplicate module name")); - } - validate_module( - module, - &mut expected_commands, - &mut expected_queries, - &mut namespace_owners, - )?; - } - if expected_commands != self.commands.keys().cloned().collect() - || expected_queries != self.queries.keys().cloned().collect() - { - return Err(Error::Registry("descriptor and function bindings differ")); - } - let expected_workflows = self - .modules - .iter() - .flat_map(|module| { - module - .workflow_definitions - .iter() - .map(|digest| (module.name.to_owned(), *digest.as_bytes())) - }) - .collect::>(); - if expected_workflows - != self - .workflow_definitions - .keys() - .cloned() - .collect::>() - { - return Err(Error::Registry( - "descriptor and workflow definition bindings differ", - )); - } - let expected_activities = self - .modules - .iter() - .flat_map(|module| { - module.workflow_definitions.iter().flat_map(|definition| { - module.activity_types.iter().map(|activity| ActivityKey { - module: module.name.to_owned(), - definition: *definition.as_bytes(), - activity: (*activity).to_owned(), - }) - }) - }) - .collect::>(); - if expected_activities != self.activities.keys().cloned().collect() - || expected_activities != self.activity_claims - { - return Err(Error::Registry("descriptor and activity bindings differ")); - } - validate_namespaces(&namespace_owners)?; - validate_workflow_effect_targets( - &self.workflow_definitions, - &namespace_owners, - &self.modules, - )?; - validate_queue_bindings(&self.queue_bindings, &namespace_owners, &self.modules)?; - validate_primitive_bindings( - &self.blob_bindings, - CatalogRole::Blob, - "Blob descriptors and compiled bindings differ", - &namespace_owners, - )?; - validate_cron_bindings(&self.cron_bindings, &namespace_owners, &self.modules)?; - validate_maintenance_bindings(&self.maintenance_bindings, &self.queue_bindings)?; - if self - .maintenance_bindings - .keys() - .copied() - .collect::>() - != self.maintenance_runners.keys().copied().collect() - { - return Err(Error::Registry( - "maintenance metadata and runner bindings differ", - )); - } - let expected_activity_runners = self - .modules - .iter() - .filter(|module| !module.activity_types.is_empty()) - .flat_map(|module| { - module - .namespaces - .iter() - .filter(|namespace| namespace.role == CatalogRole::Workflow) - .map(|namespace| namespace.id) - }) - .collect::>(); - if expected_activity_runners != self.activity_runners.keys().copied().collect() { - return Err(Error::Registry( - "activity metadata and runner bindings differ", - )); - } - - let command_descriptors = operation_descriptors(&self.modules, |module| module.commands); - let query_descriptors = operation_descriptors(&self.modules, |module| module.queries); - let module_schemas = self - .modules - .iter() - .map(|module| { - ( - module.name.to_owned(), - (module.schema_min, module.schema_max), - ) - }) - .collect(); - let module_migrations = self - .modules - .iter() - .map(|module| (module.name, module.migrations)) - .collect(); - - let (release_bytes, module_codes) = encode_release(&self.build, &self.modules)?; - let module_retained_codes = self - .modules - .iter() - .map(|module| { - let current = module_codes - .get(module.name) - .copied() - .ok_or(Error::Registry("module code is unavailable"))?; - if module - .retained_codes - .iter() - .any(|retained| retained.code == current) - { - return Err(Error::Registry( - "current module code is retained as predecessor", - )); - } - Ok((module.name, module.retained_codes)) - }) - .collect::>>()?; - let mut supported_codes = module_codes.values().copied().collect::>(); - supported_codes.extend( - module_retained_codes - .values() - .flat_map(|retained| retained.iter().map(|descriptor| descriptor.code)), - ); - if supported_codes.len() > MAX_MODULES { - return Err(Error::Registry( - "current and retained module code inventory exceeds 128", - )); - } - if release_bytes.len() > MAX_DESCRIPTOR_BYTES { - return Err(Error::Registry("release descriptor exceeds 256 KiB")); - } - let release_digest = Digest::from_bytes(*blake3::hash(&release_bytes).as_bytes()); - let blocking_modules = self - .activities - .iter() - .filter_map(|(key, handler)| { - matches!(handler, ActivityFunction::Blocking(_)).then_some(key.module.as_str()) - }) - .collect::>(); - let blocking_activity_namespaces = namespace_owners - .iter() - .filter_map(|(namespace, (module, descriptor))| { - (descriptor.role == CatalogRole::Workflow && blocking_modules.contains(*module)) - .then_some(*namespace) - }) - .collect(); - let module_names = self.modules.iter().map(|module| module.name).collect(); - Ok(Registry { - release_bytes, - release_digest, - module_codes, - module_schemas, - module_migrations, - module_retained_codes, - module_names, - commands: self.commands, - command_descriptors, - queries: self.queries, - query_descriptors, - namespace_modules: namespace_owners, - activities: self.activities, - blocking_activity_namespaces, - activity_runners: self.activity_runners, - maintenance_runners: self.maintenance_runners, - effect_runners: self.effect_runners, - maintenance_operations: self.maintenance_operations, - effect_operations: self.effect_operations, - activity_operations: self.activity_operations, - }) - } -} - -#[derive(Clone, Copy)] -pub(super) struct QueueBinding { - pub(super) module: &'static str, - pub(super) namespace: NamespaceId, - pub(super) send_command_id: u32, - pub(super) codec_version: u32, - pub(super) dead_letter: Option, -} - -#[derive(Clone, Copy)] -pub(super) struct CronBinding { - pub(super) module: &'static str, - pub(super) namespace: NamespaceId, - pub(super) targets: &'static [crate::primitives::cron::CronTarget], -} - -#[derive(Clone, Copy)] -pub(super) struct PrimitiveBinding { - pub(super) module: &'static str, - pub(super) namespace: NamespaceId, -} - -/// Immutable compiled registry shared by runtime and release inspection. -#[derive(Clone)] -pub struct Registry { - pub(super) release_bytes: Vec, - pub(super) release_digest: Digest, - pub(super) module_codes: BTreeMap, - pub(super) module_names: Vec<&'static str>, - pub(super) module_schemas: BTreeMap, - pub(super) module_migrations: BTreeMap<&'static str, &'static [MigrationDescriptor]>, - pub(super) module_retained_codes: BTreeMap<&'static str, &'static [RetainedCodeDescriptor]>, - pub(super) commands: BTreeMap, - pub(super) command_descriptors: BTreeMap, - pub(super) queries: BTreeMap, - pub(super) query_descriptors: BTreeMap, - pub(super) namespace_modules: HashMap, - pub(super) activities: BTreeMap, - pub(super) blocking_activity_namespaces: HashSet, - pub(super) activity_runners: HashMap, - pub(super) maintenance_runners: BTreeMap<&'static str, MaintenanceRunner>, - pub(super) effect_runners: BTreeMap<&'static str, EffectRunner>, - pub(super) maintenance_operations: BTreeMap<&'static str, (u32, u32)>, - pub(super) effect_operations: BTreeMap<&'static str, (u32, u32, u32, u32, u32)>, - pub(super) activity_operations: HashMap, -} - -mod run; - -impl Registry { - /// Returns the canonical release descriptor bytes. - #[must_use] - pub fn release_bytes(&self) -> &[u8] { - &self.release_bytes - } - - /// Returns the digest of the canonical release descriptor. - #[must_use] - pub const fn release_digest(&self) -> Digest { - self.release_digest - } - - /// Returns the module code digest registered under `module`. - #[must_use] - pub fn module_code(&self, module: &str) -> Option { - self.module_codes.get(module).copied() - } - - /// Returns the schema range compiled for one registered module. - #[must_use] - pub fn module_schema_range(&self, module: &str) -> Option<(u32, u32)> { - self.module_schemas.get(module).copied() - } - - /// Returns the compiled module names in sorted order. - #[must_use] - pub fn module_names(&self) -> &[&'static str] { - &self.module_names - } - - /// Returns the sorted module-code inventory advertised by eligible nodes. - #[must_use] - pub fn module_digests(&self) -> Vec { - let mut digests = self.module_codes.values().copied().collect::>(); - digests.extend( - self.module_retained_codes - .values() - .flat_map(|retained| retained.iter().map(|descriptor| descriptor.code)), - ); - digests.sort_unstable_by(|left, right| left.as_bytes().cmp(right.as_bytes())); - digests.dedup(); - digests - } - - /// Verifies that this registry can replace one previously selected release online. - /// - /// Every executable and persisted-work contract from the predecessor must remain - /// available. Removing one requires an offline maintenance activation. - pub fn verify_rolling_from(&self, previous: &[u8]) -> Result<()> { - verify_rolling_compatibility(previous, &self.release_bytes) - } - - /// Reports whether an offline rollout removes or narrows a predecessor contract. - /// - /// A true result requires complete persisted-work admission before the - /// predecessor implementation or codec can be removed. - pub fn requires_persisted_work_inventory_from(&self, previous: &[u8]) -> Result { - requires_persisted_work_inventory(previous, &self.release_bytes) - } - - /// Reports whether this binary can execute one authoritative Cell pair. - #[must_use] - pub fn supports_cell( - &self, - namespace: NamespaceId, - role: CatalogRole, - code: Digest, - schema: u32, - ) -> bool { - let Some((module, descriptor)) = self.namespace_modules.get(&namespace) else { - return false; - }; - let Some((schema_min, schema_max)) = self.module_schemas.get(*module) else { - return false; - }; - if descriptor.role != role || !(*schema_min..=*schema_max).contains(&schema) { - return false; - } - self.supports_module_code(module, code, schema) - } - - /// Reports whether one Cell already uses the module's target code and schema. - #[must_use] - pub fn is_current_cell( - &self, - namespace: NamespaceId, - role: CatalogRole, - code: Digest, - schema: u32, - ) -> bool { - self.current_cell_version(namespace, role) == Some((code, schema)) - } - - /// Returns the target code and schema for one registered namespace role. - #[must_use] - pub fn current_cell_version( - &self, - namespace: NamespaceId, - role: CatalogRole, - ) -> Option<(Digest, u32)> { - let (module, descriptor) = self.namespace_modules.get(&namespace)?; - if descriptor.role != role { - return None; - } - Some(( - *self.module_codes.get(*module)?, - self.module_schemas.get(*module)?.1, - )) - } - - /// Selects the next compiled migration for one exact Cell code/schema pair. - pub fn next_migration( - &self, - namespace: NamespaceId, - code: Digest, - schema: u32, - ) -> Result> { - let (module, _) = self - .namespace_modules - .get(&namespace) - .ok_or(Error::Registry("namespace is unavailable"))?; - let current_code = self - .module_codes - .get(*module) - .copied() - .ok_or(Error::Registry("module code is unavailable"))?; - let (schema_min, schema_max) = self - .module_schemas - .get(*module) - .copied() - .ok_or(Error::Registry("module schema range is unavailable"))?; - if !(schema_min..=schema_max).contains(&schema) - || (code != current_code - && !self - .module_retained_codes - .get(module) - .is_some_and(|retained| { - retained.iter().any(|descriptor| { - descriptor.code == code - && (descriptor.schema_min..=descriptor.schema_max).contains(&schema) - }) - })) - { - return Err(Error::Registry( - "Cell code/schema is not executable by this registry", - )); - } - let (to_schema, migration) = match schema.checked_add(1).filter(|next| *next <= schema_max) - { - Some(to_schema) => { - let migration = self - .module_migrations - .get(module) - .and_then(|migrations| { - migrations - .iter() - .find(|migration| migration.version == to_schema) - }) - .copied() - .ok_or(Error::Registry("next migration is unavailable"))?; - (to_schema, Some(migration)) - } - None if code != current_code => (schema, None), - None => return Ok(None), - }; - Ok(Some(MigrationPlan { - module, - from_code: code, - to_code: current_code, - from_schema: schema, - to_schema, - migration, - })) - } - - /// Maps one registered internal command to its exact fleet authorization. - #[must_use] - pub fn internal_command_action( - &self, - namespace: NamespaceId, - command_id: u32, - codec_version: u32, - ) -> Option<&'static str> { - let module = self.namespace_modules.get(&namespace)?.0; - if self - .maintenance_operations - .get(module) - .is_some_and(|&(tick, codec)| tick == command_id && codec == codec_version) - { - return Some("cell.scheduler.tick"); - } - if self - .effect_operations - .get(module) - .is_some_and(|&(claim, lease, _, _, codec)| { - codec == codec_version && matches!(command_id, id if id == claim || id == lease) - }) - { - return Some("cell.effect.source"); - } - if self.activity_operations.get(&namespace).is_some_and( - |&(claim, complete, extend, _, codec)| { - codec == codec_version - && matches!(command_id, id if id == claim || id == complete || id == extend) - }, - ) { - return Some("cell.activity.source"); - } - None - } - - /// Maps one registered internal query to its exact fleet authorization. - #[must_use] - pub fn internal_query_action( - &self, - namespace: NamespaceId, - query_id: u32, - codec_version: u32, - ) -> Option<&'static str> { - let module = self.namespace_modules.get(&namespace)?.0; - if self - .effect_operations - .get(module) - .is_some_and(|&(_, _, validate, status, codec)| { - (validate == query_id || status == query_id) && codec == codec_version - }) - { - return Some("cell.effect.source"); - } - if self - .activity_operations - .get(&namespace) - .is_some_and(|&(_, _, _, validate, codec)| { - validate == query_id && codec == codec_version - }) - { - return Some("cell.activity.source"); - } - None - } - - /// Returns the compiled owner and descriptor for one namespace. - pub fn namespace_contract( - &self, - namespace: NamespaceId, - ) -> Option<(&'static str, NamespaceDescriptor)> { - self.namespace_modules.get(&namespace).copied() - } - - /// Returns the number of namespaces compiled into this release. - #[must_use] - pub fn namespace_count(&self) -> usize { - self.namespace_modules.len() - } - - /// Returns the compiled contract for a typed command in one namespace. - /// - /// Fails when the command's stable module, ID, or codec is absent. - pub fn command_contract( - &self, - namespace: NamespaceId, - ) -> Result { - self.operation_contract( - namespace, - C::MODULE, - C::ID, - C::CODEC_VERSION, - &self.command_descriptors, - ) - } - - /// Returns the compiled contract for a typed query in one namespace. - /// - /// Fails when the query's stable module, ID, or codec is absent. - pub fn query_contract(&self, namespace: NamespaceId) -> Result { - self.operation_contract( - namespace, - Q::MODULE, - Q::ID, - Q::CODEC_VERSION, - &self.query_descriptors, - ) - } - - pub(crate) fn routed_command_contract( - &self, - namespace: NamespaceId, - id: u32, - codec_version: u32, - ) -> Result<(&'static str, OperationDescriptor)> { - self.routed_operation_contract(namespace, id, codec_version, &self.command_descriptors) - } - - pub(crate) fn routed_query_contract( - &self, - namespace: NamespaceId, - id: u32, - codec_version: u32, - ) -> Result<(&'static str, OperationDescriptor)> { - self.routed_operation_contract(namespace, id, codec_version, &self.query_descriptors) - } - - pub(super) fn routed_operation_contract( - &self, - namespace: NamespaceId, - id: u32, - codec_version: u32, - descriptors: &BTreeMap, - ) -> Result<(&'static str, OperationDescriptor)> { - let module = self - .namespace_modules - .get(&namespace) - .map(|(module, _)| *module) - .ok_or(Error::Registry("operation namespace is unavailable"))?; - let operation = - self.operation_contract(namespace, module, id, codec_version, descriptors)?; - Ok((module, operation)) - } - - pub(super) fn operation_contract( - &self, - namespace: NamespaceId, - module: &'static str, - id: u32, - codec_version: u32, - descriptors: &BTreeMap, - ) -> Result { - if self - .namespace_modules - .get(&namespace) - .map(|(owner, _)| *owner) - != Some(module) - { - return Err(Error::Registry("operation module does not own namespace")); - } - let key = BindingKey::new(module, id, codec_version)?; - let operation = descriptors - .get(&key) - .copied() - .ok_or(Error::Registry("operation descriptor is unavailable"))?; - Ok(operation) - } - - pub(crate) fn supports_module_code(&self, module: &str, code: Digest, schema: u32) -> bool { - if self - .module_schemas - .get(module) - .is_none_or(|(schema_min, schema_max)| !(*schema_min..=*schema_max).contains(&schema)) - { - return false; - } - self.module_codes.get(module) == Some(&code) - || self - .module_retained_codes - .get(module) - .is_some_and(|retained| { - retained.iter().any(|descriptor| { - descriptor.code == code - && (descriptor.schema_min..=descriptor.schema_max).contains(&schema) - }) - }) - } - - /// Executes one already bounded command through its exact compiled binding. - pub fn execute_command( - &self, - transaction: &Transaction<'_>, - invocation: CommandInvocation<'_>, - ) -> Result { - let issued_at_ms = invocation.now_ms; - self.execute_command_with_issue_time(transaction, invocation, issued_at_ms) - } - - pub(crate) fn execute_command_with_issue_time( - &self, - transaction: &Transaction<'_>, - invocation: CommandInvocation<'_>, - issued_at_ms: i64, - ) -> Result { - if invocation.sequence == 0 || invocation.now_ms < 0 { - return Err(Error::Command("invalid registered command context")); - } - let key = BindingKey::new( - invocation.module, - invocation.operation_id, - invocation.codec_version, - )?; - let operation = self - .command_descriptors - .get(&key) - .ok_or(Error::Registry("command descriptor is unavailable"))?; - if self - .namespace_modules - .get(&invocation.target.namespace()) - .map(|(module, _)| *module) - != Some(invocation.module) - { - return Err(Error::Registry("operation module does not own namespace")); - } - validate_invocation(operation, invocation.schema, invocation.input.len())?; - let handler = self - .commands - .get(&key) - .ok_or(Error::Registry("command binding is unavailable"))?; - let mut context = CommandContext { - transaction, - target: invocation.target.clone(), - effect_targets: self - .namespace_modules - .get(&invocation.target.namespace()) - .ok_or(Error::Registry("command target namespace is unavailable")) - .map(|(_, descriptor)| descriptor.effect_targets)?, - sequence: invocation.sequence, - now_ms: invocation.now_ms, - issued_at_ms, - input_limit: operation.input_limit, - output_limit: operation.output_limit, - effects: None, - }; - let outcome = handler(&mut context, invocation.input)?; - let output = match &outcome { - HandlerOutcome::Success(output) | HandlerOutcome::Rejected(output) => output, - }; - if output.len() > operation.output_limit as usize { - return Err(Error::Command("registered operation output exceeds limit")); - } - Ok(outcome) - } - - /// Executes one already bounded query through its exact compiled binding. - pub fn execute_query( - &self, - connection: &Connection, - invocation: QueryInvocation<'_>, - ) -> Result> { - if invocation.now_ms < 0 { - return Err(Error::Command("invalid registered query context")); - } - let key = BindingKey::new( - invocation.module, - invocation.operation_id, - invocation.codec_version, - )?; - let operation = self - .query_descriptors - .get(&key) - .ok_or(Error::Registry("query descriptor is unavailable"))?; - validate_invocation(operation, invocation.schema, invocation.input.len())?; - let handler = self - .queries - .get(&key) - .ok_or(Error::Registry("query binding is unavailable"))?; - let mut context = QueryContext { - connection, - cell: invocation.cell, - commit_sequence: invocation.commit_sequence, - now_ms: invocation.now_ms, - input_limit: operation.input_limit, - output_limit: operation.output_limit, - }; - let output = handler(&mut context, invocation.input)?; - if output.len() > operation.output_limit as usize { - return Err(Error::Command("registered operation output exceeds limit")); - } - Ok(output) - } -} - -#[derive(Clone, Debug, PartialEq, Eq, PartialOrd, Ord)] -pub(super) struct BindingKey { - pub(super) module: String, - pub(super) id: u32, - pub(super) codec_version: u32, -} - -#[derive(Clone, Debug, PartialEq, Eq, PartialOrd, Ord)] -pub(super) struct ActivityKey { - pub(super) module: String, - pub(super) definition: [u8; 32], - pub(super) activity: String, -} - -impl ActivityKey { - pub(super) fn new(module: &str, definition: Digest, activity: &str) -> Result { - if !valid_name(module) - || !valid_name(activity) - || definition.as_bytes().iter().all(|byte| *byte == 0) - { - return Err(Error::Registry("invalid activity binding key")); - } - Ok(Self { - module: module.to_owned(), - definition: *definition.as_bytes(), - activity: activity.to_owned(), - }) - } -} - -impl BindingKey { - pub(super) fn new(module: &str, id: u32, codec_version: u32) -> Result { - if !valid_name(module) || id == 0 || codec_version == 0 { - return Err(Error::Registry("invalid function binding key")); - } - Ok(Self { - module: module.to_owned(), - id, - codec_version, - }) - } -} diff --git a/crates/crab-cell-runtime/src/registry/builder/run.rs b/crates/crab-cell-runtime/src/registry/builder/run.rs deleted file mode 100644 index 0e07097d6..000000000 --- a/crates/crab-cell-runtime/src/registry/builder/run.rs +++ /dev/null @@ -1,178 +0,0 @@ -//! Running module work through a compiled registry. -//! -//! The registry decides which module owns a namespace, which runner can -//! carry its Activities, maintenance jobs, and Effects, and whether an -//! activity must run on the blocking pool. - -use super::*; - -impl Registry { - pub(crate) fn activity_support( - &self, - module: &'static str, - definition: Digest, - ) -> Result> { - let supported = self - .activities - .keys() - .filter(|key| key.module == module && key.definition == *definition.as_bytes()) - .map(|key| ActivitySupport { - activity_type: key.activity.clone(), - definition_digest: Digest::from_bytes(key.definition), - }) - .collect::>(); - if supported.is_empty() { - return Err(Error::Registry("activity support is unavailable")); - } - Ok(supported) - } - - pub(crate) fn execute_activity( - &self, - module: &'static str, - definition: Digest, - activity: &str, - context: ActivityContext, - input: Vec, - blocking: Option, - ) -> ActivityFuture { - let key = match ActivityKey::new(module, definition, activity) { - Ok(key) => key, - Err(error) => return Box::pin(async move { Err(error) }), - }; - let Some(handler) = self.activities.get(&key).copied() else { - return Box::pin(async { Err(Error::Registry("activity binding is unavailable")) }); - }; - match handler { - ActivityFunction::Async(handler) => Box::pin(async move { - let _blocking = blocking; - handler(context, input).await - }), - ActivityFunction::Blocking(handler) => { - let Some(blocking) = blocking else { - return Box::pin(async { - Err(Error::Capacity("blocking activity slot was not reserved")) - }); - }; - Box::pin(async move { - blocking - .execute(Box::new(move || handler(context, input))) - .await - }) - } - } - } - - /// Runs the statically bound maintenance command for one namespace. - pub async fn run_maintenance_once( - &self, - client: CellClient, - target: CellTarget, - identity: MutationIdentity, - request: MaintenanceTickRequest, - ) -> std::result::Result< - Committed, - InvocationError, - > { - let module = self - .namespace_modules - .get(&target.namespace()) - .map(|(module, _)| *module) - .ok_or_else(|| InvocationError::NotStarted(Error::Registry("namespace unavailable")))?; - let runner = self - .maintenance_runners - .get(module) - .copied() - .ok_or_else(|| { - InvocationError::NotStarted(Error::Registry("maintenance runner unavailable")) - })?; - runner(client, target, identity, request).await - } - - /// Reports whether one namespace has statically linked native activities. - #[must_use] - pub fn has_activity_runner(&self, namespace: NamespaceId) -> bool { - self.activity_runners.contains_key(&namespace) - } - - /// Reports whether this release contains any native blocking activity. - #[must_use] - pub fn has_blocking_activities(&self) -> bool { - !self.blocking_activity_namespaces.is_empty() - } - - /// Reports whether this namespace needs pre-claim blocking admission. - #[must_use] - pub fn requires_blocking_activity(&self, namespace: NamespaceId) -> bool { - self.blocking_activity_namespaces.contains(&namespace) - } - - /// Runs at most one statically bound native activity from one Workflow shard. - pub async fn run_activity_once( - &self, - client: CellClient, - target: &CellTarget, - lease_ms: u32, - blocking: Option, - ) -> std::result::Result { - if self.requires_blocking_activity(target.namespace()) && blocking.is_none() { - return Err(ActivitySupervisorError::Runtime(Error::Capacity( - "blocking activity slot was not reserved", - ))); - } - let runner = self - .activity_runners - .get(&target.namespace()) - .copied() - .ok_or(ActivitySupervisorError::Runtime(Error::Registry( - "activity runner unavailable", - )))?; - let shard = u32::from_be_bytes(target.partition().try_into().map_err(|_| { - ActivitySupervisorError::Runtime(Error::Identity( - "Workflow partition is not a canonical shard", - )) - })?); - runner( - client, - target.tenant(), - target.application(), - shard, - lease_ms, - blocking, - ) - .await - } - - /// Runs at most one statically bound effect from one explicit source Cell. - pub async fn run_effect_once( - &self, - client: CellClient, - target: CellTarget, - peer: EffectPeerClient, - lease_ms: u32, - ) -> std::result::Result { - let module = self - .namespace_modules - .get(&target.namespace()) - .map(|(module, _)| *module) - .ok_or(EffectSupervisorError::Runtime(Error::Registry( - "namespace unavailable", - )))?; - let runner = - self.effect_runners - .get(module) - .copied() - .ok_or(EffectSupervisorError::Runtime(Error::Registry( - "effect runner unavailable", - )))?; - runner(client, target, peer, lease_ms).await - } - - /// Reports whether one namespace's module registered effect supervision. - #[must_use] - pub fn has_effect_runner(&self, namespace: NamespaceId) -> bool { - self.namespace_modules - .get(&namespace) - .is_some_and(|(module, _)| self.effect_runners.contains_key(module)) - } -} diff --git a/crates/crab-cell-runtime/src/registry/descriptor.rs b/crates/crab-cell-runtime/src/registry/descriptor.rs deleted file mode 100644 index 24f75b91b..000000000 --- a/crates/crab-cell-runtime/src/registry/descriptor.rs +++ /dev/null @@ -1,596 +0,0 @@ -use serde::{Deserialize, Serialize}; -use std::collections::BTreeMap; - -use super::schemas::{ - BuildDescriptor, MAX_DESCRIPTOR_BYTES, MigrationDescriptor, ModuleDescriptor, - NamespaceDescriptor, OperationDescriptor, RetainedCodeDescriptor, -}; -use crate::Result; -use crate::cell::catalog::CatalogRole; -use crate::identity::Digest; -use crate::identity::encode_hex; - -pub(super) fn encode_release( - build: &BuildDescriptor, - modules: &[&ModuleDescriptor], -) -> Result<(Vec, BTreeMap)> { - let mut raw_modules = Vec::with_capacity(modules.len()); - let mut raw_namespaces = Vec::new(); - let mut codes = BTreeMap::new(); - for module in modules { - let base = RawModuleBase::from(*module); - let code = Digest::from_bytes(*blake3::hash(&serde_json::to_vec(&base)?).as_bytes()); - codes.insert(module.name.to_owned(), code); - raw_namespaces.extend(base.namespaces.iter().cloned()); - raw_modules.push(RawModule::new(base, code, module.retained_codes)); - } - raw_namespaces.sort_by(|left, right| left.id.cmp(&right.id)); - let release = RawRelease { - build: RawBuild { - cargo_lock_digest: encode_hex(build.cargo_lock_digest.as_bytes()), - source_revision: build.source_revision.clone(), - }, - modules: raw_modules, - namespaces: raw_namespaces, - peer_versions: [1], - runtime: "crab-http-server".to_owned(), - version: 1, - }; - Ok((serde_json::to_vec(&release)?, codes)) -} - -pub(super) fn verify_rolling_compatibility(previous: &[u8], candidate: &[u8]) -> Result<()> { - if let Some(reason) = rolling_incompatibility(previous, candidate)? { - return Err(crate::Error::Registry(reason)); - } - Ok(()) -} - -pub(super) fn requires_persisted_work_inventory(previous: &[u8], candidate: &[u8]) -> Result { - Ok(rolling_incompatibility(previous, candidate)?.is_some()) -} - -fn rolling_incompatibility(previous: &[u8], candidate: &[u8]) -> Result> { - let previous = decode_compatibility_release(previous)?; - let candidate = decode_compatibility_release(candidate)?; - - for old in &previous.modules { - let Some(new) = candidate - .modules - .iter() - .find(|module| module.name == old.name) - else { - return Ok(Some("rolling release removes a compiled module")); - }; - if new.schema_min > old.schema_min || new.schema_max < old.schema_max { - return Ok(Some("rolling release narrows a module schema range")); - } - if new.code != old.code - && !new.retained_codes.iter().any(|retained| { - retained.code == old.code - && retained.schema_min <= old.schema_min - && retained.schema_max >= old.schema_max - }) - { - return Ok(Some( - "rolling release does not retain predecessor module code", - )); - } - if let Some(reason) = operation_incompatibility( - &old.commands, - &new.commands, - "rolling release removes a command codec", - "rolling release narrows a command contract", - ) { - return Ok(Some(reason)); - } - if let Some(reason) = operation_incompatibility( - &old.queries, - &new.queries, - "rolling release removes a query codec", - "rolling release narrows a query contract", - ) { - return Ok(Some(reason)); - } - if !old - .migrations - .iter() - .all(|migration| new.migrations.contains(migration)) - { - return Ok(Some("rolling release removes a migration digest")); - } - if !old - .workflows - .iter() - .all(|workflow| new.workflows.contains(workflow)) - { - return Ok(Some("rolling release removes a workflow definition")); - } - if !old - .activities - .iter() - .all(|activity| new.activities.contains(activity)) - { - return Ok(Some("rolling release removes an activity type")); - } - } - if !previous - .namespaces - .iter() - .all(|namespace| candidate.namespaces.contains(namespace)) - { - return Ok(Some( - "rolling release changes or removes a namespace contract", - )); - } - Ok(None) -} - -fn operation_incompatibility( - previous: &[RawOperation], - candidate: &[RawOperation], - missing: &'static str, - narrowed: &'static str, -) -> Option<&'static str> { - for old in previous { - let Some(new) = candidate - .iter() - .find(|operation| operation.id == old.id && operation.codec == old.codec) - else { - return Some(missing); - }; - if new.schema_min > old.schema_min - || new.schema_max < old.schema_max - || new.input_limit < old.input_limit - || new.output_limit < old.output_limit - { - return Some(narrowed); - } - } - None -} - -fn decode_compatibility_release(bytes: &[u8]) -> Result { - if bytes.is_empty() || bytes.len() > MAX_DESCRIPTOR_BYTES { - return Err(crate::Error::Registry("release descriptor size is invalid")); - } - let release: RawRelease = serde_json::from_slice(bytes)?; - if release.version != 1 || release.runtime != "crab-http-server" || release.peer_versions != [1] - { - return Err(crate::Error::Registry( - "release descriptor identity is invalid", - )); - } - Ok(release) -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct RawRelease { - build: RawBuild, - modules: Vec, - namespaces: Vec, - peer_versions: [u32; 1], - runtime: String, - version: u32, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct RawBuild { - cargo_lock_digest: String, - source_revision: String, -} - -#[derive(Deserialize, Serialize)] -#[serde(deny_unknown_fields)] -struct RawModule { - activities: Vec, - code: String, - commands: Vec, - migrations: Vec, - name: String, - queries: Vec, - retained_codes: Vec, - schema_max: u32, - schema_min: u32, - source_digest: String, - workflows: Vec, -} - -impl RawModule { - fn new( - base: RawModuleBase<'_>, - code: Digest, - retained_codes: &[RetainedCodeDescriptor], - ) -> Self { - let mut retained_codes = retained_codes - .iter() - .map(RawRetainedCode::from) - .collect::>(); - retained_codes.sort(); - Self { - activities: base.activities.into_iter().map(str::to_owned).collect(), - code: encode_hex(code.as_bytes()), - commands: base.commands, - migrations: base.migrations, - name: base.name.to_owned(), - queries: base.queries, - retained_codes, - schema_max: base.schema_max, - schema_min: base.schema_min, - source_digest: base.source_digest, - workflows: base.workflows, - } - } -} - -#[derive(Deserialize, Ord, PartialOrd, Eq, PartialEq, Serialize)] -#[serde(deny_unknown_fields)] -struct RawRetainedCode { - code: String, - schema_max: u32, - schema_min: u32, -} - -impl From<&RetainedCodeDescriptor> for RawRetainedCode { - fn from(retained: &RetainedCodeDescriptor) -> Self { - Self { - code: encode_hex(retained.code.as_bytes()), - schema_max: retained.schema_max, - schema_min: retained.schema_min, - } - } -} - -#[derive(Clone, Serialize)] -struct RawModuleBase<'a> { - activities: Vec<&'a str>, - commands: Vec, - migrations: Vec, - name: &'a str, - namespaces: Vec, - queries: Vec, - schema_max: u32, - schema_min: u32, - source_digest: String, - workflows: Vec, -} - -impl<'a> From<&'a ModuleDescriptor> for RawModuleBase<'a> { - fn from(module: &'a ModuleDescriptor) -> Self { - let mut activities = module.activity_types.to_vec(); - activities.sort_unstable(); - let mut commands = module - .commands - .iter() - .map(RawOperation::from) - .collect::>(); - commands.sort(); - let mut migrations = module - .migrations - .iter() - .map(RawMigration::from) - .collect::>(); - migrations.sort(); - let mut queries = module - .queries - .iter() - .map(RawOperation::from) - .collect::>(); - queries.sort(); - let mut workflows = module - .workflow_definitions - .iter() - .map(|digest| encode_hex(digest.as_bytes())) - .collect::>(); - workflows.sort(); - let mut namespaces = module - .namespaces - .iter() - .map(|namespace| RawNamespace::new(module.name, namespace)) - .collect::>(); - namespaces.sort_by(|left, right| left.id.cmp(&right.id)); - Self { - activities, - commands, - migrations, - name: module.name, - namespaces, - queries, - schema_max: module.schema_max, - schema_min: module.schema_min, - source_digest: encode_hex(module.source_digest.as_bytes()), - workflows, - } - } -} - -#[derive(Clone, Deserialize, Ord, PartialOrd, Eq, PartialEq, Serialize)] -#[serde(deny_unknown_fields)] -struct RawOperation { - codec: u32, - id: u32, - input_limit: u32, - output_limit: u32, - schema_max: u32, - schema_min: u32, -} - -impl From<&OperationDescriptor> for RawOperation { - fn from(operation: &OperationDescriptor) -> Self { - Self { - codec: operation.codec_version, - id: operation.id, - input_limit: operation.input_limit, - output_limit: operation.output_limit, - schema_max: operation.schema_max, - schema_min: operation.schema_min, - } - } -} - -#[derive(Clone, Deserialize, Ord, PartialOrd, Eq, PartialEq, Serialize)] -#[serde(deny_unknown_fields)] -struct RawMigration { - digest: String, - version: u32, -} - -impl From<&MigrationDescriptor> for RawMigration { - fn from(migration: &MigrationDescriptor) -> Self { - Self { - digest: encode_hex(migration.digest.as_bytes()), - version: migration.version, - } - } -} - -#[derive(Clone, Deserialize, PartialEq, Eq, Serialize)] -#[serde(deny_unknown_fields)] -struct RawNamespace { - dead_letter: Option, - effect_targets: Vec, - id: String, - module: String, - name: String, - role: String, - shards: u32, -} - -impl RawNamespace { - fn new(module: &str, namespace: &NamespaceDescriptor) -> Self { - let mut effect_targets = namespace - .effect_targets - .iter() - .map(|target| encode_hex(target.as_bytes())) - .collect::>(); - effect_targets.sort(); - Self { - dead_letter: namespace - .dead_letter - .map(|target| encode_hex(target.as_bytes())), - effect_targets, - id: encode_hex(namespace.id.as_bytes()), - module: module.to_owned(), - name: namespace.name.to_owned(), - role: role_name(namespace.role).to_owned(), - shards: namespace.shards, - } - } -} - -fn role_name(role: CatalogRole) -> &'static str { - match role { - CatalogRole::Repository => "repository", - CatalogRole::Sql => "sql", - CatalogRole::Kv => "kv", - CatalogRole::Queue => "queue", - CatalogRole::Workflow => "workflow", - CatalogRole::Blob => "blob", - CatalogRole::Cron => "cron", - } -} - -#[cfg(test)] -mod tests { - use serde_json::{Value, json}; - - use super::{requires_persisted_work_inventory, verify_rolling_compatibility}; - - fn release(code: &str, retained_codes: Value) -> Value { - json!({ - "build": { - "cargo_lock_digest": "11".repeat(32), - "source_revision": "test" - }, - "modules": [{ - "activities": ["send"], - "code": code, - "commands": [{ - "codec": 1, - "id": 1, - "input_limit": 16, - "output_limit": 16, - "schema_max": 1, - "schema_min": 1 - }], - "migrations": [{"digest": "22".repeat(32), "version": 1}], - "name": "workflow", - "queries": [{ - "codec": 1, - "id": 2, - "input_limit": 8, - "output_limit": 16, - "schema_max": 1, - "schema_min": 1 - }], - "retained_codes": retained_codes, - "schema_max": 1, - "schema_min": 1, - "source_digest": "33".repeat(32), - "workflows": ["44".repeat(32)] - }], - "namespaces": [{ - "dead_letter": null, - "effect_targets": [], - "id": "55".repeat(16), - "module": "workflow", - "name": "workflow", - "role": "workflow", - "shards": 1 - }], - "peer_versions": [1], - "runtime": "crab-http-server", - "version": 1 - }) - } - - fn bytes(value: &Value) -> Vec { - serde_json::to_vec(value).unwrap() - } - - fn compatibility_error(previous: &Value, candidate: &Value) -> crate::Error { - verify_rolling_compatibility(&bytes(previous), &bytes(candidate)).unwrap_err() - } - - fn incompatible_base() -> Value { - release( - &"77".repeat(32), - json!([{"code": "66".repeat(32), "schema_max": 1, "schema_min": 1}]), - ) - } - - #[test] - fn rolling_release_rejects_a_removed_module() { - let previous = release(&"66".repeat(32), json!([])); - let mut candidate = incompatible_base(); - candidate["modules"] = json!([]); - assert!(matches!( - compatibility_error(&previous, &candidate), - crate::Error::Registry("rolling release removes a compiled module") - )); - } - - #[test] - fn rolling_release_rejects_a_narrowed_module_schema() { - let previous = release(&"66".repeat(32), json!([])); - let mut candidate = incompatible_base(); - candidate["modules"][0]["schema_min"] = json!(2); - assert!(matches!( - compatibility_error(&previous, &candidate), - crate::Error::Registry("rolling release narrows a module schema range") - )); - } - - #[test] - fn rolling_release_rejects_a_removed_query_codec() { - let previous = release(&"66".repeat(32), json!([])); - let mut candidate = incompatible_base(); - candidate["modules"][0]["queries"] = json!([]); - assert!(matches!( - compatibility_error(&previous, &candidate), - crate::Error::Registry("rolling release removes a query codec") - )); - } - - #[test] - fn rolling_release_rejects_a_narrowed_command_contract() { - let previous = release(&"66".repeat(32), json!([])); - let mut candidate = incompatible_base(); - candidate["modules"][0]["commands"][0]["input_limit"] = json!(8); - assert!(matches!( - compatibility_error(&previous, &candidate), - crate::Error::Registry("rolling release narrows a command contract") - )); - } - - #[test] - fn rolling_release_rejects_a_narrowed_query_contract() { - let previous = release(&"66".repeat(32), json!([])); - let mut candidate = incompatible_base(); - candidate["modules"][0]["queries"][0]["output_limit"] = json!(8); - assert!(matches!( - compatibility_error(&previous, &candidate), - crate::Error::Registry("rolling release narrows a query contract") - )); - } - - #[test] - fn rolling_release_rejects_a_removed_migration_digest() { - let previous = release(&"66".repeat(32), json!([])); - let mut candidate = incompatible_base(); - candidate["modules"][0]["migrations"] = json!([]); - assert!(matches!( - compatibility_error(&previous, &candidate), - crate::Error::Registry("rolling release removes a migration digest") - )); - } - - #[test] - fn rolling_release_rejects_a_removed_activity_type() { - let previous = release(&"66".repeat(32), json!([])); - let mut candidate = incompatible_base(); - candidate["modules"][0]["activities"] = json!([]); - assert!(matches!( - compatibility_error(&previous, &candidate), - crate::Error::Registry("rolling release removes an activity type") - )); - } - - #[test] - fn rolling_release_accepts_widened_contracts() { - let previous = release(&"66".repeat(32), json!([])); - let mut candidate = incompatible_base(); - candidate["modules"][0]["schema_min"] = json!(0); - candidate["modules"][0]["schema_max"] = json!(2); - candidate["modules"][0]["commands"][0]["input_limit"] = json!(32); - candidate["modules"][0]["queries"][0]["output_limit"] = json!(32); - verify_rolling_compatibility(&bytes(&previous), &bytes(&candidate)).unwrap(); - assert!(!requires_persisted_work_inventory(&bytes(&previous), &bytes(&candidate)).unwrap()); - } - - #[test] - fn rolling_release_retains_executable_and_persisted_work_contracts() { - let old_code = "66".repeat(32); - let previous = release(&old_code, json!([])); - let candidate = release( - &"77".repeat(32), - json!([{"code": old_code, "schema_max": 1, "schema_min": 1}]), - ); - - verify_rolling_compatibility(&bytes(&previous), &bytes(&candidate)).unwrap(); - assert!(!requires_persisted_work_inventory(&bytes(&previous), &bytes(&candidate)).unwrap()); - } - - #[test] - fn rolling_release_rejects_removed_runtime_contracts() { - let old_code = "66".repeat(32); - let previous = release(&old_code, json!([])); - let base = release( - &"77".repeat(32), - json!([{"code": old_code, "schema_max": 1, "schema_min": 1}]), - ); - - let mut cases = Vec::new(); - let mut missing_code = base.clone(); - missing_code["modules"][0]["retained_codes"] = json!([]); - cases.push(missing_code); - let mut missing_command = base.clone(); - missing_command["modules"][0]["commands"] = json!([]); - cases.push(missing_command); - let mut missing_workflow = base.clone(); - missing_workflow["modules"][0]["workflows"] = json!([]); - cases.push(missing_workflow); - let mut changed_namespace = base; - changed_namespace["namespaces"][0]["shards"] = json!(2); - cases.push(changed_namespace); - - for candidate in cases { - assert!(verify_rolling_compatibility(&bytes(&previous), &bytes(&candidate)).is_err()); - assert!( - requires_persisted_work_inventory(&bytes(&previous), &bytes(&candidate)).unwrap() - ); - } - } -} diff --git a/crates/crab-cell-runtime/src/registry/handlers.rs b/crates/crab-cell-runtime/src/registry/handlers.rs deleted file mode 100644 index f045f703d..000000000 --- a/crates/crab-cell-runtime/src/registry/handlers.rs +++ /dev/null @@ -1,365 +0,0 @@ -//! Command, query, and activity handler contracts. - -use std::future::Future; -use std::pin::Pin; - -use rusqlite::{Connection, Transaction}; - -use super::*; - -/// Synchronous application command context with no raw transaction accessor. -pub struct CommandContext<'borrow, 'connection> { - pub(super) transaction: &'borrow Transaction<'connection>, - pub(super) target: CellTarget, - pub(super) effect_targets: &'static [NamespaceId], - pub(super) sequence: u64, - pub(super) now_ms: i64, - pub(super) issued_at_ms: i64, - pub(super) input_limit: u32, - pub(super) output_limit: u32, - pub(super) effects: Option, -} - -impl CommandContext<'_, '_> { - /// Returns the Cell this command targets. - #[must_use] - pub fn cell_id(&self) -> CellId { - self.target.cell_id() - } - - /// Returns the runtime-validated target for deterministic cross-Cell routing. - #[must_use] - pub const fn target(&self) -> &CellTarget { - &self.target - } - - /// Returns the actor-ordered sequence the command was admitted at. - #[must_use] - pub const fn sequence(&self) -> u64 { - self.sequence - } - - /// Returns the logical runtime time for the command. - #[must_use] - pub const fn now_ms(&self) -> i64 { - self.now_ms - } - - /// Returns the timestamp at which the caller created this mutation. - /// - /// This remains crate-private because application commands should use the - /// logical runtime timestamp for domain decisions. Native primitives use - /// it only when validating an absolute expiry that is part of a request. - pub(crate) const fn issued_at_ms(&self) -> i64 { - self.issued_at_ms - } - - /// Emits one durable cross-Cell command with this command's allocator. - pub fn emit_effect(&mut self, intent: &EffectCommandIntent) -> Result<[u8; 32]> { - if intent.target.tenant() != self.target.tenant() - || intent.target.application() != self.target.application() - { - return Err(Error::Identity( - "effect target is outside the source application scope", - )); - } - if !self.effect_targets.contains(&intent.target.namespace()) { - return Err(Error::Command("effect target is not declared")); - } - self.ensure_effects()?; - let transaction = self.transaction; - self.effects - .as_mut() - .ok_or(Error::Command("effect ledger was not initialized"))? - .insert_command(transaction, intent) - } - - pub(crate) fn primitive_effects(&mut self) -> Result<(&Transaction<'_>, &mut EffectBatch)> { - self.ensure_effects()?; - let transaction = self.transaction; - let effects = self - .effects - .as_mut() - .ok_or(Error::Command("effect ledger was not initialized"))?; - Ok((transaction, effects)) - } - - pub(super) fn ensure_effects(&mut self) -> Result<&mut EffectBatch> { - if self.effects.is_none() { - self.effects = Some(EffectBatch::new( - self.transaction, - &self.target, - self.sequence, - self.now_ms, - )?); - } - self.effects - .as_mut() - .ok_or(Error::Command("effect ledger was not initialized")) - } - - /// Executes bounded application SQL under the runtime authorizer. - pub fn sql(&self, batch: &SqlBatch) -> Result> { - sql_batch(self.transaction, batch) - } - - /// Reserves database bytes under a stable key for later commands in this Cell. - /// - /// Install the capacity primitive schema first. Reusing a key requires the - /// same rounded page count. Reservations persist until explicitly released; - /// every runtime commit protects them, including its own receipt writes. - pub fn reserve_database_capacity(&self, key: &[u8], bytes: u64) -> Result<()> { - crate::primitives::capacity::reserve(self.transaction, key, bytes) - } - - /// Releases a reservation in this command, returning whether it existed. - /// - /// Release before applying deferred work. Failure of this command restores - /// the reservation with the rest of the transaction. - pub fn release_database_capacity(&self, key: &[u8]) -> Result { - crate::primitives::capacity::release(self.transaction, key) - } - - /// Returns the SQLite page size for application allocation accounting. - pub fn database_page_size(&self) -> Result { - Ok(self - .transaction - .query_row("PRAGMA page_size", [], |row| row.get(0))?) - } - - /// Writes a bounded slice into an existing application BLOB in this command. - /// - /// Allocate its fixed size with SQL `zeroblob` first. Only main-database - /// rowid tables and unindexed, non-key columns are supported. SQLite does - /// not run triggers or CHECK constraints for incremental writes: callers - /// must maintain application invariants in the same command. Protected - /// tables, writes beyond the BLOB, and operations over 1 MiB are rejected. - /// Propagate failures to roll back the command's application savepoint. - pub fn write_sql_blob( - &self, - table: &str, - column: &str, - row_id: i64, - offset: usize, - bytes: &[u8], - ) -> Result<()> { - crate::primitives::sql::write_blob(self.transaction, table, column, row_id, offset, bytes) - } - - pub(crate) const fn primitive_transaction(&self) -> &Transaction<'_> { - self.transaction - } -} - -/// Read-only application query context with no raw connection accessor. -pub struct QueryContext<'borrow> { - pub(super) connection: &'borrow Connection, - pub(super) cell: CellId, - pub(super) commit_sequence: u64, - pub(super) now_ms: i64, - pub(super) input_limit: u32, - pub(super) output_limit: u32, -} - -impl QueryContext<'_> { - /// Returns the Cell this query reads. - #[must_use] - pub const fn cell_id(&self) -> CellId { - self.cell - } - - /// Returns the highest committed sequence the query may observe. - #[must_use] - pub const fn commit_sequence(&self) -> u64 { - self.commit_sequence - } - - /// Returns the logical runtime time for the query. - #[must_use] - pub const fn now_ms(&self) -> i64 { - self.now_ms - } - - /// Executes bounded read-only application SQL under the runtime authorizer. - pub fn sql(&self, batch: &SqlBatch) -> Result> { - sql_query_batch(self.connection, batch) - } - - /// Returns occupied SQLite page bytes, including runtime and indexes but excluding the freelist. - pub fn database_used_bytes(&self) -> Result { - let pages: i64 = self - .connection - .query_row("PRAGMA page_count", [], |row| row.get(0))?; - let free_pages: i64 = self - .connection - .query_row("PRAGMA freelist_count", [], |row| row.get(0))?; - let page_size: i64 = self - .connection - .query_row("PRAGMA page_size", [], |row| row.get(0))?; - let pages = pages - .checked_sub(free_pages) - .ok_or(Error::Command("invalid database freelist count"))?; - let pages = - u64::try_from(pages).map_err(|_| Error::Command("invalid database page count"))?; - let page_size = - u64::try_from(page_size).map_err(|_| Error::Command("invalid database page size"))?; - pages - .checked_mul(page_size) - .ok_or(Error::Command("database size overflow")) - } - - pub(crate) const fn primitive_connection(&self) -> &Connection { - self.connection - } -} - -pub(super) type CommandHandler = for<'borrow, 'connection> fn( - &mut CommandContext<'borrow, 'connection>, - &[u8], -) -> Result; - -pub(super) type QueryHandler = - for<'borrow> fn(&mut QueryContext<'borrow>, &[u8]) -> Result>; -pub(super) type ActivityFuture = - Pin> + Send + 'static>>; -pub(super) type AsyncActivityFunction = fn(ActivityContext, Vec) -> ActivityFuture; -pub(super) type BlockingActivityFunction = - fn(ActivityContext, Vec) -> Result; - -#[derive(Clone, Copy)] -pub(super) enum ActivityFunction { - Async(AsyncActivityFunction), - Blocking(BlockingActivityFunction), -} -pub(super) type MaintenanceFuture = Pin< - Box< - dyn Future< - Output = std::result::Result< - Committed, - InvocationError, - >, - > + Send - + 'static, - >, ->; -pub(super) type MaintenanceRunner = - fn(CellClient, CellTarget, MutationIdentity, MaintenanceTickRequest) -> MaintenanceFuture; -pub(super) type EffectFuture = Pin< - Box< - dyn Future> - + Send - + 'static, - >, ->; -pub(super) type EffectRunner = fn(CellClient, CellTarget, EffectPeerClient, u32) -> EffectFuture; -pub(super) type ActivityRunFuture = Pin< - Box< - dyn Future> - + Send - + 'static, - >, ->; -pub(super) type ActivityRunner = fn( - CellClient, - TenantId, - ApplicationId, - u32, - u32, - Option, -) -> ActivityRunFuture; - -/// Stored command decision encoded with the command's declared output codec. -pub enum CommandResult { - /// The command committed a result the caller consumes as success. - Success(T), - /// The command committed a rejection the caller consumes as the result. - Rejected(T), -} - -/// Statically dispatched typed command implemented by compiled Crab code. -pub trait Command: Send + Sync + 'static { - /// Module the command is registered under. - const MODULE: &'static str; - /// Command id within its module. - const ID: u32; - /// Input codec version the command accepts. - const CODEC_VERSION: u32; - /// Declared input type. - type Input: WireValue; - /// Declared output type. - type Output: WireValue; - - /// Runs the command inside its savepoint and returns its decision. - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> Result>; -} - -/// Statically dispatched typed query implemented by compiled Crab code. -pub trait Query: Send + Sync + 'static { - /// Module the query is registered under. - const MODULE: &'static str; - /// Query id within its module. - const ID: u32; - /// Input codec version the query accepts. - const CODEC_VERSION: u32; - /// Declared input type. - type Input: WireValue; - /// Declared output type. - type Output: WireValue; - - /// Runs the read-only query. - fn execute(context: &mut QueryContext<'_>, input: Self::Input) -> Result; -} - -/// Validated command selection and bounded input supplied by runtime routing. -pub struct CommandInvocation<'a> { - /// Module the routing selected. - pub module: &'a str, - /// Command id the routing selected. - pub operation_id: u32, - /// Codec version the caller declared. - pub codec_version: u32, - /// Schema version the Cell serves. - pub schema: u32, - /// Validated target the command must own. - pub target: CellTarget, - /// Actor-ordered sequence of the command. - pub sequence: u64, - /// Logical runtime time for the command. - pub now_ms: i64, - /// Bounded encoded input. - pub input: &'a [u8], -} - -/// Validated query selection and bounded input supplied by runtime routing. -pub struct QueryInvocation<'a> { - /// Module the routing selected. - pub module: &'a str, - /// Query id the routing selected. - pub operation_id: u32, - /// Codec version the caller declared. - pub codec_version: u32, - /// Schema version the Cell serves. - pub schema: u32, - /// Cell the query reads. - pub cell: CellId, - /// Highest committed sequence the query may observe. - pub commit_sequence: u64, - /// Logical runtime time for the query. - pub now_ms: i64, - /// Bounded encoded input. - pub input: &'a [u8], -} - -/// Source-level module registration contract for statically linked Crab code. -pub trait CellModule: Send + Sync + 'static { - /// Source-level module name, matched against its descriptor. - const NAME: &'static str; - - /// Returns the static descriptor this module registers. - fn descriptor(&self) -> &'static ModuleDescriptor; - /// Registers this module's typed bindings. - fn register(self, registry: &mut RegistryBuilder) -> std::result::Result<(), RegistryError>; -} diff --git a/crates/crab-cell-runtime/src/registry/schemas.rs b/crates/crab-cell-runtime/src/registry/schemas.rs deleted file mode 100644 index 24ad1dec9..000000000 --- a/crates/crab-cell-runtime/src/registry/schemas.rs +++ /dev/null @@ -1,216 +0,0 @@ -//! Module, migration, and operation descriptors with their validation rules. - -use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet}; - -use super::*; -use crate::registry::builder::{BindingKey, CronBinding, PrimitiveBinding, QueueBinding}; -use crate::registry::handlers::{ - ActivityFuture, ActivityRunFuture, EffectFuture, MaintenanceFuture, -}; - -pub(super) const MAX_DESCRIPTOR_BYTES: usize = 256 * 1024; -pub(super) const MAX_MODULES: usize = 128; -pub(super) const MAX_NAMESPACES: usize = 128; -pub(super) const MAX_MIGRATION_BYTES: usize = 1024 * 1024; -pub(super) const MAX_OPERATION_BYTES: u32 = crate::codec::MAX_WIRE_BYTES as u32; -pub(super) const CODE_ONLY_MIGRATION_BYTES: usize = 2 * 32; - -/// Registry construction error returned before server readiness. -pub type RegistryError = Error; - -/// Build evidence included in the canonical release descriptor. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct BuildDescriptor { - /// Source revision the image was built from. - pub source_revision: String, - /// Digest of the Cargo lockfile the image was built with. - pub cargo_lock_digest: Digest, -} - -/// One exact application migration and its independently verified digest. -#[derive(Clone, Copy, Debug)] -pub struct MigrationDescriptor { - /// Schema version this migration installs. - pub version: u32, - /// Migration SQL, digest-bound to `digest`. - pub sql: &'static str, - /// BLAKE3 digest of `sql`. - pub digest: Digest, -} - -/// One predecessor module code intentionally retained by the current binary. -#[derive(Clone, Copy, Debug)] -pub struct RetainedCodeDescriptor { - /// Code digest the module retains. - pub code: Digest, - /// Oldest schema version the retained code serves. - pub schema_min: u32, - /// Newest schema version the retained code serves. - pub schema_max: u32, -} - -/// One registry-verified schema or code migration for an active Cell. -#[derive(Clone, Copy, Debug)] -pub struct MigrationPlan { - pub(super) module: &'static str, - pub(super) from_code: Digest, - pub(super) to_code: Digest, - pub(super) from_schema: u32, - pub(super) to_schema: u32, - pub(super) migration: Option, -} - -impl MigrationPlan { - /// Returns the module the plan migrates. - #[must_use] - pub const fn module(&self) -> &'static str { - self.module - } - - /// Returns the code digest the Cell migrates from. - #[must_use] - pub const fn from_code(&self) -> Digest { - self.from_code - } - - /// Returns the code digest the Cell migrates to. - #[must_use] - pub const fn to_code(&self) -> Digest { - self.to_code - } - - /// Returns the schema version the Cell migrates from. - #[must_use] - pub const fn from_schema(&self) -> u32 { - self.from_schema - } - - /// Returns the schema version the Cell migrates to. - #[must_use] - pub const fn to_schema(&self) -> u32 { - self.to_schema - } - - /// Returns the plan digest, when the plan declares one. - #[must_use] - pub const fn digest(&self) -> Option { - match self.migration { - Some(migration) => Some(migration.digest), - None => None, - } - } - - /// Returns the migration SQL, when the plan carries any. - #[must_use] - pub const fn sql(&self) -> Option<&'static str> { - match self.migration { - Some(migration) => Some(migration.sql), - None => None, - } - } - - /// Returns the plan's operation byte budget. - #[must_use] - pub const fn operation_bytes(&self) -> usize { - match self.migration { - Some(migration) => migration.sql.len(), - None => CODE_ONLY_MIGRATION_BYTES, - } - } -} - -/// One registered command or query codec and its schema compatibility range. -#[derive(Clone, Copy, Debug)] -pub struct OperationDescriptor { - /// Command or query id within its module. - pub id: u32, - /// Input codec version the operation accepts. - pub codec_version: u32, - /// Oldest schema version the operation serves. - pub schema_min: u32, - /// Newest schema version the operation serves. - pub schema_max: u32, - /// Largest input the operation accepts, in bytes. - pub input_limit: u32, - /// Largest output the operation returns, in bytes. - pub output_limit: u32, -} - -/// One stable Cell namespace compiled into a module. -#[derive(Clone, Copy, Debug)] -pub struct NamespaceDescriptor { - /// Namespace identity. - pub id: NamespaceId, - /// Stable namespace name. - pub name: &'static str, - /// Catalog role the namespace serves. - pub role: CatalogRole, - /// Power-of-two shard count. - pub shards: u32, - /// Namespaces this one may emit effects to. - pub effect_targets: &'static [NamespaceId], - /// Queue namespace that receives this namespace's dead letters. - pub dead_letter: Option, -} - -/// Static module inventory that must match its compiled function bindings. -#[derive(Clone, Copy, Debug)] -pub struct ModuleDescriptor { - /// Module name, matched against the compiled function bindings. - pub name: &'static str, - /// Digest of the module's source. - pub source_digest: Digest, - /// Code digests the module retains for older schemas. - pub retained_codes: &'static [RetainedCodeDescriptor], - /// Oldest schema version the module serves. - pub schema_min: u32, - /// Newest schema version the module serves. - pub schema_max: u32, - /// Contiguous migrations covering `schema_min` through `schema_max`. - pub migrations: &'static [MigrationDescriptor], - /// Commands the module declares. - pub commands: &'static [OperationDescriptor], - /// Queries the module declares. - pub queries: &'static [OperationDescriptor], - /// Workflow definition digests the module serves. - pub workflow_definitions: &'static [Digest], - /// Activity types the module registers. - pub activity_types: &'static [&'static str], - /// Namespaces the module owns. - pub namespaces: &'static [NamespaceDescriptor], -} - -mod binding; -mod validation; - -pub(in crate::registry) use binding::*; -pub(in crate::registry) use validation::*; - -pub(super) fn operation_descriptors( - modules: &[&ModuleDescriptor], - select: impl Fn(&ModuleDescriptor) -> &[OperationDescriptor], -) -> BTreeMap { - modules - .iter() - .flat_map(|module| { - select(module).iter().map(|operation| { - ( - BindingKey { - module: module.name.to_owned(), - id: operation.id, - codec_version: operation.codec_version, - }, - *operation, - ) - }) - }) - .collect() -} - -pub(super) fn valid_name(value: &str) -> bool { - !value.is_empty() - && value.len() <= 128 - && value - .bytes() - .all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'_' | b'.')) -} diff --git a/crates/crab-cell-runtime/src/registry/schemas/binding.rs b/crates/crab-cell-runtime/src/registry/schemas/binding.rs deleted file mode 100644 index 5981594d2..000000000 --- a/crates/crab-cell-runtime/src/registry/schemas/binding.rs +++ /dev/null @@ -1,107 +0,0 @@ -//! Typed command, query, activity, and maintenance bindings. - -use super::*; - -pub(in crate::registry) fn typed_command( - context: &mut CommandContext<'_, '_>, - input: &[u8], -) -> Result { - let input = decode_wire::(input, context.input_limit)?; - let outcome = C::execute(context, input)?; - Ok(match outcome { - CommandResult::Success(output) => { - HandlerOutcome::Success(encode_wire(&output, context.output_limit)?) - } - CommandResult::Rejected(output) => { - HandlerOutcome::Rejected(encode_wire(&output, context.output_limit)?) - } - }) -} - -pub(in crate::registry) fn typed_query( - context: &mut QueryContext<'_>, - input: &[u8], -) -> Result> { - let input = decode_wire::(input, context.input_limit)?; - let output = Q::execute(context, input)?; - Ok(encode_wire(&output, context.output_limit)?) -} - -pub(in crate::registry) fn typed_activity( - context: ActivityContext, - input: Vec, -) -> ActivityFuture { - Box::pin(async move { - let outcome = A::execute(context, input).await; - let payload = match &outcome { - ActivityExecution::Completed(result) => result, - ActivityExecution::Failed { details, .. } => details, - }; - if payload.len() > crate::primitives::workflow::MAX_ACTIVITY_PAYLOAD_BYTES { - return Err(Error::Command("activity handler result exceeds 256 KiB")); - } - Ok(outcome) - }) -} - -pub(in crate::registry) fn typed_blocking_activity( - context: ActivityContext, - input: Vec, -) -> Result { - let outcome = A::execute(context, input); - let payload = match &outcome { - ActivityExecution::Completed(result) => result, - ActivityExecution::Failed { details, .. } => details, - }; - if payload.len() > crate::primitives::workflow::MAX_ACTIVITY_PAYLOAD_BYTES { - return Err(Error::Command("activity handler result exceeds 256 KiB")); - } - Ok(outcome) -} - -pub(in crate::registry) fn typed_maintenance( - client: CellClient, - target: CellTarget, - identity: MutationIdentity, - request: MaintenanceTickRequest, -) -> MaintenanceFuture { - Box::pin(async move { - client - .command::>(&target, identity, request) - .await - }) -} - -pub(in crate::registry) fn typed_effect( - client: CellClient, - target: CellTarget, - peer: EffectPeerClient, - lease_ms: u32, -) -> EffectFuture { - Box::pin(async move { - EffectSupervisor::::new(crate::EffectSource::new(client, target), peer, lease_ms) - .map_err(EffectSupervisorError::Runtime)? - .run_once() - .await - }) -} - -pub(in crate::registry) fn typed_activity_runner( - client: CellClient, - tenant: TenantId, - application: ApplicationId, - shard: u32, - lease_ms: u32, - blocking: Option, -) -> ActivityRunFuture { - Box::pin(async move { - ActivitySupervisor::new( - WorkflowActivities::::new(client, tenant, application) - .map_err(ActivitySupervisorError::Runtime)?, - lease_ms, - ) - .map_err(ActivitySupervisorError::Runtime)? - .run_once(shard, blocking) - .await - }) -} diff --git a/crates/crab-cell-runtime/src/registry/schemas/validation.rs b/crates/crab-cell-runtime/src/registry/schemas/validation.rs deleted file mode 100644 index 80fd53288..000000000 --- a/crates/crab-cell-runtime/src/registry/schemas/validation.rs +++ /dev/null @@ -1,395 +0,0 @@ -//! Build, module, namespace, and primitive descriptor validation. - -use super::*; - -pub(in crate::registry) fn validate_build(build: &BuildDescriptor) -> Result<()> { - if build.source_revision.is_empty() - || build.source_revision.len() > 128 - || build - .cargo_lock_digest - .as_bytes() - .iter() - .all(|byte| *byte == 0) - { - return Err(Error::Registry("invalid build descriptor")); - } - Ok(()) -} - -pub(in crate::registry) fn validate_module( - module: &ModuleDescriptor, - commands: &mut BTreeSet, - queries: &mut BTreeSet, - namespaces: &mut HashMap, -) -> Result<()> { - if !valid_name(module.name) - || module.schema_min == 0 - || module.schema_max < module.schema_min - || module - .source_digest - .as_bytes() - .iter() - .all(|byte| *byte == 0) - { - return Err(Error::Registry("invalid module descriptor")); - } - if module.migrations.is_empty() { - return Err(Error::Registry("module has no migrations")); - } - let mut expected_version = module.schema_min; - for migration in module.migrations { - if migration.version != expected_version - || migration.sql.is_empty() - || migration.sql.len() > MAX_MIGRATION_BYTES - || migration.digest - != Digest::from_bytes(*blake3::hash(migration.sql.as_bytes()).as_bytes()) - { - return Err(Error::Registry("invalid migration inventory")); - } - expected_version = expected_version - .checked_add(1) - .ok_or(Error::Registry("migration version overflow"))?; - } - if expected_version.checked_sub(1) != Some(module.schema_max) { - return Err(Error::Registry( - "migration range does not cover module schema", - )); - } - let mut retained_codes = HashSet::new(); - for retained in module.retained_codes { - if retained.code.as_bytes().iter().all(|byte| *byte == 0) - || retained.schema_min < module.schema_min - || retained.schema_max > module.schema_max - || retained.schema_max < retained.schema_min - || !retained_codes.insert(retained.code) - { - return Err(Error::Registry("invalid retained module code inventory")); - } - } - validate_operations(module, module.commands, commands)?; - validate_operations(module, module.queries, queries)?; - let mut workflows = HashSet::new(); - for digest in module.workflow_definitions { - if digest.as_bytes().iter().all(|byte| *byte == 0) || !workflows.insert(*digest) { - return Err(Error::Registry("invalid workflow definition inventory")); - } - } - let mut activities = HashSet::new(); - for activity in module.activity_types { - if !valid_name(activity) || !activities.insert(*activity) { - return Err(Error::Registry("invalid activity inventory")); - } - } - if !activities.is_empty() && workflows.is_empty() { - return Err(Error::Registry( - "activity inventory requires a workflow definition", - )); - } - for namespace in module.namespaces { - if namespaces - .insert(namespace.id, (module.name, *namespace)) - .is_some() - { - return Err(Error::Registry("duplicate namespace ID")); - } - } - Ok(()) -} - -pub(in crate::registry) fn validate_operations( - module: &ModuleDescriptor, - operations: &[OperationDescriptor], - keys: &mut BTreeSet, -) -> Result<()> { - for operation in operations { - if operation.id == 0 - || operation.codec_version == 0 - || operation.schema_min < module.schema_min - || operation.schema_max > module.schema_max - || operation.schema_max < operation.schema_min - || !(1..=MAX_OPERATION_BYTES).contains(&operation.input_limit) - || !(1..=MAX_OPERATION_BYTES).contains(&operation.output_limit) - || !keys.insert(BindingKey::new( - module.name, - operation.id, - operation.codec_version, - )?) - { - return Err(Error::Registry("invalid operation inventory")); - } - } - Ok(()) -} - -pub(in crate::registry) fn validate_namespaces( - namespaces: &HashMap, -) -> Result<()> { - if namespaces.len() > MAX_NAMESPACES { - return Err(Error::Registry("namespace count exceeds 128")); - } - let mut names = HashSet::new(); - for (_, namespace) in namespaces.values() { - if !valid_name(namespace.name) - || !names.insert(namespace.name) - || !(1..=4096).contains(&namespace.shards) - || !namespace.shards.is_power_of_two() - { - return Err(Error::Registry("invalid namespace inventory")); - } - let mut targets = HashSet::new(); - for target in namespace.effect_targets { - if !namespaces.contains_key(target) || !targets.insert(*target) { - return Err(Error::Registry("invalid effect target inventory")); - } - } - if let Some(dead_letter) = namespace.dead_letter - && (!targets.contains(&dead_letter) - || namespaces.get(&dead_letter).map(|(_, value)| value.role) - != Some(CatalogRole::Queue)) - { - return Err(Error::Registry("invalid dead-letter target")); - } - } - for start in namespaces.keys() { - let mut visited = HashSet::new(); - let mut current = Some(*start); - while let Some(id) = current { - if !visited.insert(id) { - return Err(Error::Registry("dead-letter cycle")); - } - current = namespaces.get(&id).and_then(|(_, value)| value.dead_letter); - } - } - Ok(()) -} - -pub(in crate::registry) fn validate_workflow_effect_targets( - bindings: &HashMap<(String, [u8; 32]), Vec>, - namespaces: &HashMap, - modules: &[&ModuleDescriptor], -) -> Result<()> { - for module in modules - .iter() - .copied() - .filter(|module| !module.workflow_definitions.is_empty()) - { - let mut workflow_namespaces = module - .namespaces - .iter() - .filter(|namespace| namespace.role == CatalogRole::Workflow); - let namespace = workflow_namespaces.next().ok_or(Error::Registry( - "workflow definitions require one Workflow namespace", - ))?; - if workflow_namespaces.next().is_some() { - return Err(Error::Registry( - "workflow definitions require one Workflow namespace", - )); - } - let compiled = bindings - .iter() - .filter(|((owner, _), _)| owner == module.name) - .flat_map(|(_, targets)| targets.iter().copied()) - .collect::>(); - let declared = namespace - .effect_targets - .iter() - .copied() - .collect::>(); - if compiled != declared - || compiled - .iter() - .any(|target| !namespaces.contains_key(target)) - { - return Err(Error::Registry( - "Workflow effect targets and compiled definitions differ", - )); - } - } - Ok(()) -} - -pub(in crate::registry) fn validate_queue_bindings( - bindings: &[QueueBinding], - namespaces: &HashMap, - modules: &[&ModuleDescriptor], -) -> Result<()> { - let queues = namespaces - .iter() - .filter(|(_, (_, namespace))| namespace.role == CatalogRole::Queue) - .count(); - if bindings.len() != queues { - return Err(Error::Registry( - "Queue descriptors and compiled bindings differ", - )); - } - - for binding in bindings { - let Some((owner, namespace)) = namespaces.get(&binding.namespace) else { - return Err(Error::Registry("Queue binding namespace is unavailable")); - }; - if namespace.role != CatalogRole::Queue || *owner != binding.module { - return Err(Error::Registry("Queue binding does not own its namespace")); - } - let module = modules - .iter() - .find(|module| module.name == binding.module) - .ok_or(Error::Registry("Queue binding module is unavailable"))?; - if !module.commands.iter().any(|command| { - command.id == binding.send_command_id - && command.codec_version == binding.codec_version - && command.input_limit >= crate::primitives::queue::QUEUE_SEND_MAX_INPUT_BYTES - }) { - return Err(Error::Registry( - "Queue send binding is absent from its descriptor", - )); - } - if namespace.dead_letter != binding.dead_letter.map(|target| target.namespace()) { - return Err(Error::Registry( - "Queue dead-letter descriptor and binding differ", - )); - } - - let Some(dead_letter) = binding.dead_letter else { - continue; - }; - let target = bindings - .iter() - .find(|candidate| candidate.namespace == dead_letter.namespace()) - .ok_or(Error::Registry( - "Queue dead-letter target has no compiled binding", - ))?; - if target.module != dead_letter.module() - || target.send_command_id != dead_letter.send_command_id() - || target.codec_version != dead_letter.codec_version() - || namespaces - .get(&target.namespace) - .map(|(_, descriptor)| descriptor.shards) - != Some(dead_letter.shards()) - { - return Err(Error::Registry( - "Queue dead-letter target and compiled binding differ", - )); - } - } - Ok(()) -} - -pub(in crate::registry) fn validate_cron_bindings( - bindings: &[CronBinding], - namespaces: &HashMap, - modules: &[&ModuleDescriptor], -) -> Result<()> { - let cron_namespaces = namespaces - .values() - .filter(|(_, namespace)| namespace.role == CatalogRole::Cron) - .count(); - if bindings.len() != cron_namespaces { - return Err(Error::Registry( - "Cron descriptors and compiled bindings differ", - )); - } - for binding in bindings { - let Some((owner, namespace)) = namespaces.get(&binding.namespace) else { - return Err(Error::Registry("Cron binding namespace is unavailable")); - }; - if namespace.role != CatalogRole::Cron || *owner != binding.module { - return Err(Error::Registry("Cron binding does not own its namespace")); - } - let compiled_targets = binding - .targets - .iter() - .map(|target| target.namespace()) - .collect::>(); - let declared_targets = namespace - .effect_targets - .iter() - .copied() - .collect::>(); - if compiled_targets != declared_targets { - return Err(Error::Registry("Cron effect targets and descriptor differ")); - } - for target in binding.targets { - let Some((target_owner, _)) = namespaces.get(&target.namespace()) else { - return Err(Error::Registry("Cron target namespace is unavailable")); - }; - if *target_owner != target.module() { - return Err(Error::Registry( - "Cron target module differs from descriptor", - )); - } - let target_module = modules - .iter() - .find(|module| module.name == target.module()) - .ok_or(Error::Registry("Cron target module is unavailable"))?; - if !target_module.commands.iter().any(|command| { - command.id == target.command_id() - && command.codec_version == target.codec_version() - && command.input_limit == target.input_limit() - }) { - return Err(Error::Registry( - "Cron target command differs from descriptor", - )); - } - } - } - Ok(()) -} - -pub(in crate::registry) fn validate_primitive_bindings( - bindings: &[PrimitiveBinding], - role: CatalogRole, - mismatch: &'static str, - namespaces: &HashMap, -) -> Result<()> { - let expected = namespaces - .values() - .filter(|(_, namespace)| namespace.role == role) - .count(); - if bindings.len() != expected { - return Err(Error::Registry(mismatch)); - } - for binding in bindings { - let Some((owner, namespace)) = namespaces.get(&binding.namespace) else { - return Err(Error::Registry(mismatch)); - }; - if *owner != binding.module || namespace.role != role { - return Err(Error::Registry(mismatch)); - } - } - Ok(()) -} - -pub(in crate::registry) fn validate_maintenance_bindings( - maintenance: &BTreeMap<&'static str, Option>, - queues: &[QueueBinding], -) -> Result<()> { - for (module, configured) in maintenance { - let queue = queues.iter().find(|queue| queue.module == *module); - match (queue, configured) { - (Some(queue), configured) if queue.dead_letter == *configured => {} - (None, None) => {} - _ => { - return Err(Error::Registry( - "maintenance and Queue dead-letter bindings differ", - )); - } - } - } - Ok(()) -} - -pub(in crate::registry) fn validate_invocation( - operation: &OperationDescriptor, - schema: u32, - input_bytes: usize, -) -> Result<()> { - if !(operation.schema_min..=operation.schema_max).contains(&schema) { - return Err(Error::Command( - "registered operation does not support schema", - )); - } - if input_bytes > operation.input_limit as usize { - return Err(Error::Command("registered operation input exceeds limit")); - } - Ok(()) -} diff --git a/crates/crab-cell-runtime/src/retry.rs b/crates/crab-cell-runtime/src/retry.rs deleted file mode 100644 index f658eb1e9..000000000 --- a/crates/crab-cell-runtime/src/retry.rs +++ /dev/null @@ -1,74 +0,0 @@ -//! Shared storage-retry classification and backoff for runtime senders. -//! -//! Publication, release progress, and the Cell catalog retry the same storage -//! classes. Keeping the classifier and the delay schedule in one module stops -//! one sender from retrying a class another sender treats as fatal. - -use std::time::Duration; - -use crab_storage::{RetryClass, StorageError}; - -use crate::{Error, Result}; - -/// Longest delay any sender waits between storage retries. -pub(crate) const MAX_RETRY_DELAY_MS: u64 = 1_000; - -/// Whether a sender that holds its own attempt budget may retry this failure. -pub(crate) fn retryable_storage_error(error: &StorageError) -> bool { - matches!( - crab_storage::retry_class(error), - RetryClass::Transient - | RetryClass::Throttled { .. } - | RetryClass::StateDependent - | RetryClass::InspectErrno - ) -} - -/// Delay the provider asked for, when the failure carried one. -pub(crate) fn retry_hint(error: &StorageError) -> Option { - match crab_storage::retry_class(error) { - RetryClass::Throttled { retry_after } => retry_after, - _ => None, - } -} - -/// Bounded exponential backoff for one sender's retry loop. -pub(crate) struct Backoff { - delay_ms: u64, -} - -impl Default for Backoff { - fn default() -> Self { - Self { delay_ms: 100 } - } -} - -impl Backoff { - /// Sleeps at least the current delay — and at least `minimum` when the - /// provider asked for longer — then doubles the delay. - pub(crate) async fn wait(&mut self, minimum: Option) { - let delay = Duration::from_millis(self.delay_ms); - tokio::time::sleep(minimum.map_or(delay, |minimum| minimum.max(delay))).await; - self.delay_ms = self.delay_ms.saturating_mul(2).min(MAX_RETRY_DELAY_MS); - } - - /// [`Backoff::wait`] bounded by `deadline`: a wait that would run past it - /// reports the fence instead, so a sender cannot outlive its ownership. - pub(crate) async fn wait_until( - &mut self, - minimum: Option, - deadline: std::time::Instant, - ) -> Result<()> { - let delay = Duration::from_millis(self.delay_ms); - let delay = minimum.map_or(delay, |minimum| minimum.max(delay)); - let remaining = deadline - .checked_duration_since(std::time::Instant::now()) - .ok_or(Error::Fenced)?; - if delay >= remaining { - return Err(Error::Fenced); - } - tokio::time::sleep(delay).await; - self.delay_ms = self.delay_ms.saturating_mul(2).min(MAX_RETRY_DELAY_MS); - Ok(()) - } -} diff --git a/crates/crab-cell-runtime/src/test_support.rs b/crates/crab-cell-runtime/src/test_support.rs deleted file mode 100644 index ee7956a3e..000000000 --- a/crates/crab-cell-runtime/src/test_support.rs +++ /dev/null @@ -1,205 +0,0 @@ -//! Test-only object-store instrumentation, gated behind the `test-support` feature. -use std::{ - fmt, - fs::OpenOptions, - path::{Path as FilePath, PathBuf}, - sync::{ - Arc, - atomic::{AtomicBool, Ordering}, - }, - time::Duration, -}; - -use async_trait::async_trait; - -use crate::identity::encode_hex; -use fs4::fs_std::FileExt; -use futures_util::stream::BoxStream; -use object_store::{ - CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, - ObjectStoreExt, PutMode, PutMultipartOptions, PutOptions, PutPayload, PutResult, RenameOptions, - Result, UpdateVersion, path::Path, -}; - -/// Test-only local object store with cross-process conditional updates. -/// -/// `LocalFileSystem` deliberately does not implement `PutMode::Update`. This -/// adapter supplies the conditional control-record operation needed by the -/// process movement qualification while retaining the real local filesystem -/// and atomic object writes for every other operation. -#[derive(Clone, Debug)] -pub struct FilesystemCasStore { - inner: Arc, - root: Arc, - drop_next_update_response: Arc, - dropped_update_response: Arc, -} - -impl FilesystemCasStore { - /// Wraps a local directory, creating the lock directory conditional updates - /// need. - pub fn new(root: &FilePath) -> Result { - let inner = object_store::local::LocalFileSystem::new_with_prefix(root)?; - std::fs::create_dir_all(root.join(".crab-cas-locks")).map_err(|source| { - object_store::Error::Generic { - store: "filesystem-cas-store", - source: Box::new(source), - } - })?; - Ok(Self { - inner: Arc::new(inner), - root: Arc::new(root.to_owned()), - drop_next_update_response: Arc::new(AtomicBool::new(false)), - dropped_update_response: Arc::new(AtomicBool::new(false)), - }) - } - - /// Makes the next successful conditional update return a transport error. - /// - /// The write remains committed, modeling a lost response after the - /// authority has durably applied the release. - #[cfg_attr(test, allow(dead_code))] - pub fn drop_next_update_response(&self) { - self.drop_next_update_response - .store(true, Ordering::Release); - } - - /// Reports whether `drop_next_update_response` fired since construction. - #[must_use] - #[cfg_attr(test, allow(dead_code))] - pub fn dropped_update_response(&self) -> bool { - self.dropped_update_response.load(Ordering::Acquire) - } - - fn lock_path(&self, location: &Path) -> PathBuf { - let digest = blake3::hash(location.as_ref().as_bytes()); - self.root - .join(".crab-cas-locks") - .join(format!("{}.lock", encode_hex(digest.as_bytes()))) - } - - async fn update( - &self, - location: &Path, - payload: PutPayload, - mut options: PutOptions, - expected: UpdateVersion, - ) -> Result { - let lock_path = self.lock_path(location); - let lock = OpenOptions::new() - .create(true) - .truncate(false) - .read(true) - .write(true) - .open(lock_path) - .map_err(|source| object_store::Error::Generic { - store: "filesystem-cas-store", - source: Box::new(source), - })?; - loop { - if FileExt::try_lock_exclusive(&lock).map_err(|source| { - object_store::Error::Generic { - store: "filesystem-cas-store", - source: Box::new(source), - } - })? { - break; - } - tokio::time::sleep(Duration::from_millis(1)).await; - } - - let result = async { - let current = self.inner.head(location).await?; - let actual = UpdateVersion { - e_tag: current.e_tag, - version: current.version, - }; - if actual != expected { - return Err(object_store::Error::Precondition { - path: location.to_string(), - source: Box::new(std::io::Error::new( - std::io::ErrorKind::WouldBlock, - "conditional filesystem update lost the race", - )), - }); - } - options.mode = PutMode::Overwrite; - let result = self.inner.put_opts(location, payload, options).await?; - if self.drop_next_update_response.swap(false, Ordering::AcqRel) { - self.dropped_update_response.store(true, Ordering::Release); - return Err(object_store::Error::Generic { - store: "filesystem-cas-store", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::ConnectionReset, - "conditional update response lost after commit", - )), - }); - } - Ok(result) - } - .await; - FileExt::unlock(&lock).map_err(|source| object_store::Error::Generic { - store: "filesystem-cas-store", - source: Box::new(source), - })?; - result - } -} - -#[async_trait] -impl ObjectStore for FilesystemCasStore { - async fn put_opts( - &self, - location: &Path, - payload: PutPayload, - options: PutOptions, - ) -> Result { - match options.mode.clone() { - PutMode::Update(expected) => self.update(location, payload, options, expected).await, - PutMode::Create | PutMode::Overwrite => { - self.inner.put_opts(location, payload, options).await - } - } - } - - async fn put_multipart_opts( - &self, - location: &Path, - options: PutMultipartOptions, - ) -> Result> { - self.inner.put_multipart_opts(location, options).await - } - - async fn get_opts(&self, location: &Path, options: GetOptions) -> Result { - self.inner.get_opts(location, options).await - } - - fn delete_stream( - &self, - locations: BoxStream<'static, Result>, - ) -> BoxStream<'static, Result> { - self.inner.delete_stream(locations) - } - - fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, Result> { - self.inner.list(prefix) - } - - async fn list_with_delimiter(&self, prefix: Option<&Path>) -> Result { - self.inner.list_with_delimiter(prefix).await - } - - async fn copy_opts(&self, from: &Path, to: &Path, options: CopyOptions) -> Result<()> { - self.inner.copy_opts(from, to, options).await - } - - async fn rename_opts(&self, from: &Path, to: &Path, options: RenameOptions) -> Result<()> { - self.inner.rename_opts(from, to, options).await - } -} - -impl fmt::Display for FilesystemCasStore { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter.write_str("filesystem-cas-store") - } -} diff --git a/crates/crab-cell-runtime/tests-allow-list.txt b/crates/crab-cell-runtime/tests-allow-list.txt deleted file mode 100644 index 63d1de925..000000000 --- a/crates/crab-cell-runtime/tests-allow-list.txt +++ /dev/null @@ -1,94 +0,0 @@ -bin/qualification_receipt.rs # binary target; libtest cannot be imported by tests/ -cell/actor/tests.rs # asserts private projection helpers -cell/actor.rs # declares crate-private test modules -cell/executor/tests.rs # asserts private pending-publication state -cell/executor.rs # declares crate-private test modules -cell/schema.rs # compares the private RUNTIME_SCHEMA constant -cell/worker/tests.rs # exercises private worker admission and bootstrap -cell/worker.rs # declares crate-private test modules -client/routing.rs # exercises private in-flight selection and cancellation guards -client/tests.rs # exercises the private transport/encoding seam -client.rs # declares crate-private test modules -codec.rs # declares the shared bounded wire round trip for in-src codec tests -control.rs # declares crate-private test modules -control/tests.rs # asserts private control transition and codec helpers -coordination/sim.rs # deterministic simulator for the pub(crate) kernel -coordination/tests.rs # declares crate-private kernel test modules -coordination/tests/effects.rs # asserts exact effect completion and deactivation fences -coordination/tests/lifecycle.rs # asserts private admission, drain, and fence transitions -coordination/tests/publication.rs # asserts private publication outcomes and CAS loss -coordination/tests/scheduler.rs # asserts private kernel scheduling decisions -coordination/tests/transfer.rs # asserts private transfer and migration transitions -coordination.rs # pure pub(crate) kernel; declares its test modules -fleet/eviction.rs # asserts private eviction observations and victim selection -fleet/placement/tests.rs # uses private observation-age bounds -fleet/placement.rs # declares crate-private test modules -fleet/resource.rs # asserts the private resource ledger -fleet/scheduler.rs # asserts the private primitive probe and class budget -fleet/telemetry.rs # uses the private telemetry install seam -follower/tests.rs # asserts private lane and index constants -follower/tests/append.rs # asserts private append, seal, and retire framing -follower/tests/budget.rs # asserts private lane budget and collection bounds -follower/tests/scan.rs # asserts private tail scan and fail-closed reads -follower.rs # declares crate-private test modules -identity.rs # asserts private hex codec helpers -node/durability/tests.rs # asserts private durability gate state -node/durability.rs # declares crate-private test modules -node/lease.rs # uses the private lease bound constant -node/log/tests.rs # asserts private proof state -node/log.rs # declares crate-private test modules -node/log_recovery/tests.rs # uses private recovery page bounds and witness records -node/log_recovery.rs # declares crate-private test modules -node/log_shipper/tests.rs # asserts private shipping queue internals -node/log_shipper.rs # declares crate-private test modules -node/tests/candidates.rs # asserts private recovery-candidate windows and snapshots -node/tests/log.rs # asserts private node-log enrollment and takeover authority -node/tests/placement.rs # asserts private signed-capacity placement -node/tests/records.rs # asserts private advertisement records -node/tests/sessions.rs # asserts private session listing, claims, and collection -node/tests.rs # declares crate-private test modules; holds the shared fixtures -node.rs # declares crate-private test modules -peer/tests.rs # uses the private peer wire bound -peer.rs # declares crate-private test modules -primitives/activity_pool/tests.rs # drives the private activity job type -primitives/activity_pool.rs # declares crate-private test modules -primitives/blob/api/codec.rs # asserts private blob wire envelopes -primitives/blob/tests.rs # uses private schema and digest helpers -primitives/blob.rs # declares crate-private test modules -primitives/cron.rs # asserts private cron wire envelopes -primitives/effects/api.rs # asserts private effect wire envelopes -primitives/effects/tests.rs # drives the private effect batch -primitives/effects.rs # declares crate-private test modules -primitives/kv/api.rs # asserts private kv wire envelopes -primitives/kv.rs # uses the private wire bound -primitives/maintenance.rs # uses private transfer-work inventory constants -primitives/queue/api/codec.rs # asserts private queue wire envelopes and limits -primitives/queue/tests.rs # asserts private queue schema and limits -primitives/queue.rs # declares crate-private test modules -primitives/sql/api.rs # asserts private SQL wire envelopes -primitives/workflow/activity_api.rs # asserts private activity lease bounds -primitives/workflow/activity_api/supervisor.rs # asserts the private attempt cancellation guard -primitives/workflow/api/codec.rs # asserts private workflow wire envelopes -primitives/workflow/tests.rs # uses the private workflow schema and codec -primitives/workflow.rs # declares crate-private test modules -publication/tests.rs # asserts private publisher state -publication.rs # declares crate-private test modules -qualification/cluster.rs # validates private cluster receipt helpers -qualification/tests/evidence.rs # asserts private execution evidence watermarks -qualification/tests/matrix.rs # asserts private matrix manifests and row recomputation -qualification/tests/profile.rs # asserts private threshold and environment profiles -qualification/tests/receipt.rs # declares crate-private receipt test modules -qualification/tests/receipt/artifacts.rs # asserts private run-artifact binding and lifecycle coverage -qualification/tests/receipt/receipts.rs # asserts private receipt encoding and measurement bounds -qualification/tests/receipt/verification.rs # asserts private runner and protected verification rules -qualification/tests/workload.rs # asserts private workload generation and bounds -qualification/tests.rs # declares crate-private test modules; holds the shared executor fixtures -qualification.rs # declares crate-private test modules -recovery/artifacts.rs # asserts the private artifact cache -recovery/manifest/tests.rs # uses private manifest bounds and codecs -recovery/manifest.rs # declares crate-private test modules -recovery/release/tests.rs # asserts private release record fields -recovery/release.rs # declares crate-private test modules -recovery/retention/tests.rs # asserts private retention scratch handling -recovery/retention.rs # declares crate-private test modules -registry/descriptor.rs # asserts private descriptor codecs diff --git a/crates/crab-cell-runtime/tests/contracts.rs b/crates/crab-cell-runtime/tests/contracts.rs deleted file mode 100644 index 721f90e00..000000000 --- a/crates/crab-cell-runtime/tests/contracts.rs +++ /dev/null @@ -1,9 +0,0 @@ -//! Wire codec and registry contract tests. - -mod contracts { - pub mod application; - pub mod authority; - pub mod codec; - pub mod codec_properties; - pub mod registry; -} diff --git a/crates/crab-cell-runtime/tests/contracts/application.rs b/crates/crab-cell-runtime/tests/contracts/application.rs deleted file mode 100644 index a12cbfc64..000000000 --- a/crates/crab-cell-runtime/tests/contracts/application.rs +++ /dev/null @@ -1,66 +0,0 @@ -//! Application identity tests extracted from `src/application.rs`. -use crab_cell_runtime::cell::application::{ApplicationIdentity, ApplicationIdentityStore}; -use crab_cell_runtime::ltx::CellStorageLayout; - -use std::sync::Arc; - -use bytes::Bytes; -use crab_storage::Store; -use object_store::memory::InMemory; -use object_store::path::Path; - -use crab_cell_runtime::*; - -fn identity(byte: u8) -> ApplicationIdentity { - ApplicationIdentity::new( - TenantId::from_bytes([byte; 16]), - ApplicationId::from_bytes([byte.wrapping_add(1); 16]), - ) -} - -#[tokio::test] -async fn initialization_is_idempotent_and_rejects_another_application() { - let store = Store::new(Arc::new(InMemory::new())); - let identities = ApplicationIdentityStore::new(store, Path::from("root")); - let first = identity(1); - - assert_eq!(identities.initialize(first).await.unwrap(), first); - assert_eq!(identities.initialize(first).await.unwrap(), first); - assert!(matches!( - identities.initialize(identity(9)).await, - Err(Error::Identity(_)) - )); - assert_eq!(identities.load().await.unwrap(), Some(first)); - assert_eq!( - identities.layout(first).await.unwrap().application_id(), - first.application().as_bytes() - ); -} - -#[tokio::test] -async fn identity_reader_rejects_noncanonical_or_unknown_fields() { - let store = Store::new(Arc::new(InMemory::new())); - let identities = ApplicationIdentityStore::new(store.clone(), Path::from("root")); - store - .create_strict( - &CellStorageLayout::root_identity_path(&Path::from("root")), - Bytes::from_static( - br#"{ "application":"02020202020202020202020202020202","tenant":"01010101010101010101010101010101","version":1}"#, - ), - ) - .await - .unwrap(); - assert!(matches!(identities.load().await, Err(Error::Identity(_)))); - - let other = ApplicationIdentityStore::new(store.clone(), Path::from("other")); - store - .create_strict( - &CellStorageLayout::root_identity_path(&Path::from("other")), - Bytes::from_static( - br#"{"application":"02020202020202020202020202020202","tenant":"01010101010101010101010101010101","unexpected":true,"version":1}"#, - ), - ) - .await - .unwrap(); - assert!(matches!(other.load().await, Err(Error::Json(_)))); -} diff --git a/crates/crab-cell-runtime/tests/contracts/authority.rs b/crates/crab-cell-runtime/tests/contracts/authority.rs deleted file mode 100644 index c7f6aa1b3..000000000 --- a/crates/crab-cell-runtime/tests/contracts/authority.rs +++ /dev/null @@ -1,90 +0,0 @@ -//! Authority CAS tests. -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::control::{Control, Transition}; -use crab_cell_runtime::ltx::CellStorageLayout; - -use std::sync::Arc; - -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use bytes::Bytes; -use crab_cell_runtime::*; - -use crab_cell_runtime::control::{ControlState, Owner, RootRef}; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{Digest, SessionId}; - -fn control() -> Control { - Control::initial( - CellId::from_bytes([1; 32]), - IncarnationId::from_bytes([2; 16]), - Owner { - session: SessionId::from_bytes([3; 16]), - endpoint: "https://node.internal:8081".into(), - }, - Digest::from_bytes([4; 32]), - 1, - ) - .unwrap() -} - -#[tokio::test] -async fn stale_control_token_cannot_overwrite_winning_publication() { - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store, Path::from("root"), [8; 16]); - let authority = CellAuthority::new(layout.clone()); - let initial = control(); - layout - .store() - .create_strict( - &layout.control_path(initial.cell.as_bytes()), - Bytes::from(initial.encode().unwrap()), - ) - .await - .unwrap(); - let first = authority.load(initial.cell).await.unwrap().unwrap(); - let stale = authority.load(initial.cell).await.unwrap().unwrap(); - - let mut published = initial.clone(); - published.revision += 1; - published.progress += 1; - published.state = ControlState::Serving; - published.root = Some(RootRef { - digest: Digest::from_bytes([9; 32]), - txid: 1, - checksum: (1 << 63) | 7, - commit_sequence: 1, - }); - let winner = authority - .transition(&first, published.clone(), Transition::Publish) - .await - .unwrap(); - assert_eq!(winner.value(), &published); - - let mut stale_publish = initial; - stale_publish.revision += 1; - stale_publish.progress += 1; - stale_publish.state = ControlState::Serving; - stale_publish.root = Some(RootRef { - digest: Digest::from_bytes([5; 32]), - txid: 1, - checksum: (1 << 63) | 8, - commit_sequence: 1, - }); - assert!( - authority - .transition(&stale, stale_publish, Transition::Publish) - .await - .is_err() - ); - assert_eq!( - authority - .load(published.cell) - .await - .unwrap() - .unwrap() - .value(), - &published - ); -} diff --git a/crates/crab-cell-runtime/tests/contracts/codec.rs b/crates/crab-cell-runtime/tests/contracts/codec.rs deleted file mode 100644 index e17f716ff..000000000 --- a/crates/crab-cell-runtime/tests/contracts/codec.rs +++ /dev/null @@ -1,113 +0,0 @@ -use crab_cell_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; - -fn roundtrip(value: T, limit: u32) -> T -where - T: WireValue, -{ - let mut encoder = BoundedEncoder::new(limit).unwrap(); - value.encode(&mut encoder).unwrap(); - let bytes = encoder.finish(); - let mut decoder = BoundedDecoder::new(&bytes, limit).unwrap(); - let decoded = T::decode(&mut decoder).unwrap(); - decoder.finish().unwrap(); - decoded -} - -#[test] -fn signed_extremes_and_length_delimited_values_roundtrip_exactly() { - assert_eq!(roundtrip(i64::MIN, 8), i64::MIN); - assert_eq!(roundtrip(i64::MAX, 8), i64::MAX); - assert_eq!( - roundtrip(Some("crab".to_owned()), 16), - Some("crab".to_owned()) - ); - assert_eq!(roundtrip(vec![0, 1, 255], 16), vec![0, 1, 255]); -} - -#[test] -fn invalid_tags_noncanonical_floats_and_trailing_bytes_fail_closed() { - let mut invalid_bool = BoundedDecoder::new(&[2], 1).unwrap(); - assert!(matches!( - bool::decode(&mut invalid_bool), - Err(CodecError::Invalid("invalid bool tag")) - )); - - let negative_zero_bytes = (-0.0_f64).to_bits().to_be_bytes(); - let mut negative_zero = BoundedDecoder::new(&negative_zero_bytes, 8).unwrap(); - assert!(matches!( - f64::decode(&mut negative_zero), - Err(CodecError::Invalid("noncanonical f64")) - )); - - let mut trailing = BoundedDecoder::new(&[1, 0], 2).unwrap(); - assert!(bool::decode(&mut trailing).unwrap()); - assert!(matches!( - trailing.finish(), - Err(CodecError::Invalid("trailing wire bytes")) - )); -} - -#[test] -fn encoder_and_decoder_enforce_declared_limits_before_allocation() { - let mut encoder = BoundedEncoder::new(4).unwrap(); - assert!(matches!( - b"payload".to_vec().encode(&mut encoder), - Err(CodecError::Limit) - )); - assert!(matches!( - BoundedDecoder::new(&[0; 5], 4), - Err(CodecError::Limit) - )); - assert!(matches!( - BoundedEncoder::new(4 * 1024 * 1024 + 64 * 1024 + 1), - Err(CodecError::Limit) - )); -} - -#[test] -fn decoder_rejects_invalid_utf8_text() { - let bytes = [0, 0, 0, 2, 0xff, 0xfe]; - let mut decoder = BoundedDecoder::new(&bytes, 6).unwrap(); - assert!(matches!( - String::decode(&mut decoder), - Err(CodecError::Utf8(_)) - )); -} - -#[test] -fn length_delimited_values_reject_a_length_past_the_input() { - let bytes = [0, 0, 0, 8]; - let mut decoder = BoundedDecoder::new(&bytes, 4).unwrap(); - assert!(matches!( - Vec::::decode(&mut decoder), - Err(CodecError::Invalid("truncated wire value")) - )); -} - -#[test] -fn zero_limits_are_rejected() { - assert!(matches!(BoundedEncoder::new(0), Err(CodecError::Limit))); - assert!(matches!( - BoundedDecoder::new(&[], 0), - Err(CodecError::Limit) - )); -} - -#[test] -fn non_finite_floats_cannot_be_encoded() { - let mut encoder = BoundedEncoder::new(8).unwrap(); - assert!(matches!( - f64::NAN.encode(&mut encoder), - Err(CodecError::Invalid("non-finite f64")) - )); - assert!(matches!( - f64::INFINITY.encode(&mut encoder), - Err(CodecError::Invalid("non-finite f64")) - )); -} - -#[test] -fn negative_zero_encodes_as_canonical_zero() { - let decoded = roundtrip(-0.0_f64, 8); - assert_eq!(decoded.to_bits(), 0.0_f64.to_bits()); -} diff --git a/crates/crab-cell-runtime/tests/contracts/codec_properties.rs b/crates/crab-cell-runtime/tests/contracts/codec_properties.rs deleted file mode 100644 index daa5739d5..000000000 --- a/crates/crab-cell-runtime/tests/contracts/codec_properties.rs +++ /dev/null @@ -1,142 +0,0 @@ -//! Properties every bounded wire decoder must hold for arbitrary input. -//! -//! The decoders parse peer, client, and storage bytes, so two properties matter -//! beyond the fixtures: a decode never panics, and whatever it accepts whole is -//! canonical — re-encoding the value reproduces exactly the input bytes. - -use crab_cell_runtime::codec::{BoundedDecoder, BoundedEncoder, WireValue}; -use crab_cell_runtime::primitives::effects::{ - EffectAckRequest, EffectClaimRequest, EffectLeaseRequest, EffectStatusRequest, - EffectValidateRequest, -}; -use crab_cell_runtime::primitives::kv::{KvGetRequest, KvListRequest}; -use crab_cell_runtime::primitives::queue::{ - QueueClaimRequest, QueueInfoRequest, QueueLeaseRequest, QueueValidateRequest, -}; -use crab_cell_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue}; -use crab_cell_runtime::primitives::workflow::{ - ActivityClaim, ActivityCompletion, ActivityCompletionOutcome, ActivityLeaseOutcome, - WorkflowActivityClaimRequest, WorkflowActivityExtendRequest, WorkflowActivityValidateRequest, -}; -use proptest::prelude::*; - -const LIMIT: u32 = 4 * 1024 * 1024; - -/// Decodes arbitrary bytes, then re-encodes anything the decoder accepted whole. -fn accepts_only_canonical(bytes: &[u8]) { - let Ok(mut decoder) = BoundedDecoder::new(bytes, LIMIT) else { - return; - }; - let Ok(value) = T::decode(&mut decoder) else { - return; - }; - if decoder.finish().is_err() { - // A value that does not consume its input is not the canonical encoding - // of anything; `finish` is what rejects it. - return; - } - let mut encoder = BoundedEncoder::new(LIMIT).expect("the limit is valid"); - value - .encode(&mut encoder) - .expect("an accepted value re-encodes"); - assert_eq!(encoder.finish(), bytes, "accepted input is not canonical"); -} - -/// One decoder per index, so one property covers every public wire shape. -fn check(index: usize, bytes: &[u8]) { - match index { - 0 => accepts_only_canonical::(bytes), - 1 => accepts_only_canonical::(bytes), - 2 => accepts_only_canonical::(bytes), - 3 => accepts_only_canonical::(bytes), - 4 => accepts_only_canonical::(bytes), - 5 => accepts_only_canonical::(bytes), - 6 => accepts_only_canonical::(bytes), - 7 => accepts_only_canonical::(bytes), - 8 => accepts_only_canonical::(bytes), - 9 => accepts_only_canonical::(bytes), - 10 => accepts_only_canonical::(bytes), - 11 => accepts_only_canonical::(bytes), - 12 => accepts_only_canonical::(bytes), - 13 => accepts_only_canonical::(bytes), - 14 => accepts_only_canonical::(bytes), - 15 => accepts_only_canonical::(bytes), - 16 => accepts_only_canonical::(bytes), - 17 => accepts_only_canonical::(bytes), - 18 => accepts_only_canonical::(bytes), - 19 => accepts_only_canonical::(bytes), - 20 => accepts_only_canonical::(bytes), - _ => unreachable!("every decoder under test has an index"), - } -} - -/// A valid encoding of one representative value per index, for the mutation -/// property below: random bytes rarely reach a decoder's deeper paths. -fn fixture_encoding(index: usize) -> Vec { - let mut encoder = BoundedEncoder::new(LIMIT).expect("the limit is valid"); - match index { - 0 => KvGetRequest { - scope: b"tenant".to_vec(), - key: b"key".to_vec(), - } - .encode(&mut encoder) - .expect("a bounded value encodes"), - _ => QueueClaimRequest { - limit: 4, - lease_ms: 30_000, - } - .encode(&mut encoder) - .expect("a bounded value encodes"), - } - encoder.finish() -} - -proptest! { - #![proptest_config(ProptestConfig { cases: 256, ..ProptestConfig::default() })] - - /// Arbitrary bytes never panic a decoder, and an accepted value re-encodes to - /// exactly the bytes it was decoded from. - #[test] - fn primitive_decoders_accept_only_canonical_bytes( - bytes in prop::collection::vec(any::(), 0..512), - index in 0usize..21, - ) { - check(index, &bytes); - } - - /// A valid encoding with a run of bytes replaced must still decode without - /// panicking, and anything it accepts whole must stay canonical. - #[test] - fn mutated_valid_encodings_stay_total_and_canonical( - index in 0usize..2, - position in any::(), - replacement in any::(), - run in 1usize..8, - ) { - let mut bytes = fixture_encoding(index); - if !bytes.is_empty() { - let mut at = position % bytes.len(); - for _ in 0..run { - bytes[at] = replacement; - at = (at + 1) % bytes.len(); - } - } - match index { - 0 => accepts_only_canonical::(&bytes), - _ => accepts_only_canonical::(&bytes), - } - } -} - -/// Both branches of the property above, spelled out: a canonical encoding passes -/// the check, and one with a trailing byte is rejected by `finish` rather than -/// being reported as non-canonical. -#[test] -fn canonical_encodings_roundtrip_and_trailing_bytes_are_rejected() { - let bytes = fixture_encoding(0); - accepts_only_canonical::(&bytes); - - let mut extended = bytes.clone(); - extended.push(0); - accepts_only_canonical::(&extended); -} diff --git a/crates/crab-cell-runtime/tests/contracts/registry.rs b/crates/crab-cell-runtime/tests/contracts/registry.rs deleted file mode 100644 index 437d2d49f..000000000 --- a/crates/crab-cell-runtime/tests/contracts/registry.rs +++ /dev/null @@ -1,671 +0,0 @@ -use std::sync::OnceLock; - -use crab_cell_runtime::Error; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::executor::HandlerOutcome; -use crab_cell_runtime::codec::{BoundedDecoder, BoundedEncoder, WireValue}; -use crab_cell_runtime::identity::{ - ApplicationId, CellId, CellTarget, Digest, NamespaceId, TenantId, -}; -use crab_cell_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, Command, ModuleDescriptor, NamespaceDescriptor, Query, Registry, - RegistryBuilder, -}; -use crab_cell_runtime::registry::{ - CommandContext, CommandInvocation, CommandResult, MigrationDescriptor, OperationDescriptor, - QueryContext, QueryInvocation, -}; - -const MIGRATION: &str = "CREATE TABLE items(value BLOB NOT NULL)"; -const COMMAND: OperationDescriptor = OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 16, - output_limit: 16, -}; -const QUERY: OperationDescriptor = OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 1, - output_limit: 16, -}; - -struct FirstModule; - -impl CellModule for FirstModule { - const NAME: &'static str = "first"; - - fn descriptor(&self) -> &'static ModuleDescriptor { - first_descriptor() - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - registry.bind_command::()?; - registry.bind_query::()?; - Ok(()) - } -} - -struct SecondModule; - -impl CellModule for SecondModule { - const NAME: &'static str = "second"; - - fn descriptor(&self) -> &'static ModuleDescriptor { - second_descriptor() - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - registry.bind_command::()?; - registry.bind_query::()?; - Ok(()) - } -} - -struct MissingBinding; - -impl CellModule for MissingBinding { - const NAME: &'static str = "first"; - - fn descriptor(&self) -> &'static ModuleDescriptor { - first_descriptor() - } - - fn register(self, _registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - Ok(()) - } -} - -fn first_descriptor() -> &'static ModuleDescriptor { - static DESCRIPTOR: OnceLock = OnceLock::new(); - DESCRIPTOR.get_or_init(|| descriptor("first", 1)) -} - -fn second_descriptor() -> &'static ModuleDescriptor { - static DESCRIPTOR: OnceLock = OnceLock::new(); - DESCRIPTOR.get_or_init(|| descriptor("second", 2)) -} - -fn descriptor(name: &'static str, namespace: u8) -> ModuleDescriptor { - let migrations = Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: MIGRATION, - digest: Digest::from_bytes(*blake3::hash(MIGRATION.as_bytes()).as_bytes()), - }])); - let namespaces = Box::leak(Box::new([NamespaceDescriptor { - id: NamespaceId::from_bytes([namespace; 16]), - name, - role: CatalogRole::Repository, - shards: 1, - effect_targets: &[], - dead_letter: None, - }])); - ModuleDescriptor { - name, - source_digest: Digest::from_bytes([namespace; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations, - commands: &[COMMAND], - queries: &[QUERY], - workflow_definitions: &[], - activity_types: &[], - namespaces, - } -} - -fn build() -> BuildDescriptor { - BuildDescriptor { - source_revision: "0123456789abcdef".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - } -} - -fn build_registry(reverse: bool) -> Registry { - let mut builder = RegistryBuilder::new(build()); - if reverse { - builder.register(SecondModule).unwrap(); - builder.register(FirstModule).unwrap(); - } else { - builder.register(FirstModule).unwrap(); - builder.register(SecondModule).unwrap(); - } - builder.finish().unwrap() -} - -/// A module whose descriptor a test mutates before registration. -struct DescriptorModule(&'static ModuleDescriptor); - -impl CellModule for DescriptorModule { - const NAME: &'static str = "first"; - - fn descriptor(&self) -> &'static ModuleDescriptor { - self.0 - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - registry.bind_command::()?; - registry.bind_query::()?; - Ok(()) - } -} - -fn finish_with(descriptor: &'static ModuleDescriptor) -> crab_cell_runtime::Result<()> { - let mut builder = RegistryBuilder::new(build()); - builder.register(DescriptorModule(descriptor))?; - builder.finish().map(|_| ()) -} - -fn leaked(descriptor: ModuleDescriptor) -> &'static ModuleDescriptor { - Box::leak(Box::new(descriptor)) -} - -struct Insert; - -impl Command for Insert { - const MODULE: &'static str = "first"; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - type Input = Vec; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crab_cell_runtime::Result> { - let result = context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "INSERT INTO items(value) VALUES (?)".into(), - parameters: vec![SqlValue::Blob(input.clone())], - }], - })?; - if result[0].rows_affected != 1 { - return Err(Error::Command("test insert did not affect one row")); - } - Ok(CommandResult::Success(input)) - } -} - -struct Read; - -impl Query for Read { - const MODULE: &'static str = "first"; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - type Input = (); - type Output = Vec; - - fn execute( - context: &mut QueryContext<'_>, - _input: Self::Input, - ) -> crab_cell_runtime::Result { - let result = context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT value FROM items ORDER BY rowid".into(), - parameters: Vec::new(), - }], - })?; - match result[0].rows.first().and_then(|row| row.first()) { - Some(SqlValue::Blob(value)) => Ok(value.clone()), - _ => Err(Error::Command("test query returned no blob")), - } - } -} - -struct SecondInsert; - -impl Command for SecondInsert { - const MODULE: &'static str = "second"; - const ID: u32 = Insert::ID; - const CODEC_VERSION: u32 = Insert::CODEC_VERSION; - type Input = Vec; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crab_cell_runtime::Result> { - Insert::execute(context, input) - } -} - -struct SecondRead; - -impl Query for SecondRead { - const MODULE: &'static str = "second"; - const ID: u32 = Read::ID; - const CODEC_VERSION: u32 = Read::CODEC_VERSION; - type Input = (); - type Output = Vec; - - fn execute( - context: &mut QueryContext<'_>, - input: Self::Input, - ) -> crab_cell_runtime::Result { - Read::execute(context, input) - } -} - -struct Extra; - -impl Command for Extra { - const MODULE: &'static str = "first"; - const ID: u32 = 2; - const CODEC_VERSION: u32 = 1; - type Input = (); - type Output = (); - - fn execute( - _context: &mut CommandContext<'_, '_>, - _input: Self::Input, - ) -> crab_cell_runtime::Result> { - Ok(CommandResult::Success(())) - } -} - -fn wire(value: &T, limit: u32) -> Vec { - let mut encoder = BoundedEncoder::new(limit).unwrap(); - value.encode(&mut encoder).unwrap(); - encoder.finish() -} - -fn decode(bytes: &[u8], limit: u32) -> T { - let mut decoder = BoundedDecoder::new(bytes, limit).unwrap(); - let value = T::decode(&mut decoder).unwrap(); - decoder.finish().unwrap(); - value -} - -#[test] -fn compiled_registry_is_canonical_and_executes_only_declared_bindings() { - let registry = build_registry(false); - let reversed = build_registry(true); - assert_eq!(registry.release_bytes(), reversed.release_bytes()); - assert_eq!(registry.release_digest(), reversed.release_digest()); - let first_code = registry.module_code("first").unwrap(); - assert!(registry.supports_cell( - NamespaceId::from_bytes([1; 16]), - CatalogRole::Repository, - first_code, - 1, - )); - for (namespace, role, code, schema) in [ - ( - NamespaceId::from_bytes([9; 16]), - CatalogRole::Repository, - first_code, - 1, - ), - ( - NamespaceId::from_bytes([1; 16]), - CatalogRole::Kv, - first_code, - 1, - ), - ( - NamespaceId::from_bytes([1; 16]), - CatalogRole::Repository, - Digest::from_bytes([8; 32]), - 1, - ), - ( - NamespaceId::from_bytes([1; 16]), - CatalogRole::Repository, - first_code, - 2, - ), - ] { - assert!(!registry.supports_cell(namespace, role, code, schema)); - } - - let mut connection = crab_ltx::rusqlite::Connection::open_in_memory().unwrap(); - connection.execute_batch(MIGRATION).unwrap(); - let transaction = connection.transaction().unwrap(); - let input = wire(&b"value".to_vec(), 16); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NamespaceId::from_bytes([1; 16]), - b"registry-test", - ) - .unwrap(); - let outcome = registry - .execute_command( - &transaction, - CommandInvocation { - module: "first", - operation_id: 1, - codec_version: 1, - schema: 1, - target, - sequence: 1, - now_ms: 10, - input: &input, - }, - ) - .unwrap(); - assert!( - matches!(outcome, HandlerOutcome::Success(ref value) if decode::>(value, 16) == b"value") - ); - transaction.commit().unwrap(); - assert_eq!( - registry - .execute_query( - &connection, - QueryInvocation { - module: "first", - operation_id: 1, - codec_version: 1, - schema: 1, - cell: CellId::from_bytes([3; 32]), - commit_sequence: 1, - now_ms: 10, - input: b"", - }, - ) - .unwrap(), - wire(&b"value".to_vec(), 16) - ); - assert!(matches!( - registry.execute_query( - &connection, - QueryInvocation { - module: "first", - operation_id: 1, - codec_version: 1, - schema: 2, - cell: CellId::from_bytes([3; 32]), - commit_sequence: 1, - now_ms: 10, - input: b"", - }, - ), - Err(Error::Command( - "registered operation does not support schema" - )) - )); -} - -#[test] -fn command_execution_rejects_a_module_targeting_another_namespace_owner() { - let registry = build_registry(false); - let mut connection = crab_ltx::rusqlite::Connection::open_in_memory().unwrap(); - connection.execute_batch(MIGRATION).unwrap(); - let transaction = connection.transaction().unwrap(); - let input = wire(&b"value".to_vec(), 16); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NamespaceId::from_bytes([2; 16]), - b"registry-test", - ) - .unwrap(); - - assert!(matches!( - registry.execute_command( - &transaction, - CommandInvocation { - module: "first", - operation_id: 1, - codec_version: 1, - schema: 1, - target, - sequence: 1, - now_ms: 10, - input: &input, - }, - ), - Err(Error::Registry("operation module does not own namespace")) - )); - let count: i64 = transaction - .query_row("SELECT COUNT(*) FROM items", [], |row| row.get(0)) - .unwrap(); - assert_eq!(count, 0); - transaction.rollback().unwrap(); -} - -#[test] -fn compiled_registry_rejects_descriptor_binding_drift() { - let mut missing = RegistryBuilder::new(build()); - missing.register(MissingBinding).unwrap(); - assert!(matches!( - missing.finish(), - Err(Error::Registry("descriptor and function bindings differ")) - )); - - let mut extra = RegistryBuilder::new(build()); - extra.register(FirstModule).unwrap(); - extra.bind_command::().unwrap(); - assert!(matches!( - extra.finish(), - Err(Error::Registry("descriptor and function bindings differ")) - )); -} - -#[test] -fn build_descriptor_bounds_are_enforced() { - let mut longest = RegistryBuilder::new(BuildDescriptor { - source_revision: "x".repeat(128), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }); - longest.register(FirstModule).unwrap(); - assert!(longest.finish().is_ok()); - - for (revision, digest) in [ - (String::new(), [9; 32]), - ("x".repeat(129), [9; 32]), - ("0123456789abcdef".into(), [0; 32]), - ] { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: revision, - cargo_lock_digest: Digest::from_bytes(digest), - }); - builder.register(FirstModule).unwrap(); - assert!(matches!( - builder.finish(), - Err(Error::Registry("invalid build descriptor")) - )); - } -} - -#[test] -fn module_descriptor_requires_a_live_schema_range() { - for descriptor in [ - ModuleDescriptor { - schema_min: 0, - ..*first_descriptor() - }, - ModuleDescriptor { - schema_max: 0, - ..*first_descriptor() - }, - ModuleDescriptor { - source_digest: Digest::from_bytes([0; 32]), - ..*first_descriptor() - }, - ] { - assert!(matches!( - finish_with(leaked(descriptor)), - Err(Error::Registry("invalid module descriptor")) - )); - } -} - -#[test] -fn migrations_must_be_contiguous_and_digest_bound() { - let digest = |sql: &str| Digest::from_bytes(*blake3::hash(sql.as_bytes()).as_bytes()); - let migration = |version| MigrationDescriptor { - version, - sql: MIGRATION, - digest: digest(MIGRATION), - }; - for descriptor in [ - ModuleDescriptor { - migrations: Box::leak(Box::new([migration(2)])), - ..*first_descriptor() - }, - ModuleDescriptor { - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: MIGRATION, - digest: Digest::from_bytes([7; 32]), - }])), - ..*first_descriptor() - }, - ModuleDescriptor { - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: "", - digest: digest(""), - }])), - ..*first_descriptor() - }, - ] { - assert!(matches!( - finish_with(leaked(descriptor)), - Err(Error::Registry("invalid migration inventory")) - )); - } - - let uncovered = ModuleDescriptor { - schema_max: 2, - ..*first_descriptor() - }; - assert!(matches!( - finish_with(leaked(uncovered)), - Err(Error::Registry( - "migration range does not cover module schema" - )) - )); -} - -#[test] -fn operation_limits_must_fit_the_wire_bound() { - const WIRE_BOUND: u32 = 4 * 1024 * 1024 + 64 * 1024; - for operation in [ - OperationDescriptor { - input_limit: 0, - ..COMMAND - }, - OperationDescriptor { - output_limit: 0, - ..COMMAND - }, - OperationDescriptor { - input_limit: WIRE_BOUND + 1, - ..COMMAND - }, - ] { - let descriptor = ModuleDescriptor { - commands: Box::leak(Box::new([operation])), - ..*first_descriptor() - }; - assert!(matches!( - finish_with(leaked(descriptor)), - Err(Error::Registry("invalid operation inventory")) - )); - } -} - -#[test] -fn namespace_ids_and_names_are_unique() { - let namespace = |id: u8, name: &'static str| NamespaceDescriptor { - id: NamespaceId::from_bytes([id; 16]), - name, - role: CatalogRole::Repository, - shards: 1, - effect_targets: &[], - dead_letter: None, - }; - let duplicate_id = ModuleDescriptor { - namespaces: Box::leak(Box::new([namespace(1, "first"), namespace(1, "other")])), - ..*first_descriptor() - }; - assert!(matches!( - finish_with(leaked(duplicate_id)), - Err(Error::Registry("duplicate namespace ID")) - )); - - let duplicate_name = ModuleDescriptor { - namespaces: Box::leak(Box::new([namespace(1, "first"), namespace(2, "first")])), - ..*first_descriptor() - }; - assert!(matches!( - finish_with(leaked(duplicate_name)), - Err(Error::Registry("invalid namespace inventory")) - )); -} - -#[test] -fn namespace_count_is_bounded() { - let namespaces = (0..129_u8) - .map(|index| NamespaceDescriptor { - id: NamespaceId::from_bytes([index; 16]), - name: Box::leak(format!("namespace-{index}").into_boxed_str()), - role: CatalogRole::Repository, - shards: 1, - effect_targets: &[], - dead_letter: None, - }) - .collect::>(); - let descriptor = ModuleDescriptor { - namespaces: Box::leak(namespaces.into_boxed_slice()), - ..*first_descriptor() - }; - assert!(matches!( - finish_with(leaked(descriptor)), - Err(Error::Registry("namespace count exceeds 128")) - )); -} - -#[test] -fn namespace_shards_must_be_a_bounded_power_of_two() { - for shards in [0, 3, 8192] { - let descriptor = ModuleDescriptor { - namespaces: Box::leak(Box::new([NamespaceDescriptor { - id: NamespaceId::from_bytes([1; 16]), - name: "first", - role: CatalogRole::Repository, - shards, - effect_targets: &[], - dead_letter: None, - }])), - ..*first_descriptor() - }; - assert!(matches!( - finish_with(leaked(descriptor)), - Err(Error::Registry("invalid namespace inventory")) - )); - } -} - -#[test] -fn dead_letter_cycles_are_rejected() { - let first = NamespaceId::from_bytes([11; 16]); - let second = NamespaceId::from_bytes([12; 16]); - let namespace = |id, name: &'static str, target| NamespaceDescriptor { - id, - name, - role: CatalogRole::Queue, - shards: 1, - effect_targets: Box::leak(Box::new([target])), - dead_letter: Some(target), - }; - let descriptor = ModuleDescriptor { - namespaces: Box::leak(Box::new([ - namespace(first, "first-queue", second), - namespace(second, "second-queue", first), - ])), - ..*first_descriptor() - }; - assert!(matches!( - finish_with(leaked(descriptor)), - Err(Error::Registry("dead-letter cycle")) - )); -} diff --git a/crates/crab-cell-runtime/tests/fleet.rs b/crates/crab-cell-runtime/tests/fleet.rs deleted file mode 100644 index b9c40140d..000000000 --- a/crates/crab-cell-runtime/tests/fleet.rs +++ /dev/null @@ -1,11 +0,0 @@ -//! Fleet-level integration tests: node leases, placement, pressure, cluster receipts. - -mod fleet { - pub mod node_log_transport; - pub mod placement_properties; - pub mod pressure; - pub mod pressure_properties; - pub mod read_placement; - pub mod read_policy; - pub mod takeover; -} diff --git a/crates/crab-cell-runtime/tests/fleet/node_log_transport.rs b/crates/crab-cell-runtime/tests/fleet/node_log_transport.rs deleted file mode 100644 index 20c134042..000000000 --- a/crates/crab-cell-runtime/tests/fleet/node_log_transport.rs +++ /dev/null @@ -1,70 +0,0 @@ -//! Node log transport tests extracted from `src/node_log_transport.rs`. -use crab_cell_runtime::follower::FollowerReceipt; -use crab_cell_runtime::identity::NodeId; -use crab_cell_runtime::node::log_transport::{ - AppendRequest, NodeLogTransport, RetireRequest, SealRequest, TailRequest, -}; - -use bytes::Bytes; -use futures_util::future::BoxFuture; - -use crab_cell_runtime::*; - -struct LegacyTransport { - frames: Vec, -} - -impl NodeLogTransport for LegacyTransport { - fn append<'a>( - &'a self, - _member: NodeId, - _request: AppendRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("append is not used by this test")) }) - } - - fn seal<'a>( - &'a self, - _member: NodeId, - _request: SealRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("seal is not used by this test")) }) - } - - fn retire<'a>( - &'a self, - _member: NodeId, - _request: RetireRequest, - ) -> BoxFuture<'a, Result> { - Box::pin(async { Err(Error::Node("retire is not used by this test")) }) - } - - fn tail<'a>( - &'a self, - _member: NodeId, - _request: TailRequest, - ) -> BoxFuture<'a, Result>> { - let frames = self.frames.clone(); - Box::pin(async move { Ok(frames) }) - } -} - -#[tokio::test] -async fn legacy_tail_fallback_preserves_page_boundaries() { - let transport = LegacyTransport { - frames: (0..4_097).map(|_| Bytes::from_static(b"frame")).collect(), - }; - let page = transport - .tail_page( - NodeId::from_bytes([1; 16]), - TailRequest { - leader_session: SessionId::from_bytes([2; 16]), - log_epoch: 1, - first_sequence: 10, - }, - ) - .await - .unwrap(); - assert_eq!(page.frames.len(), 4_096); - assert_eq!(page.next_sequence, Some(4_106)); -} diff --git a/crates/crab-cell-runtime/tests/fleet/placement_properties.rs b/crates/crab-cell-runtime/tests/fleet/placement_properties.rs deleted file mode 100644 index 70a1285f1..000000000 --- a/crates/crab-cell-runtime/tests/fleet/placement_properties.rs +++ /dev/null @@ -1,353 +0,0 @@ -//! Properties the weighted placement planner promises for one fleet snapshot. -//! -//! The planner is pure and deterministic, so its contracts are properties of a -//! generated snapshot: the same view must elect the same donor whatever order -//! the observations arrive in, an ineligible member never receives, a -//! pre-batch view moves nothing, and one snapshot never plans more than the -//! tick cap. - -use crab_cell_runtime::fleet::placement::{ - CellTransferDemand, FleetBalance, PlacementEligibility, PlacementObservation, PlacementPlanner, - PlacementPressure, -}; -use crab_cell_runtime::identity::{CellId, NodeId, SessionId}; -use proptest::prelude::*; - -const NOW_MS: i64 = 1_000_000; - -fn node(byte: u8) -> NodeId { - NodeId::from_bytes([byte; 16]) -} - -fn session(byte: u8) -> SessionId { - SessionId::from_bytes([byte; 16]) -} - -fn observation( - index: u8, - active_cells: u32, - max_active_cells: u32, - pressure: PlacementPressure, - draining: bool, -) -> PlacementObservation { - PlacementObservation { - node: node(index), - session: session(index), - observed_at_ms: NOW_MS, - memory_capacity_bytes: 1 << 30, - free_memory_bytes: 1 << 29, - disk_capacity_bytes: 1 << 30, - free_disk_bytes: 1 << 29, - active_cells, - max_active_cells, - running_jobs: 0, - job_capacity: 8, - publication_backlog: 0, - hydration_backlog: 0, - primitive_backlog: 0, - pressure, - draining, - authenticated: true, - current_owner: false, - } -} - -fn pressure() -> impl Strategy { - prop_oneof![ - Just(PlacementPressure::Normal), - Just(PlacementPressure::Constrained), - Just(PlacementPressure::Shedding), - Just(PlacementPressure::Critical), - ] -} - -/// Generates a small fleet whose capacities and counts collide often enough for -/// ties to appear. -fn fleet() -> impl Strategy> { - prop::collection::vec((0u32..4, 1u32..4, pressure(), any::()), 1..5).prop_map(|entries| { - entries - .into_iter() - .enumerate() - .map(|(index, (active, max_active, pressure, draining))| { - let active = active.min(max_active); - observation(index as u8, active, max_active, pressure, draining) - }) - .collect() - }) -} - -fn balance(observations: &[PlacementObservation], since_ms: i64) -> Option { - PlacementPlanner::default() - .fleet_balance(NOW_MS, observations, since_ms) - .expect("a generated snapshot is a valid view") -} - -/// A dense member hands over to an empty one, which keeps the generated -/// properties above from passing vacuously: the planner does return a plan for -/// a skewed snapshot, and the plan is bounded by the tick cap. -#[test] -fn a_dense_member_hands_over_to_an_empty_one() { - let planner = PlacementPlanner::default(); - let dense = observation(1, 3, 3, PlacementPressure::Normal, false); - let empty = observation(2, 0, 3, PlacementPressure::Normal, false); - let observations = [dense, empty]; - let balance = planner - .fleet_balance(NOW_MS, &observations, NOW_MS - 1) - .expect("a compact snapshot is a valid view") - .expect("a skewed snapshot elects one donor"); - assert_eq!(balance.donor, dense.session); - assert_eq!(balance.receivers, vec![empty.session]); - assert!(balance.surplus > 0 && balance.surplus <= 2); - - let demand = CellTransferDemand { - cell: CellId::from_bytes([9; 32]), - source: dense.session, - generation: 1, - memory_bytes: 1 << 20, - disk_bytes: 1 << 20, - job_credits: 1, - resident_since_ms: NOW_MS - 120_000, - last_moved_at_ms: None, - stable_observations: 3, - settled: true, - }; - let intents = planner - .plan_transfers(NOW_MS, &observations, &[demand], Some(&balance)) - .expect("a compact snapshot is a valid view"); - assert_eq!(intents.len(), 1); - assert_eq!(intents[0].source, dense.session); - assert_eq!(intents[0].destination, empty.session); - assert_eq!(intents[0].cell, demand.cell); -} - -#[test] -fn small_fleets_converge_after_owner_loss_with_fresh_settled_views() { - let planner = PlacementPlanner::default(); - let mut unconverged = Vec::new(); - for nodes in [3u8, 5, 10, 20] { - let mut observations = (0..nodes) - .map(|index| { - observation( - index, - if index == 0 { 20 } else { 0 }, - 64, - PlacementPressure::Normal, - false, - ) - }) - .collect::>(); - let mut owners = vec![session(0); 20]; - for tick in 0..40 { - let now = NOW_MS + tick * 120_000; - for value in &mut observations { - value.observed_at_ms = now; - } - let Some(balance) = planner.fleet_balance(now, &observations, now - 1).unwrap() else { - break; - }; - let demands = owners - .iter() - .enumerate() - .filter(|(_, owner)| **owner == balance.donor) - .map(|(cell, owner)| CellTransferDemand { - cell: CellId::from_bytes([cell as u8; 32]), - source: *owner, - generation: 1, - memory_bytes: 1 << 20, - disk_bytes: 1 << 20, - job_credits: 1, - resident_since_ms: now - 120_000, - last_moved_at_ms: None, - stable_observations: 3, - settled: true, - }) - .collect::>(); - let intents = planner - .plan_transfers(now, &observations, &demands, Some(&balance)) - .unwrap(); - for intent in intents { - owners[intent.cell.as_bytes()[0] as usize] = intent.destination; - observations - .iter_mut() - .find(|node| node.session == intent.source) - .unwrap() - .active_cells -= 1; - observations - .iter_mut() - .find(|node| node.session == intent.destination) - .unwrap() - .active_cells += 1; - } - } - let counts = observations - .iter() - .map(|node| node.active_cells) - .collect::>(); - if counts.iter().any(|count| { - *count < 20 / u32::from(nodes) || *count > 20u32.div_ceil(u32::from(nodes)) - }) { - unconverged.push((nodes, counts)); - } - } - assert!( - unconverged.is_empty(), - "ownership did not converge: {unconverged:?}" - ); -} - -#[test] -fn donations_do_not_overfill_a_preferred_receivers_weighted_share() { - let planner = PlacementPlanner::default(); - let mut observations = [5, 1, 0] - .into_iter() - .enumerate() - .map(|(index, count)| observation(index as u8, count, 64, PlacementPressure::Normal, false)) - .collect::>(); - // The first receiver has enough headroom advantage to rank first for both - // Cells, but neither receiver exceeds the source's sticky score. - observations[0].free_memory_bytes = 1 << 30; - observations[1].free_memory_bytes = 3 << 28; - let demands = (0..4) - .map(|index| CellTransferDemand { - cell: CellId::from_bytes([index; 32]), - source: session(0), - generation: 1, - memory_bytes: 1, - disk_bytes: 1, - job_credits: 1, - resident_since_ms: NOW_MS - 120_000, - last_moved_at_ms: None, - stable_observations: 3, - settled: true, - }) - .collect::>(); - // Its whole-Cell margin is zero at target two; the existing Cell leaves - // room for only one more donation in this snapshot. - let balance = planner - .fleet_balance(NOW_MS, &observations, NOW_MS - 1) - .unwrap() - .unwrap(); - let intents = planner - .plan_transfers(NOW_MS, &observations, &demands, Some(&balance)) - .unwrap(); - let destinations = intents - .iter() - .map(|intent| intent.destination) - .collect::>(); - assert_eq!(destinations, vec![session(1), session(2)]); -} - -proptest! { - #![proptest_config(ProptestConfig { cases: 32, ..ProptestConfig::default() })] - - /// One snapshot elects one donor and a bounded batch, whatever order the - /// observations arrive in: two hosts scanning the same directory must not - /// pick different donors for the same generation. - #[test] - fn balance_is_permutation_invariant(observations in fleet()) { - let mut reversed = observations.clone(); - reversed.reverse(); - let mut rotated = observations.clone(); - rotated.rotate_left(1); - - let expected = balance(&observations, NOW_MS - 1); - prop_assert_eq!(balance(&reversed, NOW_MS - 1), expected.clone()); - prop_assert_eq!(balance(&rotated, NOW_MS - 1), expected); - } - - /// A receiver is eligible, below the shedding tier, and never the donor. - #[test] - fn balance_never_names_an_ineligible_receiver(observations in fleet()) { - let planner = PlacementPlanner::default(); - let Some(balance) = balance(&observations, NOW_MS - 1) else { - return Ok(()); - }; - let scores = planner - .rank(CellId::from_bytes([7; 32]), NOW_MS, &observations) - .expect("a generated snapshot is a valid view"); - prop_assert!(balance.surplus > 0); - for receiver in &balance.receivers { - let observation = observations - .iter() - .find(|observation| observation.session == *receiver) - .expect("a receiver came from the snapshot"); - prop_assert_ne!(*receiver, balance.donor); - prop_assert!(observation.pressure < PlacementPressure::Shedding); - prop_assert_eq!( - scores - .iter() - .find(|score| score.session == *receiver) - .expect("every observation is ranked") - .eligibility, - PlacementEligibility::Eligible - ); - } - } - - /// A sample from the previous movement batch, or one outside the freshness - /// window, moves nothing on the count rule. - #[test] - fn balance_fails_closed_on_a_pre_batch_or_stale_view( - observations in fleet(), - age_ms in 30_001i64..600_000, - ) { - let mut stale = observations.clone(); - for observation in &mut stale { - observation.observed_at_ms = NOW_MS - age_ms; - } - prop_assert_eq!(balance(&stale, NOW_MS - age_ms - 1), None); - // The same view at the sample instant, before any batch: nothing moves. - prop_assert_eq!(balance(&observations, NOW_MS), None); - } - - /// One snapshot never plans more transfers than the tick cap, and no Cell is - /// planned twice or moved onto its own node. - #[test] - fn transfer_plan_stays_within_the_tick_cap( - observations in fleet(), - demand_bytes in prop::collection::vec(1u64 << 20..6u64 << 30, 1..5), - ) { - let planner = PlacementPlanner::default(); - let balance = balance(&observations, NOW_MS - 1); - let demands = observations - .iter() - .enumerate() - .map(|(index, observation)| { - let bytes = demand_bytes[index % demand_bytes.len()]; - CellTransferDemand { - cell: CellId::from_bytes([index as u8 + 1; 32]), - source: observation.session, - generation: 1, - memory_bytes: bytes, - disk_bytes: bytes, - job_credits: 1, - resident_since_ms: NOW_MS - 1_000, - last_moved_at_ms: None, - stable_observations: 2, - settled: true, - } - }) - .collect::>(); - let intents = planner - .plan_transfers(NOW_MS, &observations, &demands, balance.as_ref()) - .expect("a generated snapshot is a valid view"); - - prop_assert!(intents.len() <= 2); - let projected = intents - .iter() - .map(|intent| intent.disk_bytes) - .sum::(); - prop_assert!(projected <= 8 * 1024 * 1024 * 1024); - let mut cells = intents - .iter() - .map(|intent| *intent.cell.as_bytes()) - .collect::>(); - cells.sort(); - let before = cells.len(); - cells.dedup(); - prop_assert_eq!(cells.len(), before); - for intent in &intents { - prop_assert_ne!(intent.destination, intent.source); - } - } -} diff --git a/crates/crab-cell-runtime/tests/fleet/pressure.rs b/crates/crab-cell-runtime/tests/fleet/pressure.rs deleted file mode 100644 index 9fb085ec7..000000000 --- a/crates/crab-cell-runtime/tests/fleet/pressure.rs +++ /dev/null @@ -1,143 +0,0 @@ -//! Pressure shedding tests extracted from `src/pressure.rs`. -use std::sync::{Arc, Mutex}; - -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::fleet::pressure::{ - MovementBudget, PressureClassifier, PressureSample, PressureState, -}; -use crab_cell_runtime::fleet::telemetry::CellTelemetry; -use crab_cell_runtime::identity::SessionId; - -/// Records the pressure tiers one node reports to its telemetry sink. -#[derive(Default)] -struct RecordingPressureTelemetry { - states: Mutex>, -} - -impl CellTelemetry for RecordingPressureTelemetry { - fn pressure_state(&self, state: PressureState) { - self.states.lock().unwrap().push(state); - } -} - -impl RecordingPressureTelemetry { - fn observed(&self) -> Vec { - self.states.lock().unwrap().clone() - } -} - -fn sample(at_ms: i64, memory: u16) -> PressureSample { - PressureSample { - at_ms, - memory_used_permille: memory, - disk_used_permille: 100, - jobs_used_permille: 100, - stale: false, - } -} - -#[test] -fn pressure_requires_dwell_and_hysteresis() { - let mut classifier = PressureClassifier::new(800, 600, 10).unwrap(); - assert_eq!( - classifier.observe(sample(0, 850)).unwrap(), - PressureState::Normal - ); - assert_eq!( - classifier.observe(sample(9, 850)).unwrap(), - PressureState::Normal - ); - assert_eq!( - classifier.observe(sample(10, 850)).unwrap(), - PressureState::Shedding - ); - assert_eq!( - classifier.observe(sample(11, 700)).unwrap(), - PressureState::Shedding - ); - assert_eq!( - classifier.observe(sample(20, 500)).unwrap(), - PressureState::Shedding - ); - assert_eq!( - classifier.observe(sample(29, 500)).unwrap(), - PressureState::Shedding - ); - assert_eq!( - classifier.observe(sample(30, 500)).unwrap(), - PressureState::Normal - ); -} - -#[test] -fn stale_samples_cannot_recover_or_move_time_backwards() { - let mut classifier = PressureClassifier::new(800, 600, 10).unwrap(); - classifier.observe(sample(0, 850)).unwrap(); - assert_eq!( - classifier - .observe(PressureSample { - stale: true, - ..sample(10, 500) - }) - .unwrap(), - PressureState::Constrained - ); - assert!(classifier.observe(sample(9, 500)).is_err()); -} - -#[test] -fn movement_budget_bounds_concurrency_and_rate() { - let mut budget = MovementBudget::new(1, 100).unwrap(); - let mut permit = budget.try_start(0).unwrap(); - assert!(budget.try_start(0).is_err()); - budget.complete(&mut permit); - assert!(budget.try_start(0).is_err()); - let _next = budget.try_start(100).unwrap(); - assert_eq!(budget.in_flight(), 1); -} - -#[tokio::test] -async fn node_reports_every_pressure_tier_it_classifies() { - let session = SessionId::from_bytes([123; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 1_024, session).unwrap(); - let recording = Arc::new(RecordingPressureTelemetry::default()); - runtime.install_telemetry(recording.clone()).unwrap(); - - // The actor samples this node's own ledger on the wall clock, so keep the - // observations ahead of anything its tick could already have reported. - let base_ms = i64::try_from( - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap() - + 5_000; - let high = PressureSample { - at_ms: base_ms, - memory_used_permille: 900, - disk_used_permille: 100, - jobs_used_permille: 100, - stale: false, - }; - assert_eq!( - runtime.observe_pressure(high).await.unwrap(), - PressureState::Normal - ); - assert_eq!( - runtime - .observe_pressure(PressureSample { - at_ms: base_ms + 1_000, - ..high - }) - .await - .unwrap(), - PressureState::Shedding - ); - - let observed = recording.observed(); - assert!(observed.contains(&PressureState::Normal), "{observed:?}"); - assert!(observed.contains(&PressureState::Shedding), "{observed:?}"); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/fleet/pressure_properties.proptest-regressions b/crates/crab-cell-runtime/tests/fleet/pressure_properties.proptest-regressions deleted file mode 100644 index 52e908d25..000000000 --- a/crates/crab-cell-runtime/tests/fleet/pressure_properties.proptest-regressions +++ /dev/null @@ -1,7 +0,0 @@ -# Seeds for failure cases proptest has generated in the past. It is -# automatically read and these particular cases re-run before any -# novel cases are generated. -# -# It is recommended to check this file in to source control so that -# everyone who runs the test benefits from these saved cases. -cc 2b99a298de64683f7849af3a530e928d548b154a0c0e1048f0c1db941c94ccd0 # shrinks to readings = [(800, 800, 0), (800, 800, 0), (0, 0, 0)], step_ms = 1000 diff --git a/crates/crab-cell-runtime/tests/fleet/pressure_properties.rs b/crates/crab-cell-runtime/tests/fleet/pressure_properties.rs deleted file mode 100644 index 5313ec60b..000000000 --- a/crates/crab-cell-runtime/tests/fleet/pressure_properties.rs +++ /dev/null @@ -1,135 +0,0 @@ -//! Properties of the hysteretic pressure classifier that drives shedding. -//! -//! The classifier turns a node's own ledger samples into the shedding decision, -//! so its contracts are worth pinning beyond the scenario cases: a single -//! elevated sample cannot shed, critical pressure needs memory and disk, a calm -//! trace never sheds, and a stale sample cannot report a recovered node. - -use crab_cell_runtime::fleet::pressure::{PressureClassifier, PressureSample, PressureState}; -use proptest::prelude::*; - -/// The thresholds the actor builds its classifier with. -const ENTER_PERMILLE: u16 = 800; -const EXIT_PERMILLE: u16 = 600; -const DWELL_MS: i64 = 1_000; - -type Reading = (u16, u16, u16); - -fn trace() -> impl Strategy> { - prop::collection::vec((0u16..1_000, 0u16..1_000, 0u16..1_000), 1..8) -} - -fn sample(at_ms: i64, (memory, disk, jobs): Reading) -> PressureSample { - PressureSample { - at_ms, - memory_used_permille: memory, - disk_used_permille: disk, - jobs_used_permille: jobs, - stale: false, - } -} - -fn elevated((memory, disk, jobs): Reading) -> bool { - memory >= ENTER_PERMILLE || disk >= ENTER_PERMILLE || jobs >= ENTER_PERMILLE -} - -fn classifier() -> PressureClassifier { - PressureClassifier::new(ENTER_PERMILLE, EXIT_PERMILLE, DWELL_MS) - .expect("the production thresholds are valid") -} - -proptest! { - #![proptest_config(ProptestConfig { cases: 128, ..ProptestConfig::default() })] - - /// Shedding needs sustained evidence: the tier can only appear at least one - /// dwell window after the first elevated sample, so a brief spike cannot - /// evict a Cell. - #[test] - fn shedding_needs_a_dwell_after_the_first_elevated_sample( - readings in trace(), - step_ms in 1i64..2_000, - ) { - let mut classifier = classifier(); - let mut first_elevated_at = None; - for (index, reading) in readings.iter().enumerate() { - let at_ms = index as i64 * step_ms; - if elevated(*reading) && first_elevated_at.is_none() { - first_elevated_at = Some(at_ms); - } - let state = classifier.observe(sample(at_ms, *reading)).expect("time advances"); - if matches!(state, PressureState::Shedding | PressureState::Critical) { - let first = first_elevated_at.expect("shedding implies an elevated sample"); - prop_assert!(at_ms - first >= DWELL_MS, "shed at {} after {}", at_ms, first); - } - } - } - - /// Critical pressure is reserved for a node whose memory *and* disk crossed - /// the enter threshold together and then held for a dwell window. The tier - /// deliberately lags the current sample while it waits out that window, so - /// the assertion is about sustained evidence rather than the latest reading. - #[test] - fn critical_pressure_needs_sustained_memory_and_disk( - readings in trace(), - step_ms in 1i64..2_000, - ) { - let mut classifier = classifier(); - let mut first_critical_evidence = None; - for (index, reading) in readings.iter().enumerate() { - let at_ms = index as i64 * step_ms; - if reading.0 >= ENTER_PERMILLE - && reading.1 >= ENTER_PERMILLE - && first_critical_evidence.is_none() - { - first_critical_evidence = Some(at_ms); - } - let state = classifier - .observe(sample(at_ms, *reading)) - .expect("time advances"); - if state == PressureState::Critical { - let first = first_critical_evidence - .expect("critical pressure implies memory and disk were elevated"); - prop_assert!( - at_ms - first >= DWELL_MS, - "critical at {} after {}", - at_ms, - first - ); - } - } - } - - /// A trace that never crosses the enter threshold never sheds. - #[test] - fn a_calm_trace_never_sheds( - readings in prop::collection::vec( - ( - 0u16..ENTER_PERMILLE, - 0u16..ENTER_PERMILLE, - 0u16..ENTER_PERMILLE, - ), - 1..8, - ), - step_ms in 1i64..2_000, - ) { - let mut classifier = classifier(); - for (index, reading) in readings.iter().enumerate() { - let state = classifier - .observe(sample(index as i64 * step_ms, *reading)) - .expect("time advances"); - prop_assert!(!matches!(state, PressureState::Shedding | PressureState::Critical)); - } - } - - /// A stale sample can pace work but can never report a recovered node. - #[test] - fn a_stale_sample_never_reports_normal(readings in trace()) { - let mut classifier = classifier(); - for (index, reading) in readings.iter().enumerate() { - let mut stale = sample(index as i64 * DWELL_MS, *reading); - stale.stale = true; - let state = classifier.observe(stale).expect("time advances"); - prop_assert_ne!(state, PressureState::Normal); - } - } -} diff --git a/crates/crab-cell-runtime/tests/fleet/read_placement.rs b/crates/crab-cell-runtime/tests/fleet/read_placement.rs deleted file mode 100644 index a6676015f..000000000 --- a/crates/crab-cell-runtime/tests/fleet/read_placement.rs +++ /dev/null @@ -1,322 +0,0 @@ -//! Advisory reader selection from signed live node advertisements. - -use std::collections::HashSet; -use std::sync::Arc; -use std::sync::atomic::{AtomicUsize, Ordering}; - -use crab_cell_runtime::identity::{CellId, Digest, NodeId, SessionId}; -use crab_cell_runtime::node::{NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain}; -use crab_ltx::CellStorageLayout; -use crab_storage::Store; -use object_store::{ - memory::InMemory, - path::Path, - throttle::{ThrottleConfig, ThrottledStore}, -}; - -#[tokio::test(flavor = "multi_thread")] -async fn concurrent_reader_selection_shares_discovery_across_cells() { - let reads = Arc::new(AtomicUsize::new(0)); - let counted = reads.clone(); - let store = Store::new(Arc::new(ThrottledStore::new( - InMemory::new(), - ThrottleConfig { - wait_get_per_call: std::time::Duration::from_millis(5), - ..ThrottleConfig::default() - }, - ))) - .with_read_request_observer(Arc::new(move |_| { - counted.fetch_add(1, Ordering::Relaxed); - })); - let directory = NodeDirectory::new( - CellStorageLayout::new(store, Path::from("shared-reader-discovery"), [1; 16]), - Digest::from_bytes([6; 32]), - Digest::from_bytes([8; 32]), - Digest::from_bytes([9; 32]), - ); - let code = Digest::from_bytes([10; 32]); - for node in 1..=5 { - advertise( - &directory, - node, - node, - "zone-a", - code, - 32 << 20, - 1_000, - 16_000, - ) - .await; - } - reads.store(0, Ordering::Relaxed); - let selections = (1..=16).map(|cell| { - let directory = directory.clone(); - async move { - directory - .select_readers( - CellId::from_bytes([cell; 32]), - SessionId::from_bytes([1; 16]), - code, - 4, - 2_000, - 16, - ) - .await - .unwrap() - } - }); - for selected in futures_util::future::join_all(selections).await { - assert_eq!(selected.len(), 4); - assert!( - selected - .iter() - .all(|node| node.session() != SessionId::from_bytes([1; 16])) - ); - } - assert!( - reads.load(Ordering::Relaxed) <= 5, - "selection repeated discovery reads: {} provider reads", - reads.load(Ordering::Relaxed) - ); -} - -#[tokio::test] -async fn advisory_discovery_preserves_expiry_fresh_authority_and_failed_refresh() { - use bytes::Bytes; - use object_store::ObjectStoreExt; - - let inner = Arc::new(InMemory::new()); - let layout = CellStorageLayout::new( - Store::new(inner.clone()), - Path::from("reader-discovery-expiry"), - [1; 16], - ); - let directory = NodeDirectory::new( - layout.clone(), - Digest::from_bytes([6; 32]), - Digest::from_bytes([8; 32]), - Digest::from_bytes([9; 32]), - ); - let remote = NodeDirectory::new( - layout.clone(), - Digest::from_bytes([6; 32]), - Digest::from_bytes([8; 32]), - Digest::from_bytes([9; 32]), - ); - let code = Digest::from_bytes([10; 32]); - let cell = CellId::from_bytes([11; 32]); - let owner = SessionId::from_bytes([1; 16]); - advertise(&directory, 1, 1, "zone-a", code, 32 << 20, 1_000, 16_000).await; - advertise(&directory, 2, 2, "zone-b", code, 32 << 20, 1_000, 2_100).await; - assert_eq!( - directory - .select_readers(cell, owner, code, 4, 2_000, 16) - .await - .unwrap() - .len(), - 1 - ); - assert!( - directory - .select_readers(cell, owner, code, 4, 2_000, 1) - .await - .is_err() - ); - let reader = remote - .load(SessionId::from_bytes([2; 16]), 2_000) - .await - .unwrap() - .unwrap(); - remote.withdraw(&reader, 2_000).await.unwrap(); - assert!( - !directory - .is_live(reader.advertisement().session(), 2_000) - .await - .unwrap() - ); - assert_eq!(directory.live(2_000, 16).await.unwrap().len(), 1); - assert!( - directory - .select_readers(cell, owner, code, 4, 2_200, 16) - .await - .unwrap() - .is_empty() - ); - - advertise(&remote, 3, 3, "zone-c", code, 32 << 20, 2_200, 16_000).await; - let refreshed = directory - .select_readers(cell, owner, code, 4, 3_000, 16) - .await - .unwrap(); - assert_eq!(refreshed[0].session(), SessionId::from_bytes([3; 16])); - inner - .put( - &layout.node_path(&[3; 16]), - Bytes::from_static(b"corrupt signed membership").into(), - ) - .await - .unwrap(); - // Even a caller holding the same wall-clock sample cannot extend discovery - // beyond its monotonic lifetime or use it after a failed provider refresh. - tokio::time::sleep(std::time::Duration::from_millis(1_050)).await; - assert!( - directory - .select_readers(cell, owner, code, 4, 3_000, 16) - .await - .is_err() - ); -} - -async fn advertise( - directory: &NodeDirectory, - node: u8, - session: u8, - zone: &str, - code: Digest, - free_memory_bytes: u64, - issued_at_ms: i64, - expires_at_ms: i64, -) { - let advertisement = NodeAdvertisement::sign( - NodeId::from_bytes([node; 16]), - SessionId::from_bytes([session; 16]), - format!("https://node-{node}.internal:8081"), - directory.fleet(), - Digest::from_bytes([7; 32]), - Digest::from_bytes([8; 32]), - Digest::from_bytes([9; 32]), - &ed25519_dalek::SigningKey::from_bytes(&[node; 32]), - 1, - issued_at_ms, - expires_at_ms, - vec![code], - vec![1], - NodeFailureDomain::new(Some(zone.into()), Some(format!("host-{node}"))).unwrap(), - NodeCapacity { - free_memory_bytes, - free_disk_bytes: 1 << 20, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .unwrap(); - directory.create(advertisement, issued_at_ms).await.unwrap(); -} - -#[tokio::test] -async fn desired_readers_follow_live_distinct_nodes_and_replace_a_lost_member() { - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - [1; 16], - ); - let code = Digest::from_bytes([10; 32]); - let directory = NodeDirectory::new( - layout, - Digest::from_bytes([6; 32]), - Digest::from_bytes([8; 32]), - Digest::from_bytes([9; 32]), - ); - advertise(&directory, 1, 1, "zone-a", code, 32 << 20, 1_000, 16_000).await; - for (node, zone) in [(2, "zone-b"), (3, "zone-c"), (4, "zone-d"), (5, "zone-a")] { - let expires_at_ms = if node == 2 { 8_000 } else { 16_000 }; - advertise( - &directory, - node, - node, - zone, - code, - 32 << 20, - 1_000, - expires_at_ms, - ) - .await; - } - advertise( - &directory, - 7, - 7, - "zone-e", - Digest::from_bytes([11; 32]), - 32 << 20, - 1_000, - 16_000, - ) - .await; - advertise(&directory, 8, 8, "zone-f", code, 1 << 20, 1_000, 16_000).await; - let cell = CellId::from_bytes([12; 32]); - let owner = SessionId::from_bytes([1; 16]); - assert!( - directory - .select_readers(cell, owner, code, 0, 2_000, 16) - .await - .unwrap() - .is_empty() - ); - let first = directory - .select_readers(cell, owner, code, 1, 2_000, 16) - .await - .unwrap(); - assert_eq!(first.len(), 1); - assert_ne!(first[0].failure_domain().zone(), Some("zone-a")); - let two = directory - .select_readers(cell, owner, code, 2, 2_000, 16) - .await - .unwrap(); - assert_ne!( - two[0].failure_domain().zone(), - two[1].failure_domain().zone() - ); - let four = directory - .select_readers(cell, owner, code, 4, 2_000, 16) - .await - .unwrap(); - assert_eq!(four.len(), 4); - assert!( - four.iter() - .all(|candidate| candidate.node() != NodeId::from_bytes([8; 16])) - ); - assert_eq!( - four.iter() - .map(NodeAdvertisement::node) - .collect::>() - .len(), - 4 - ); - let short = directory - .select_readers(cell, owner, code, 4, 9_000, 16) - .await - .unwrap(); - assert_eq!(short.len(), 3); - advertise(&directory, 6, 6, "zone-b", code, 32 << 20, 9_000, 16_000).await; - let replacement = directory - .select_readers(cell, owner, code, 4, 9_000, 16) - .await - .unwrap(); - assert_eq!(replacement.len(), 4); - assert!( - replacement - .iter() - .any(|candidate| candidate.node() == NodeId::from_bytes([6; 16])) - ); - assert!( - replacement - .iter() - .all(|candidate| candidate.node() != NodeId::from_bytes([2; 16])) - ); - let advertisement = directory.load(owner, 9_000).await.unwrap().unwrap(); - directory.withdraw(&advertisement, 9_000).await.unwrap(); - let warm = directory - .select_readers(cell, owner, code, 4, 9_000, 16) - .await - .unwrap(); - assert_eq!( - warm.iter() - .map(NodeAdvertisement::node) - .collect::>(), - replacement - .iter() - .map(NodeAdvertisement::node) - .collect::>() - ); -} diff --git a/crates/crab-cell-runtime/tests/fleet/read_policy.rs b/crates/crab-cell-runtime/tests/fleet/read_policy.rs deleted file mode 100644 index ae6951361..000000000 --- a/crates/crab-cell-runtime/tests/fleet/read_policy.rs +++ /dev/null @@ -1,42 +0,0 @@ -//! Conditional desired-reader policy storage tests. - -use std::sync::Arc; - -use crab_cell_runtime::{CellId, identity::IncarnationId, read_policy::ReadPolicyStore}; -use crab_ltx::CellStorageLayout; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -#[tokio::test] -async fn desired_reader_updates_require_the_observed_etag() { - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - [1; 16], - ); - let store = ReadPolicyStore::new(layout); - let cell = CellId::from_bytes([2; 32]); - let incarnation = IncarnationId::from_bytes([3; 16]); - assert!(store.load(cell).await.unwrap().is_none()); - - let first = store.create(cell, incarnation, 0).await.unwrap(); - assert_eq!(first.value().desired_readers(), 0); - let second = store.update(&first, 2).await.unwrap(); - assert_eq!(second.value().revision(), 2); - let current = store.update(&second, 4).await.unwrap(); - assert_eq!(current.value().desired_readers(), 4); - assert!(store.update(&first, 1).await.is_err()); - assert_eq!( - store.load(cell).await.unwrap().unwrap().value(), - current.value() - ); - assert!(store.create(cell, incarnation, 1).await.is_err()); - let next_incarnation = IncarnationId::from_bytes([4; 16]); - let replacement = store - .replace_incarnation(¤t, next_incarnation, 1) - .await - .unwrap(); - assert_eq!(replacement.value().incarnation(), next_incarnation); - assert_eq!(replacement.value().revision(), 4); - assert!(store.update(¤t, 5).await.is_err()); -} diff --git a/crates/crab-cell-runtime/tests/fleet/takeover.rs b/crates/crab-cell-runtime/tests/fleet/takeover.rs deleted file mode 100644 index ccca7551a..000000000 --- a/crates/crab-cell-runtime/tests/fleet/takeover.rs +++ /dev/null @@ -1,305 +0,0 @@ -//! Session fencing is shared; Cell takeover remains independently authorized. - -use std::sync::Arc; -use std::sync::atomic::{AtomicUsize, Ordering}; - -use crab_cell_runtime::Error; -use crab_cell_runtime::identity::{Digest, NodeId, SessionId}; -use crab_cell_runtime::node::{ - NODE_LOG_PROTOCOL_VERSION, NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain, -}; -use crab_ltx::CellStorageLayout; -use crab_storage::Store; -use ed25519_dalek::SigningKey; -use object_store::{memory::InMemory, path::Path}; - -#[tokio::test(flavor = "multi_thread")] -async fn scheduler_membership_and_recovery_share_one_fresh_directory_scan() { - scheduler_discovery(Store::new(Arc::new(InMemory::new())), "scheduler-discovery").await; -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires a pre-created RustFS bucket, isolated prefix and test credentials"] -async fn rustfs_scheduler_membership_and_recovery_share_one_fresh_directory_scan() { - let required = |name| std::env::var(name).unwrap_or_else(|_| panic!("missing {name}")); - let store = crab_storage::build_explicit_store( - &required("CRAB_CELL_TEST_BUCKET"), - crab_storage::ObjectStoreCredentials::Aws { - access_key_id: required("AWS_ACCESS_KEY_ID"), - secret_access_key: required("AWS_SECRET_ACCESS_KEY"), - session_token: None, - region: "us-east-1".into(), - }, - Some(&required("CRAB_CELL_TEST_ENDPOINT")), - true, - ) - .unwrap(); - scheduler_discovery(store, &required("CRAB_CELL_TEST_PREFIX")).await; -} - -async fn scheduler_discovery(store: Store, prefix: &str) { - for nodes in [3_u8, 5, 10, 20] { - let reads = Arc::new(AtomicUsize::new(0)); - let counted = Arc::clone(&reads); - let store = store.clone().with_read_request_observer(Arc::new(move |_| { - counted.fetch_add(1, Ordering::Relaxed); - })); - let directory = NodeDirectory::new( - CellStorageLayout::new(store, Path::from(format!("{prefix}/{nodes}")), [1; 16]), - Digest::from_bytes([2; 32]), - Digest::from_bytes([3; 32]), - Digest::from_bytes([4; 32]), - ); - for index in 1..=nodes { - directory - .create( - advertisement(SessionId::from_bytes([index; 16]), 2_000), - 2_000, - ) - .await - .unwrap(); - let retired = directory - .create( - advertisement(SessionId::from_bytes([index + nodes; 16]), 1_000), - 1_000, - ) - .await - .unwrap(); - directory.withdraw(&retired, 2_000).await.unwrap(); - } - let failed = SessionId::from_bytes([2 * nodes + 1; 16]); - let enrolled = directory - .create(advertisement(failed, 1_000), 1_000) - .await - .unwrap(); - let enrolled = directory - .recruit_log(&enrolled, 1, 1, usize::from(nodes) + 1, 2_001) - .await - .unwrap(); - directory.activate_log(&enrolled, 2_002).await.unwrap(); - - reads.store(0, Ordering::Relaxed); - let live = directory - .live_for_recovery(11_000, usize::from(nodes)) - .await - .unwrap(); - assert_eq!(live.len(), usize::from(nodes)); - let candidates = directory - .clone() - .recovery_candidates(SessionId::from_bytes([1; 16]), 11_000, 16) - .await - .unwrap(); - assert_eq!(candidates, [failed]); - // Every live/retired/failed record is read once. The extra GET is the - // fresh claimant admission check, which discovery must never replace. - assert_eq!(reads.load(Ordering::Relaxed), usize::from(nodes) * 2 + 2); - - let claimant = SessionId::from_bytes([1; 16]); - let observed = directory.load(claimant, 11_000).await.unwrap().unwrap(); - directory.withdraw(&observed, 11_000).await.unwrap(); - assert!( - directory - .recovery_candidates(claimant, 11_000, 16) - .await - .is_err() - ); - assert!( - directory - .claim_expired_for_recovery(failed, claimant, 11_000) - .await - .is_err() - ); - } -} - -#[tokio::test] -async fn recovery_membership_refresh_is_fresh_and_drops_failed_observations() { - use bytes::Bytes; - use object_store::ObjectStoreExt; - - let inner = Arc::new(InMemory::new()); - let layout = - CellStorageLayout::new(Store::new(inner.clone()), Path::from("fresh-scan"), [1; 16]); - let directory = NodeDirectory::new( - layout.clone(), - Digest::from_bytes([2; 32]), - Digest::from_bytes([3; 32]), - Digest::from_bytes([4; 32]), - ); - let claimant = SessionId::from_bytes([1; 16]); - let other = SessionId::from_bytes([2; 16]); - directory - .create(advertisement(claimant, 1_000), 1_000) - .await - .unwrap(); - assert_eq!( - directory.live_for_recovery(2_000, 2).await.unwrap().len(), - 1 - ); - let remote = NodeDirectory::new( - layout.clone(), - directory.fleet(), - Digest::from_bytes([3; 32]), - Digest::from_bytes([4; 32]), - ); - remote - .create(advertisement(other, 1_000), 1_000) - .await - .unwrap(); - assert_eq!( - directory.live_for_recovery(2_000, 2).await.unwrap().len(), - 2 - ); - assert!(directory.live_for_recovery(2_000, 1).await.is_err()); - directory.live_for_recovery(2_000, 2).await.unwrap(); - inner - .put( - &layout.node_path(other.as_bytes()), - Bytes::from_static(b"invalid node").into(), - ) - .await - .unwrap(); - assert!(directory.live_for_recovery(2_000, 2).await.is_err()); - assert!( - directory - .recovery_candidates(claimant, 2_000, 2) - .await - .is_err() - ); -} - -#[tokio::test] -async fn independent_takeovers_share_a_fence_only_without_an_active_log() { - for log_state in ["absent", "inactive", "active"] { - let directory = NodeDirectory::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from(log_state), - [1; 16], - ), - Digest::from_bytes([2; 32]), - Digest::from_bytes([3; 32]), - Digest::from_bytes([4; 32]), - ); - let leader = SessionId::from_bytes([1; 16]); - let first = SessionId::from_bytes([2; 16]); - let second = SessionId::from_bytes([3; 16]); - let mut enrollment = directory - .create(advertisement(leader, 1_000), 1_000) - .await - .unwrap(); - for session in [first, second] { - directory - .create(advertisement(session, 2_000), 2_000) - .await - .unwrap(); - } - if log_state != "absent" { - enrollment = directory - .recruit_log(&enrollment, 1, 1, 3, 2_001) - .await - .unwrap(); - if log_state == "active" { - enrollment = directory.activate_log(&enrollment, 2_002).await.unwrap(); - } - } - let now_ms = 11_000; - // Both request paths observed the expired advertisement before either - // fenced it. The later claim must resolve the other caller's CAS. - for claimant in [first, second] { - assert!( - directory - .takeover_proof(leader, claimant, now_ms) - .await - .unwrap() - .is_none() - ); - } - if log_state == "active" { - let fenced = directory - .claim_expired_for_recovery(leader, first, now_ms) - .await - .unwrap(); - assert!(matches!( - fenced.direct_takeover(), - Err(Error::PendingPublication) - )); - for claimant in [first, second] { - assert!( - directory - .takeover_proof(leader, claimant, now_ms) - .await - .unwrap() - .is_none() - ); - assert!(matches!( - directory - .claim_expired_for_takeover(leader, claimant, now_ms) - .await, - Err(Error::PendingPublication) - )); - } - assert!( - directory - .claim_expired_for_recovery(leader, second, now_ms) - .await - .is_err() - ); - } else { - for claimant in [first, second] { - let proof = directory - .claim_expired_for_takeover(leader, claimant, now_ms) - .await - .unwrap_or_else(|error| panic!("{log_state}: {claimant:?}: {error}")); - assert_eq!((proof.session(), proof.claimant()), (leader, claimant)); - assert_eq!( - directory - .takeover_proof(leader, claimant, now_ms) - .await - .unwrap(), - Some(proof) - ); - } - } - // Sharing proof neither revives the old session nor admits an expired - // successor, and an old enrollment cannot activate a log after fencing. - assert!(!directory.is_live(leader, now_ms).await.unwrap()); - if log_state == "inactive" { - assert!(directory.activate_log(&enrollment, 2_003).await.is_err()); - } - assert!( - directory - .takeover_proof(leader, second, 12_000) - .await - .is_err() - ); - } -} - -fn advertisement(session: SessionId, issued_at_ms: i64) -> NodeAdvertisement { - NodeAdvertisement::sign( - NodeId::from_bytes(*session.as_bytes()), - session, - "https://node.internal:8789".into(), - Digest::from_bytes([2; 32]), - Digest::from_bytes([5; 32]), - Digest::from_bytes([3; 32]), - Digest::from_bytes([4; 32]), - &SigningKey::from_bytes(&[7; 32]), - 1, - issued_at_ms, - issued_at_ms + 10_000, - vec![Digest::from_bytes([6; 32])], - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 1_024, - free_disk_bytes: 1_024, - follower_free_bytes: 1_024, - job_credits: 1, - log_protocol: NODE_LOG_PROTOCOL_VERSION, - ..NodeCapacity::default() - }, - ) - .unwrap() -} diff --git a/crates/crab-cell-runtime/tests/primitives.rs b/crates/crab-cell-runtime/tests/primitives.rs deleted file mode 100644 index dc7aa5acb..000000000 --- a/crates/crab-cell-runtime/tests/primitives.rs +++ /dev/null @@ -1,14 +0,0 @@ -//! Distributed primitive integration tests: SQL, KV, Blob, Cron, Queue, Workflow. - -mod support; - -mod primitives { - pub mod blob_cron; - pub mod cron_api; - pub mod kv; - pub mod queue; - pub mod sql; - pub mod workflow; - pub mod workflow_activity_codec; - pub mod workflow_api; -} diff --git a/crates/crab-cell-runtime/tests/primitives/blob_cron.rs b/crates/crab-cell-runtime/tests/primitives/blob_cron.rs deleted file mode 100644 index badd89e3d..000000000 --- a/crates/crab-cell-runtime/tests/primitives/blob_cron.rs +++ /dev/null @@ -1,595 +0,0 @@ -use std::sync::Arc; - -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::catalog::{CatalogEntry, CellCatalog}; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::client::CellClient; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::primitives::blob::{ - BlobCondition, BlobMutation, BlobMutationOutcome, BlobQuery, BlobQueryResult, register_blob, -}; -use crab_cell_runtime::primitives::blob::{BlobModule, BlobNamespace}; -use crab_cell_runtime::primitives::cron::{ - CronInvocation, CronMutation, CronQueryResult, CronTarget, register_cron, -}; -use crab_cell_runtime::primitives::cron::{CronModule, CronNamespace}; -use crab_cell_runtime::primitives::maintenance::{ - MaintenanceModule, MaintenanceTickOutcome, MaintenanceTickRequest, -}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, Command, ModuleDescriptor, NamespaceDescriptor, RegistryBuilder, -}; -use crab_cell_runtime::registry::{ - CommandContext, CommandResult, MigrationDescriptor, OperationDescriptor, -}; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use crate::support::fixtures::{mutation_identity_window, now_ms}; - -const BLOB_MODULE: &str = "blob-test"; -const BLOB_NAMESPACE: NamespaceId = NamespaceId::from_bytes([1; 16]); -const BLOB_MIGRATION: &str = include_str!("../../src/migrations/blob.sql"); -const CRON_MODULE: &str = "cron-test"; -const CRON_NAMESPACE: NamespaceId = NamespaceId::from_bytes([2; 16]); -const CRON_MIGRATION: &str = include_str!("../../src/migrations/cron.sql"); -const TARGET_MODULE: &str = "cron-target-test"; -const TARGET_NAMESPACE: NamespaceId = NamespaceId::from_bytes([3; 16]); -const TARGET_MIGRATION: &str = "CREATE TABLE cron_target(value BLOB) STRICT;"; -const TARGET_INPUT_LIMIT: u32 = 300 * 1024; -const CRON_TARGETS: &[CronTarget] = &[CronTarget::new( - TARGET_MODULE, - TARGET_NAMESPACE, - 9, - 1, - TARGET_INPUT_LIMIT, -)]; - -const BLOB_COMMANDS: &[OperationDescriptor] = &[operation(1, 300 * 1024, 64), operation(2, 8, 5)]; -const BLOB_QUERIES: &[OperationDescriptor] = &[operation(1, 4 * 1024, 600 * 1024)]; -const CRON_COMMANDS: &[OperationDescriptor] = &[operation(1, 300 * 1024, 16), operation(2, 8, 5)]; -const CRON_QUERIES: &[OperationDescriptor] = &[operation(1, 64, 600 * 1024)]; -const TARGET_COMMANDS: &[OperationDescriptor] = &[operation(9, TARGET_INPUT_LIMIT, 1)]; - -struct TestBlob; - -impl MaintenanceModule for TestBlob { - const MODULE: &'static str = BLOB_MODULE; - const TICK_COMMAND_ID: u32 = 2; -} - -impl BlobModule for TestBlob { - const NAMESPACE: NamespaceId = BLOB_NAMESPACE; - const MUTATE_COMMAND_ID: u32 = 1; - const QUERY_ID: u32 = 1; -} - -impl CellModule for TestBlob { - const NAME: &'static str = BLOB_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - descriptor( - BLOB_MODULE, - BLOB_NAMESPACE, - CatalogRole::Blob, - BLOB_MIGRATION, - BLOB_COMMANDS, - BLOB_QUERIES, - &[], - 10, - ) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - register_blob::(registry) - } -} - -struct TestCron; - -impl MaintenanceModule for TestCron { - const MODULE: &'static str = CRON_MODULE; - const TICK_COMMAND_ID: u32 = 2; - const CRON_TARGETS: &'static [CronTarget] = CRON_TARGETS; -} - -impl CronModule for TestCron { - const NAMESPACE: NamespaceId = CRON_NAMESPACE; - const MUTATE_COMMAND_ID: u32 = 1; - const QUERY_ID: u32 = 1; -} - -impl CellModule for TestCron { - const NAME: &'static str = CRON_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - descriptor( - CRON_MODULE, - CRON_NAMESPACE, - CatalogRole::Cron, - CRON_MIGRATION, - CRON_COMMANDS, - CRON_QUERIES, - &[TARGET_NAMESPACE], - 11, - ) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - register_cron::(registry) - } -} - -struct CronTargetModule; - -struct ReceiveCron; - -impl Command for ReceiveCron { - const MODULE: &'static str = TARGET_MODULE; - const ID: u32 = 9; - const CODEC_VERSION: u32 = 1; - type Input = CronInvocation; - type Output = (); - - fn execute( - _: &mut CommandContext<'_, '_>, - _: Self::Input, - ) -> crab_cell_runtime::Result> { - Ok(CommandResult::Success(())) - } -} - -impl CellModule for CronTargetModule { - const NAME: &'static str = TARGET_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - descriptor( - TARGET_MODULE, - TARGET_NAMESPACE, - CatalogRole::Repository, - TARGET_MIGRATION, - TARGET_COMMANDS, - &[], - &[], - 12, - ) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - registry.bind_command::() - } -} - -#[test] -fn blob_and_cron_bindings_match_release_descriptors() { - let mut registry = RegistryBuilder::new(BuildDescriptor { - source_revision: "blob-cron-test".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }); - registry.register(CronTargetModule).unwrap(); - registry.register(TestBlob).unwrap(); - registry.register(TestCron).unwrap(); - registry.finish().unwrap(); -} - -#[tokio::test] -async fn typed_blob_and_cron_recover_after_owner_loss() { - let registry = registry(); - let tenant = TenantId::from_bytes([20; 16]); - let application = ApplicationId::from_bytes([21; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new( - store.clone(), - Path::from("blob-cron-runtime"), - *application.as_bytes(), - ); - - let blob_target = - CellTarget::new(tenant, application, BLOB_NAMESPACE, &0_u32.to_be_bytes()).unwrap(); - let blob_incarnation = IncarnationId::from_bytes([23; 16]); - let blob_replica = CellReplica::new( - layout.clone(), - *blob_target.cell_id().as_bytes(), - *blob_incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), tenant); - let blob_proof = catalog - .provision( - CatalogEntry::new( - &blob_target, - CatalogRole::Blob, - registry.module_code(BLOB_MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let blob_session = SessionId::from_bytes([24; 16]); - let blob_control = authority - .create_initial( - &blob_proof, - blob_incarnation, - Owner { - session: blob_session, - endpoint: "https://blob.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let blob_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - blob_session, - ) - .unwrap(); - let blob_handle = blob_runtime - .bootstrap( - blob_proof, - blob_replica, - authority.clone(), - blob_control, - directory.path().join("blob.sqlite"), - crab_cell_runtime::primitives::blob::install_blob_schema, - ) - .await - .unwrap(); - let blobs = BlobNamespace::::new( - CellClient::local(registry.clone(), blob_handle.clone()) - .with_blob_artifact_store(crab_cell_runtime::BlobArtifactStore::new(store.clone())), - tenant, - application, - ) - .unwrap(); - let key = b"artifacts/result".to_vec(); - let upload_id = [25; 16]; - let start_now_ms = now_ms(); - blobs - .mutate( - mutation_identity_window(26, start_now_ms, start_now_ms + 60_000), - BlobMutation::Begin { - key: key.clone(), - upload_id, - condition: BlobCondition::Missing, - content_type: None, - metadata: Vec::new(), - expires_at_ms: start_now_ms + 60_000, - }, - ) - .await - .unwrap(); - blobs - .mutate( - mutation_identity_window(27, start_now_ms, start_now_ms + 60_000), - BlobMutation::PutPart { - key: key.clone(), - upload_id, - part_number: 1, - payload: b"published".to_vec(), - }, - ) - .await - .unwrap(); - let committed = blobs - .mutate( - mutation_identity_window(28, start_now_ms, start_now_ms + 60_000), - BlobMutation::Complete { - key: key.clone(), - upload_id, - part_count: 1, - }, - ) - .await - .unwrap(); - assert!(matches!( - committed.output, - BlobMutationOutcome::Committed { size: 9, .. } - )); - let read = blobs - .query( - BlobQuery::Read { - key, - offset: 0, - limit: 32, - }, - Some(committed.receipt), - ) - .await - .unwrap(); - assert!(matches!( - read.output, - BlobQueryResult::Read(Some(ref value)) if value.bytes == b"published" - )); - drop(blobs); - drop(blob_handle); - drop(blob_runtime); - let stale_blob = authority - .load(blob_target.cell_id()) - .await - .unwrap() - .unwrap(); - let blob_successor = SessionId::from_bytes([34; 16]); - let blob_takeover = - crate::support::fencing::fence_session(&layout, blob_session, blob_successor).await; - let blob_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - blob_successor, - ) - .unwrap(); - let blob_restored = blob_runtime - .takeover_restored( - catalog - .lookup(blob_target.cell_id()) - .await - .unwrap() - .unwrap(), - CellReplica::new( - layout.clone(), - *blob_target.cell_id().as_bytes(), - *blob_incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(), - authority.clone(), - stale_blob, - blob_takeover.direct_takeover().unwrap(), - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - layout.clone(), - Limits::default(), - ), - directory.path().join("blob-takeover.sqlite"), - Owner { - session: blob_successor, - endpoint: "https://blob-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - let restored_blobs = BlobNamespace::::new( - CellClient::local(registry.clone(), blob_restored.clone()) - .with_blob_artifact_store(crab_cell_runtime::BlobArtifactStore::new(store)), - tenant, - application, - ) - .unwrap(); - assert!(matches!( - restored_blobs - .query( - BlobQuery::Read { - key: b"artifacts/result".to_vec(), - offset: 0, - limit: 32, - }, - Some(committed.receipt), - ) - .await - .unwrap() - .output, - BlobQueryResult::Read(Some(ref value)) if value.bytes == b"published" - )); - drop(restored_blobs); - blob_restored.drain().await.unwrap(); - blob_runtime.shutdown().await.unwrap(); - - let cron_target = - CellTarget::new(tenant, application, CRON_NAMESPACE, &0_u32.to_be_bytes()).unwrap(); - let cron_incarnation = IncarnationId::from_bytes([29; 16]); - let cron_replica = CellReplica::new( - layout.clone(), - *cron_target.cell_id().as_bytes(), - *cron_incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let cron_proof = catalog - .provision( - CatalogEntry::new( - &cron_target, - CatalogRole::Cron, - registry.module_code(CRON_MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let cron_session = SessionId::from_bytes([30; 16]); - let cron_control = authority - .create_initial( - &cron_proof, - cron_incarnation, - Owner { - session: cron_session, - endpoint: "https://cron.internal:8081".into(), - }, - ) - .await - .unwrap(); - let cron_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - cron_session, - ) - .unwrap(); - let cron_handle = cron_runtime - .bootstrap( - cron_proof, - cron_replica, - authority.clone(), - cron_control, - directory.path().join("cron.sqlite"), - crab_cell_runtime::primitives::cron::install_cron_schema, - ) - .await - .unwrap(); - let cron_client = CellClient::local(registry.clone(), cron_handle.clone()); - let cron = CronNamespace::::new(cron_client.clone(), tenant, application).unwrap(); - let schedule_id = [31; 16]; - let cron_start_now_ms = now_ms(); - let scheduled = cron - .mutate( - mutation_identity_window(32, cron_start_now_ms, cron_start_now_ms + 60_000), - CronMutation::Upsert { - schedule_id, - target_index: 0, - target_partition: b"destination".to_vec(), - payload: b"run".to_vec(), - interval_ms: 1_000, - next_due_ms: cron_start_now_ms + 100, - }, - ) - .await - .unwrap(); - tokio::time::sleep(std::time::Duration::from_millis(150)).await; - let tick_now_ms = now_ms(); - let tick = registry - .run_maintenance_once( - cron_client, - cron_target.clone(), - mutation_identity_window(33, tick_now_ms, tick_now_ms + 60_000), - MaintenanceTickRequest { - expected_commit_sequence: scheduled.receipt.commit_sequence, - }, - ) - .await - .unwrap(); - assert_eq!( - tick.output, - MaintenanceTickOutcome::Applied { processed: 1 } - ); - let state = cron.get(schedule_id, Some(tick.receipt)).await.unwrap(); - assert!(matches!( - state.output, - CronQueryResult::Get(Some(ref schedule)) if schedule.occurrence == 1 - )); - drop(cron); - drop(cron_handle); - drop(cron_runtime); - let stale_cron = authority - .load(cron_target.cell_id()) - .await - .unwrap() - .unwrap(); - let cron_successor = SessionId::from_bytes([35; 16]); - let cron_takeover = - crate::support::fencing::fence_session(&layout, cron_session, cron_successor).await; - let cron_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - cron_successor, - ) - .unwrap(); - let cron_restored = cron_runtime - .takeover_restored( - catalog - .lookup(cron_target.cell_id()) - .await - .unwrap() - .unwrap(), - CellReplica::new( - layout.clone(), - *cron_target.cell_id().as_bytes(), - *cron_incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(), - authority.clone(), - stale_cron, - cron_takeover.direct_takeover().unwrap(), - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - layout.clone(), - Limits::default(), - ), - directory.path().join("cron-takeover.sqlite"), - Owner { - session: cron_successor, - endpoint: "https://cron-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - let restored_cron = CronNamespace::::new( - CellClient::local(registry, cron_restored.clone()), - tenant, - application, - ) - .unwrap(); - assert!(matches!( - restored_cron.get(schedule_id, Some(tick.receipt)).await.unwrap().output, - CronQueryResult::Get(Some(ref schedule)) if schedule.occurrence == 1 - )); - drop(restored_cron); - cron_restored.drain().await.unwrap(); - cron_runtime.shutdown().await.unwrap(); -} - -fn registry() -> Arc { - let mut registry = RegistryBuilder::new(BuildDescriptor { - source_revision: "blob-cron-test".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }); - registry.register(CronTargetModule).unwrap(); - registry.register(TestBlob).unwrap(); - registry.register(TestCron).unwrap(); - Arc::new(registry.finish().unwrap()) -} - -fn descriptor( - name: &'static str, - namespace: NamespaceId, - role: CatalogRole, - migration: &'static str, - commands: &'static [OperationDescriptor], - queries: &'static [OperationDescriptor], - effect_targets: &'static [NamespaceId], - digest: u8, -) -> &'static ModuleDescriptor { - Box::leak(Box::new(ModuleDescriptor { - name, - source_digest: Digest::from_bytes([digest; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: migration, - digest: Digest::from_bytes(*blake3::hash(migration.as_bytes()).as_bytes()), - }])), - commands, - queries, - workflow_definitions: &[], - activity_types: &[], - namespaces: Box::leak(Box::new([NamespaceDescriptor { - id: namespace, - name, - role, - shards: 1, - effect_targets, - dead_letter: None, - }])), - })) -} - -const fn operation(id: u32, input_limit: u32, output_limit: u32) -> OperationDescriptor { - OperationDescriptor { - id, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit, - output_limit, - } -} diff --git a/crates/crab-cell-runtime/tests/primitives/cron_api.rs b/crates/crab-cell-runtime/tests/primitives/cron_api.rs deleted file mode 100644 index d59b342c4..000000000 --- a/crates/crab-cell-runtime/tests/primitives/cron_api.rs +++ /dev/null @@ -1,27 +0,0 @@ -use crab_cell_runtime::primitives::cron::CronInvocation; -use crab_cell_runtime::primitives::cron::{CronMutation, CronQuery}; - -use crate::support::fixtures::codec_roundtrip; - -#[test] -fn cron_codecs_roundtrip_schedule_and_invocation() { - codec_roundtrip(CronMutation::Upsert { - schedule_id: [1; 16], - target_index: 2, - target_partition: b"shard".to_vec(), - payload: b"run".to_vec(), - interval_ms: 1_000, - next_due_ms: 10, - }); - codec_roundtrip(CronInvocation { - schedule_id: [1; 16], - generation: 2, - occurrence: 3, - scheduled_at_ms: 10, - payload: b"run".to_vec(), - }); - codec_roundtrip(CronQuery::List { - after: Some([2; 16]), - limit: 128, - }); -} diff --git a/crates/crab-cell-runtime/tests/primitives/kv.rs b/crates/crab-cell-runtime/tests/primitives/kv.rs deleted file mode 100644 index d2ed335fe..000000000 --- a/crates/crab-cell-runtime/tests/primitives/kv.rs +++ /dev/null @@ -1,632 +0,0 @@ -use std::sync::Arc; - -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::catalog::{CatalogEntry, CellCatalog}; -use crab_cell_runtime::cell::schema::install_runtime_schema; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::client::{CellClient, InvocationError}; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::primitives::kv::{ - KvAtomicOutcome, KvAtomicRequest, KvCheck, KvCondition, KvListRequest, KvMutation, - install_kv_schema, kv_atomic, kv_cleanup_expired, kv_get, kv_list, register_kv, -}; -use crab_cell_runtime::primitives::kv::{KvModule, KvNamespace}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, ModuleDescriptor, NamespaceDescriptor, RegistryBuilder, -}; -use crab_cell_runtime::registry::{MigrationDescriptor, OperationDescriptor}; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use crate::support::fixtures::mutation_identity; - -const KV_MODULE: &str = "kv-test"; -const KV_NAMESPACE: NamespaceId = NamespaceId::from_bytes([6; 16]); -const KV_MIGRATION: &str = include_str!("../../src/migrations/kv.sql"); - -struct TestKv; - -impl KvModule for TestKv { - const MODULE: &'static str = KV_MODULE; - const ATOMIC_COMMAND_ID: u32 = 1; - const GET_QUERY_ID: u32 = 1; - const LIST_QUERY_ID: u32 = 2; -} - -impl CellModule for TestKv { - const NAME: &'static str = KV_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: std::sync::OnceLock = std::sync::OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: KV_MODULE, - source_digest: Digest::from_bytes([4; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: KV_MIGRATION, - digest: Digest::from_bytes(*blake3::hash(KV_MIGRATION.as_bytes()).as_bytes()), - }])), - commands: &[OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: (4 * 1024 * 1024 + 64 * 1024), - output_limit: 1024 * 1024, - }], - queries: &[ - OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 4096, - output_limit: (4 * 1024 * 1024 + 64 * 1024), - }, - OperationDescriptor { - id: 2, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 4096, - output_limit: (4 * 1024 * 1024 + 64 * 1024), - }, - ], - workflow_definitions: &[], - activity_types: &[], - namespaces: &[NamespaceDescriptor { - id: KV_NAMESPACE, - name: KV_MODULE, - role: CatalogRole::Kv, - shards: 1, - effect_targets: &[], - dead_letter: None, - }], - }) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - register_kv::(registry) - } -} - -fn kv_registry() -> Arc { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "kv-api-test".into(), - cargo_lock_digest: Digest::from_bytes([5; 32]), - }); - builder.register(TestKv).unwrap(); - Arc::new(builder.finish().unwrap()) -} - -fn connection() -> crab_ltx::rusqlite::Connection { - let mut connection = crab_ltx::rusqlite::Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - crab_cell_runtime::CellId::from_bytes([1; 32]), - IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - let transaction = connection.transaction().unwrap(); - install_kv_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - connection -} - -fn put(key: &[u8], value: &[u8], expires_at_ms: Option) -> KvMutation { - KvMutation::Put { - key: key.to_vec(), - value: value.to_vec(), - expires_at_ms, - } -} - -#[test] -fn atomic_checks_precede_ordered_mutations_and_versions_never_repeat() { - let mut connection = connection(); - let first = KvAtomicRequest { - scope: b"repo".to_vec(), - checks: vec![KvCheck { - key: b"key".to_vec(), - condition: KvCondition::Absent, - }], - mutations: vec![put(b"key", b"one", None)], - }; - let transaction = connection.transaction().unwrap(); - let outcome = kv_atomic(&transaction, 10, &first).unwrap(); - let KvAtomicOutcome::Applied(results) = outcome else { - panic!("expected applied KV mutation"); - }; - let first_version = results[0].version.unwrap(); - transaction - .execute( - "UPDATE sys_meta SET commit_sequence = 1 WHERE singleton = 1", - [], - ) - .unwrap(); - transaction.commit().unwrap(); - assert_eq!(&first_version[..16], &[2; 16]); - assert_eq!(&first_version[16..24], &1_u64.to_be_bytes()); - assert_eq!(&first_version[24..], &0_u32.to_be_bytes()); - - let failed = KvAtomicRequest { - scope: b"repo".to_vec(), - checks: vec![KvCheck { - key: b"key".to_vec(), - condition: KvCondition::Absent, - }], - mutations: vec![put(b"key", b"wrong", None)], - }; - let transaction = connection.transaction().unwrap(); - assert_eq!( - kv_atomic(&transaction, 11, &failed).unwrap(), - KvAtomicOutcome::PreconditionFailed { - key: b"key".to_vec() - } - ); - transaction.rollback().unwrap(); - assert_eq!( - kv_get(&connection, b"repo", b"key", 11) - .unwrap() - .unwrap() - .value, - b"one" - ); - - let replace = KvAtomicRequest { - scope: b"repo".to_vec(), - checks: vec![KvCheck { - key: b"key".to_vec(), - condition: KvCondition::Version(first_version), - }], - mutations: vec![KvMutation::Delete { - key: b"key".to_vec(), - }], - }; - let transaction = connection.transaction().unwrap(); - assert!(matches!( - kv_atomic(&transaction, 12, &replace).unwrap(), - KvAtomicOutcome::Applied(_) - )); - transaction.commit().unwrap(); - assert!(kv_get(&connection, b"repo", b"key", 12).unwrap().is_none()); -} - -#[test] -fn ttl_cleanup_and_binary_prefix_pagination_are_bounded() { - let mut connection = connection(); - let request = KvAtomicRequest { - scope: b"scope".to_vec(), - checks: Vec::new(), - mutations: vec![ - put(&[0x10], b"a", Some(50)), - put(&[0x10, 0xff], b"b", None), - put(&[0x11], b"c", None), - ], - }; - let transaction = connection.transaction().unwrap(); - kv_atomic(&transaction, 10, &request).unwrap(); - transaction.commit().unwrap(); - - let first = kv_list(&connection, b"scope", &[0x10], None, 1, 20).unwrap(); - assert_eq!(first.entries[0].key, vec![0x10]); - assert_eq!(first.next_after, Some(vec![0x10])); - let second = kv_list( - &connection, - b"scope", - &[0x10], - first.next_after.as_deref(), - 1, - 20, - ) - .unwrap(); - assert_eq!(second.entries[0].key, vec![0x10, 0xff]); - assert!(second.next_after.is_none()); - assert!( - kv_get(&connection, b"scope", &[0x10], 50) - .unwrap() - .is_none() - ); - - let transaction = connection.transaction().unwrap(); - assert_eq!(kv_cleanup_expired(&transaction, 50).unwrap(), 1); - transaction.commit().unwrap(); - assert_eq!( - connection - .query_row("SELECT count(*) FROM kv_entries", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 2 - ); - - let transaction = connection.transaction().unwrap(); - for key in 1_u8..=20 { - transaction - .execute( - "INSERT INTO kv_entries(scope, key, version, value, expires_at_ms) VALUES (?1, ?2, ?3, ?4, NULL)", - ( - b"large".as_slice(), - [key].as_slice(), - [0; 28].as_slice(), - vec![key; 60_000], - ), - ) - .unwrap(); - } - transaction.commit().unwrap(); - let bounded = kv_list(&connection, b"large", &[], None, 1_000, 20).unwrap(); - assert!(bounded.entries.len() < 20); - assert!(bounded.next_after.is_some()); -} - -#[test] -fn duplicate_mutation_keys_and_expired_puts_fail_before_writes() { - let mut connection = connection(); - let duplicate = KvAtomicRequest { - scope: Vec::new(), - checks: Vec::new(), - mutations: vec![put(b"key", b"one", None), put(b"key", b"two", None)], - }; - let transaction = connection.transaction().unwrap(); - assert!(kv_atomic(&transaction, 10, &duplicate).is_err()); - transaction.rollback().unwrap(); - - let expired = KvAtomicRequest { - scope: Vec::new(), - checks: Vec::new(), - mutations: vec![put(b"key", b"one", Some(10))], - }; - let transaction = connection.transaction().unwrap(); - assert!(kv_atomic(&transaction, 10, &expired).is_err()); - transaction.rollback().unwrap(); - - let oversized = KvAtomicRequest { - scope: Vec::new(), - checks: Vec::new(), - mutations: vec![put(b"key", &vec![7; 4 * 1024 * 1024 + 1], None)], - }; - let transaction = connection.transaction().unwrap(); - assert!(kv_atomic(&transaction, 10, &oversized).is_err()); - transaction.rollback().unwrap(); - - let too_large_batch = KvAtomicRequest { - scope: Vec::new(), - checks: Vec::new(), - mutations: vec![ - put(b"a", &vec![1; 2 * 1024 * 1024 + 64 * 1024], None), - put(b"b", &vec![2; 2 * 1024 * 1024 + 64 * 1024], None), - ], - }; - let transaction = connection.transaction().unwrap(); - assert!(kv_atomic(&transaction, 10, &too_large_batch).is_err()); - transaction.rollback().unwrap(); - assert_eq!( - connection - .query_row("SELECT count(*) FROM kv_entries", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 0 - ); -} - -#[test] -fn maximum_value_get_and_list_progress_under_page_budget() { - let mut connection = connection(); - let value = vec![7; 4 * 1024 * 1024]; - let transaction = connection.transaction().unwrap(); - kv_atomic( - &transaction, - 10, - &KvAtomicRequest { - scope: b"scope".to_vec(), - checks: Vec::new(), - mutations: vec![put(b"a", &value, None)], - }, - ) - .unwrap(); - transaction.commit().unwrap(); - - let transaction = connection.transaction().unwrap(); - kv_atomic( - &transaction, - 11, - &KvAtomicRequest { - scope: b"scope".to_vec(), - checks: Vec::new(), - mutations: vec![put(b"b", &vec![8; 128 * 1024], None)], - }, - ) - .unwrap(); - transaction.commit().unwrap(); - - assert_eq!( - kv_get(&connection, b"scope", b"a", 12) - .unwrap() - .unwrap() - .value, - value - ); - let first = kv_list(&connection, b"scope", b"", None, 10, 12).unwrap(); - assert_eq!(first.entries.len(), 1); - assert_eq!(first.next_after.as_deref(), Some(b"a".as_slice())); - let second = kv_list(&connection, b"scope", b"", Some(b"a"), 10, 12).unwrap(); - assert_eq!(second.entries.len(), 1); - assert_eq!(second.entries[0].key, b"b"); - assert!(second.next_after.is_none()); -} - -#[tokio::test] -async fn typed_kv_namespace_recovers_after_owner_loss() { - let registry = kv_registry(); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([3; 16]), - KV_NAMESPACE, - &0_u32.to_be_bytes(), - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([2; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store, Path::from("runtime"), [3; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Kv, - registry.module_code(KV_MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let first_session = SessionId::from_bytes([4; 16]); - let observed = authority - .create_initial( - &proof, - incarnation, - Owner { - session: first_session, - endpoint: "https://first.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - first_session, - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - observed, - directory.path().join("first.sqlite"), - install_kv_schema, - ) - .await - .unwrap(); - let namespace = KvNamespace::::new( - CellClient::local(registry.clone(), handle.clone()), - target.tenant(), - target.application(), - KV_NAMESPACE, - ) - .unwrap(); - let large_value = vec![42; 4 * 1024 * 1024]; - let request = KvAtomicRequest { - scope: b"repository".to_vec(), - checks: Vec::new(), - mutations: vec![put(b"branch", &large_value, None)], - }; - let committed = namespace - .atomic(mutation_identity(7), request) - .await - .unwrap(); - assert!(matches!(committed.output, KvAtomicOutcome::Applied(_))); - let entry = namespace - .get( - b"repository".to_vec(), - b"branch".to_vec(), - Some(committed.receipt), - ) - .await - .unwrap() - .output - .unwrap(); - assert_eq!(entry.value, large_value); - assert_eq!( - namespace - .list( - KvListRequest { - scope: b"repository".to_vec(), - prefix: b"br".to_vec(), - after_key: None, - limit: 10, - }, - Some(committed.receipt), - ) - .await - .unwrap() - .output - .entries, - vec![entry] - ); - let rejected = namespace - .atomic( - mutation_identity(8), - KvAtomicRequest { - scope: b"repository".to_vec(), - checks: vec![KvCheck { - key: b"branch".to_vec(), - condition: KvCondition::Absent, - }], - mutations: vec![put(b"branch", b"wrong", None)], - }, - ) - .await; - assert!(matches!( - rejected, - Err(InvocationError::Rejected(outcome)) - if outcome.output == KvAtomicOutcome::PreconditionFailed { - key: b"branch".to_vec() - } - )); - drop(namespace); - drop(handle); - drop(runtime); - let stale = authority.load(cell).await.unwrap().unwrap(); - let second_session = SessionId::from_bytes([9; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - second_session, - ) - .unwrap(); - let restored = runtime - .takeover_restored( - proof, - replica, - authority.clone(), - stale, - crate::support::fencing::fence_session(&layout, first_session, second_session) - .await - .direct_takeover() - .unwrap(), - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - layout.clone(), - Limits::default(), - ), - directory.path().join("second.sqlite"), - Owner { - session: second_session, - endpoint: "https://second.internal:8081".into(), - }, - ) - .await - .unwrap(); - let restored_namespace = KvNamespace::::new( - CellClient::local(registry, restored.clone()), - target.tenant(), - target.application(), - KV_NAMESPACE, - ) - .unwrap(); - assert_eq!( - restored_namespace - .get( - b"repository".to_vec(), - b"branch".to_vec(), - Some(committed.receipt), - ) - .await - .unwrap() - .output - .unwrap() - .value, - large_value - ); - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[test] -fn atomic_accepts_the_full_item_budget() { - let mut connection = connection(); - let mutations = (0..128) - .map(|index| put(format!("key-{index}").as_bytes(), b"v", None)) - .collect::>(); - let request = KvAtomicRequest { - scope: b"repo".to_vec(), - checks: Vec::new(), - mutations, - }; - let transaction = connection.transaction().unwrap(); - let KvAtomicOutcome::Applied(results) = kv_atomic(&transaction, 10, &request).unwrap() else { - panic!("128 items are the documented atomic budget"); - }; - assert_eq!(results.len(), 128); -} - -#[test] -fn atomic_rejects_one_item_past_the_budget() { - let mut connection = connection(); - let mutations = (0..129) - .map(|index| put(format!("key-{index}").as_bytes(), b"v", None)) - .collect::>(); - let request = KvAtomicRequest { - scope: b"repo".to_vec(), - checks: Vec::new(), - mutations, - }; - let transaction = connection.transaction().unwrap(); - assert!(kv_atomic(&transaction, 10, &request).is_err()); -} - -#[test] -fn entry_expiring_at_the_logical_time_satisfies_an_absent_check() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - let seeded = KvAtomicRequest { - scope: b"repo".to_vec(), - checks: Vec::new(), - mutations: vec![put(b"key", b"one", Some(20))], - }; - kv_atomic(&transaction, 10, &seeded).unwrap(); - transaction.commit().unwrap(); - - let after_expiry = KvAtomicRequest { - scope: b"repo".to_vec(), - checks: vec![KvCheck { - key: b"key".to_vec(), - condition: KvCondition::Absent, - }], - mutations: vec![put(b"key", b"two", None)], - }; - let transaction = connection.transaction().unwrap(); - assert!(matches!( - kv_atomic(&transaction, 20, &after_expiry).unwrap(), - KvAtomicOutcome::Applied(_) - )); -} - -#[test] -fn entry_expiring_at_the_logical_time_is_not_readable() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - let seeded = KvAtomicRequest { - scope: b"repo".to_vec(), - checks: Vec::new(), - mutations: vec![put(b"key", b"one", Some(20))], - }; - kv_atomic(&transaction, 10, &seeded).unwrap(); - transaction.commit().unwrap(); - - assert!(kv_get(&connection, b"repo", b"key", 20).unwrap().is_none()); -} diff --git a/crates/crab-cell-runtime/tests/primitives/queue.rs b/crates/crab-cell-runtime/tests/primitives/queue.rs deleted file mode 100644 index 2ed55977b..000000000 --- a/crates/crab-cell-runtime/tests/primitives/queue.rs +++ /dev/null @@ -1,333 +0,0 @@ -use std::sync::Arc; - -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::catalog::{CatalogEntry, CellCatalog}; -use crab_cell_runtime::cell::schema::install_runtime_schema; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::client::{CellClient, InvocationError}; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::primitives::maintenance::MaintenanceModule; -use crab_cell_runtime::primitives::queue::{ - QueueClaimRequest, QueueLeaseAction, QueueLeaseOutcome, QueueSendOutcome, QueueSendRequest, - QueueState, QueueTokenSource, install_queue_schema, queue_apply_lease, queue_claim, - queue_cleanup_expired, queue_send, queue_validate_claim, register_queue, -}; -use crab_cell_runtime::primitives::queue::{QueueModule, QueueNamespace}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, ModuleDescriptor, NamespaceDescriptor, RegistryBuilder, -}; -use crab_cell_runtime::registry::{MigrationDescriptor, OperationDescriptor}; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use crate::support::fixtures::{mutation_identity, now_ms}; - -const QUEUE_MODULE: &str = "queue-test"; -const QUEUE_NAMESPACE: NamespaceId = NamespaceId::from_bytes([6; 16]); -const QUEUE_MIGRATION: &str = include_str!("../../src/migrations/queue.sql"); -const QUEUE_COMMANDS: &[OperationDescriptor] = &[ - operation(1, 270 * 1024, 32), - operation(2, 8, 530 * 1024), - operation(3, 64, 16), - operation(4, 8, 5), - operation(5, 8, 16), -]; -const QUEUE_QUERIES: &[OperationDescriptor] = &[operation(1, 530 * 1024, 1), operation(2, 1, 64)]; - -struct TestQueue; - -impl QueueModule for TestQueue { - const NAMESPACE: NamespaceId = QUEUE_NAMESPACE; - const SEND_COMMAND_ID: u32 = 1; - const CLAIM_COMMAND_ID: u32 = 2; - const LEASE_COMMAND_ID: u32 = 3; - const VALIDATE_QUERY_ID: u32 = 1; - const CONTROL_COMMAND_ID: u32 = 5; - const INFO_QUERY_ID: u32 = 2; -} - -impl MaintenanceModule for TestQueue { - const MODULE: &'static str = QUEUE_MODULE; - const TICK_COMMAND_ID: u32 = 4; -} - -impl CellModule for TestQueue { - const NAME: &'static str = QUEUE_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: std::sync::OnceLock = std::sync::OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: QUEUE_MODULE, - source_digest: Digest::from_bytes([4; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: QUEUE_MIGRATION, - digest: Digest::from_bytes(*blake3::hash(QUEUE_MIGRATION.as_bytes()).as_bytes()), - }])), - commands: QUEUE_COMMANDS, - queries: QUEUE_QUERIES, - workflow_definitions: &[], - activity_types: &[], - namespaces: &[NamespaceDescriptor { - id: QUEUE_NAMESPACE, - name: QUEUE_MODULE, - role: CatalogRole::Queue, - shards: 1, - effect_targets: &[], - dead_letter: None, - }], - }) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - register_queue::(registry) - } -} - -const DEAD_LETTER_MODULE: &str = "dead-letter-test"; -const DEAD_LETTER_NAMESPACE: NamespaceId = NamespaceId::from_bytes([7; 16]); -const SOURCE_MODULE: &str = "source-queue-test"; -const SOURCE_NAMESPACE: NamespaceId = NamespaceId::from_bytes([8; 16]); - -struct DeadLetterQueue; - -impl QueueModule for DeadLetterQueue { - const NAMESPACE: NamespaceId = DEAD_LETTER_NAMESPACE; - const SEND_COMMAND_ID: u32 = 1; - const CLAIM_COMMAND_ID: u32 = 2; - const LEASE_COMMAND_ID: u32 = 3; - const VALIDATE_QUERY_ID: u32 = 1; - const CONTROL_COMMAND_ID: u32 = 5; - const INFO_QUERY_ID: u32 = 2; -} - -impl MaintenanceModule for DeadLetterQueue { - const MODULE: &'static str = DEAD_LETTER_MODULE; - const TICK_COMMAND_ID: u32 = 4; -} - -impl CellModule for DeadLetterQueue { - const NAME: &'static str = DEAD_LETTER_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - queue_descriptor(DEAD_LETTER_MODULE, DEAD_LETTER_NAMESPACE, &[], None, 10) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - register_queue::(registry) - } -} - -struct SourceQueue; - -impl QueueModule for SourceQueue { - const NAMESPACE: NamespaceId = SOURCE_NAMESPACE; - const SEND_COMMAND_ID: u32 = 1; - const CLAIM_COMMAND_ID: u32 = 2; - const LEASE_COMMAND_ID: u32 = 3; - const VALIDATE_QUERY_ID: u32 = 1; - const CONTROL_COMMAND_ID: u32 = 5; - const INFO_QUERY_ID: u32 = 2; -} - -impl MaintenanceModule for SourceQueue { - const MODULE: &'static str = SOURCE_MODULE; - const TICK_COMMAND_ID: u32 = 4; - const QUEUE_DEAD_LETTER: Option = - Some( - crab_cell_runtime::primitives::queue::QueueDeadLetterTarget::new( - DEAD_LETTER_MODULE, - DEAD_LETTER_NAMESPACE, - 1, - 1, - 1, - ), - ); -} - -impl CellModule for SourceQueue { - const NAME: &'static str = SOURCE_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - queue_descriptor( - SOURCE_MODULE, - SOURCE_NAMESPACE, - &[DEAD_LETTER_NAMESPACE], - Some(DEAD_LETTER_NAMESPACE), - 11, - ) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - register_queue::(registry) - } -} - -struct BadSourceQueue; - -impl QueueModule for BadSourceQueue { - const NAMESPACE: NamespaceId = NamespaceId::from_bytes([9; 16]); - const SEND_COMMAND_ID: u32 = 1; - const CLAIM_COMMAND_ID: u32 = 2; - const LEASE_COMMAND_ID: u32 = 3; - const VALIDATE_QUERY_ID: u32 = 1; - const CONTROL_COMMAND_ID: u32 = 5; - const INFO_QUERY_ID: u32 = 2; -} - -impl MaintenanceModule for BadSourceQueue { - const MODULE: &'static str = "bad-source-queue-test"; - const TICK_COMMAND_ID: u32 = 4; - const QUEUE_DEAD_LETTER: Option = - Some( - crab_cell_runtime::primitives::queue::QueueDeadLetterTarget::new( - DEAD_LETTER_MODULE, - DEAD_LETTER_NAMESPACE, - 2, - 1, - 1, - ), - ); -} - -impl CellModule for BadSourceQueue { - const NAME: &'static str = "bad-source-queue-test"; - - fn descriptor(&self) -> &'static ModuleDescriptor { - queue_descriptor( - Self::NAME, - Self::NAMESPACE, - &[DEAD_LETTER_NAMESPACE], - Some(DEAD_LETTER_NAMESPACE), - 12, - ) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - register_queue::(registry) - } -} - -fn queue_descriptor( - name: &'static str, - namespace: NamespaceId, - effect_targets: &'static [NamespaceId], - dead_letter: Option, - digest: u8, -) -> &'static ModuleDescriptor { - Box::leak(Box::new(ModuleDescriptor { - name, - source_digest: Digest::from_bytes([digest; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: QUEUE_MIGRATION, - digest: Digest::from_bytes(*blake3::hash(QUEUE_MIGRATION.as_bytes()).as_bytes()), - }])), - commands: QUEUE_COMMANDS, - queries: QUEUE_QUERIES, - workflow_definitions: &[], - activity_types: &[], - namespaces: Box::leak(Box::new([NamespaceDescriptor { - id: namespace, - name, - role: CatalogRole::Queue, - shards: 1, - effect_targets, - dead_letter, - }])), - })) -} - -const fn operation(id: u32, input_limit: u32, output_limit: u32) -> OperationDescriptor { - OperationDescriptor { - id, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit, - output_limit, - } -} - -fn queue_registry() -> Arc { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "queue-api-test".into(), - cargo_lock_digest: Digest::from_bytes([5; 32]), - }); - builder.register(TestQueue).unwrap(); - Arc::new(builder.finish().unwrap()) -} - -#[test] -fn queue_registry_validates_dead_letter_module_ids_and_shards() { - let mut valid = RegistryBuilder::new(BuildDescriptor { - source_revision: "queue-dead-letter-test".into(), - cargo_lock_digest: Digest::from_bytes([13; 32]), - }); - valid.register(SourceQueue).unwrap(); - valid.register(DeadLetterQueue).unwrap(); - valid.finish().unwrap(); - - let mut invalid = RegistryBuilder::new(BuildDescriptor { - source_revision: "queue-dead-letter-drift-test".into(), - cargo_lock_digest: Digest::from_bytes([14; 32]), - }); - invalid.register(BadSourceQueue).unwrap(); - invalid.register(DeadLetterQueue).unwrap(); - assert!(invalid.finish().is_err()); -} - -struct Tokens(u8); - -impl QueueTokenSource for Tokens { - fn next_token(&mut self) -> crab_cell_runtime::Result<[u8; 16]> { - self.0 = self - .0 - .checked_add(1) - .ok_or(crab_cell_runtime::Error::Command( - "test queue token overflow", - ))?; - Ok([self.0; 16]) - } -} - -fn connection() -> crab_ltx::rusqlite::Connection { - let mut connection = crab_ltx::rusqlite::Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - crab_cell_runtime::CellId::from_bytes([1; 32]), - IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - connection -} - -fn send_request(producer: u8, payload: &[u8], available_at_ms: i64) -> QueueSendRequest { - QueueSendRequest { - producer_id: [producer; 16], - payload: payload.to_vec(), - available_at_ms, - } -} - -mod lease; -mod namespace; -mod send; diff --git a/crates/crab-cell-runtime/tests/primitives/queue/lease.rs b/crates/crab-cell-runtime/tests/primitives/queue/lease.rs deleted file mode 100644 index 3af629708..000000000 --- a/crates/crab-cell-runtime/tests/primitives/queue/lease.rs +++ /dev/null @@ -1,156 +0,0 @@ -//! Claim tokens, lease transitions, retry limits, and reclaim bounds. - -use super::*; - -#[test] -fn claim_tokens_require_published_live_lease_for_ack_retry_and_extend() { - let mut connection = connection(); - let namespace = NamespaceId::from_bytes([3; 16]); - let transaction = connection.transaction().unwrap(); - for producer in 1..=2 { - queue_send( - &transaction, - namespace, - 10, - &send_request(producer, &[producer], 10), - ) - .unwrap(); - } - transaction.commit().unwrap(); - - let transaction = connection.transaction().unwrap(); - let mut tokens = Tokens(0); - let claimed = queue_claim(&transaction, 20, 2, 5_000, &mut tokens).unwrap(); - assert_eq!(claimed.len(), 2); - assert_eq!(claimed[0].attempt, 1); - transaction.commit().unwrap(); - assert!(queue_validate_claim(&connection, 21, &claimed).unwrap()); - assert!(!queue_validate_claim(&connection, 4_021, &claimed).unwrap()); - - let first = &claimed[0]; - let transaction = connection.transaction().unwrap(); - assert_eq!( - queue_apply_lease( - &transaction, - 100, - first.message_id, - [99; 16], - QueueLeaseAction::Ack, - ) - .unwrap(), - QueueLeaseOutcome::LeaseLost - ); - assert_eq!( - queue_apply_lease( - &transaction, - 100, - first.message_id, - first.token, - QueueLeaseAction::Extend { - extension_ms: 10_000, - }, - ) - .unwrap(), - QueueLeaseOutcome::Applied { - state: QueueState::Leased, - lease_until_ms: Some(10_100), - } - ); - transaction.commit().unwrap(); - - let transaction = connection.transaction().unwrap(); - assert_eq!( - queue_apply_lease( - &transaction, - 101, - first.message_id, - first.token, - QueueLeaseAction::Ack, - ) - .unwrap(), - QueueLeaseOutcome::Applied { - state: QueueState::Acked, - lease_until_ms: None, - } - ); - transaction.commit().unwrap(); -} -#[test] -fn retry_and_expired_reclaim_preserve_attempt_limits_and_cleanup_bounds() { - let mut connection = connection(); - let namespace = NamespaceId::from_bytes([3; 16]); - let transaction = connection.transaction().unwrap(); - queue_send(&transaction, namespace, 0, &send_request(1, b"payload", 0)).unwrap(); - transaction.commit().unwrap(); - - let mut tokens = Tokens(0); - let transaction = connection.transaction().unwrap(); - let first = queue_claim(&transaction, 0, 1, 5_000, &mut tokens) - .unwrap() - .remove(0); - assert_eq!( - queue_apply_lease( - &transaction, - 1, - first.message_id, - first.token, - QueueLeaseAction::Retry { delay_ms: 100 }, - ) - .unwrap(), - QueueLeaseOutcome::Applied { - state: QueueState::Ready, - lease_until_ms: None, - } - ); - transaction.commit().unwrap(); - - let transaction = connection.transaction().unwrap(); - assert!( - queue_claim(&transaction, 100, 1, 5_000, &mut tokens) - .unwrap() - .is_empty() - ); - let second = queue_claim(&transaction, 101, 1, 5_000, &mut tokens) - .unwrap() - .remove(0); - assert_eq!(second.attempt, 2); - transaction - .execute( - "UPDATE queue_messages SET attempt = 20, lease_until_ms = 102 WHERE message_id = ?1", - [second.message_id.as_slice()], - ) - .unwrap(); - transaction.commit().unwrap(); - - let transaction = connection.transaction().unwrap(); - assert!( - queue_claim(&transaction, 102, 1, 5_000, &mut tokens) - .unwrap() - .is_empty() - ); - assert_eq!( - transaction - .query_row( - "SELECT state FROM queue_messages WHERE message_id = ?1", - [second.message_id.as_slice()], - |row| row.get::<_, i64>(0), - ) - .unwrap(), - 3 - ); - transaction - .execute( - "UPDATE queue_messages SET expires_at_ms = 102 WHERE message_id = ?1", - [second.message_id.as_slice()], - ) - .unwrap(); - assert_eq!(queue_cleanup_expired(&transaction, 102).unwrap(), 1); - assert_eq!( - transaction - .query_row("SELECT count(*) FROM queue_dedup", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 1 - ); - transaction.commit().unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/queue/namespace.rs b/crates/crab-cell-runtime/tests/primitives/queue/namespace.rs deleted file mode 100644 index 1086ff212..000000000 --- a/crates/crab-cell-runtime/tests/primitives/queue/namespace.rs +++ /dev/null @@ -1,199 +0,0 @@ -//! Typed queue namespace recovery after owner loss. - -use super::*; - -#[tokio::test] -async fn typed_queue_namespace_recovers_after_owner_loss() { - let registry = queue_registry(); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([3; 16]), - QUEUE_NAMESPACE, - &0_u32.to_be_bytes(), - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([2; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store, Path::from("runtime"), [3; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Queue, - registry.module_code(QUEUE_MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let first_session = SessionId::from_bytes([4; 16]); - let observed = authority - .create_initial( - &proof, - incarnation, - Owner { - session: first_session, - endpoint: "https://first.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - first_session, - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - observed, - directory.path().join("first.sqlite"), - install_queue_schema, - ) - .await - .unwrap(); - let queue = QueueNamespace::::new( - CellClient::local(registry.clone(), handle.clone()), - target.tenant(), - target.application(), - ) - .unwrap(); - let available_at_ms = now_ms() + 100; - let sent = queue - .send( - mutation_identity(7), - QueueSendRequest { - producer_id: [7; 16], - payload: b"job".to_vec(), - available_at_ms, - }, - ) - .await - .unwrap(); - assert!(matches!(sent.output, QueueSendOutcome::Sent { .. })); - let conflict = queue - .send( - mutation_identity(8), - QueueSendRequest { - producer_id: [7; 16], - payload: b"different".to_vec(), - available_at_ms, - }, - ) - .await; - assert!(matches!( - conflict, - Err(InvocationError::Rejected(outcome)) - if outcome.output == QueueSendOutcome::ProducerConflict - )); - tokio::time::sleep(std::time::Duration::from_millis(120)).await; - let claimed = queue - .claim( - mutation_identity(9), - 0, - QueueClaimRequest { - limit: 1, - lease_ms: 10_000, - }, - ) - .await - .unwrap(); - assert_eq!(claimed.output.len(), 1); - assert_eq!(claimed.output[0].payload, b"job"); - assert_eq!( - authority - .load(cell) - .await - .unwrap() - .unwrap() - .value() - .next_due_ms, - Some(claimed.output[0].lease_until_ms) - ); - assert!( - queue - .validate_claim(0, claimed.output.clone(), Some(claimed.receipt)) - .await - .unwrap() - .output - ); - drop(queue); - drop(handle); - drop(runtime); - let stale = authority.load(cell).await.unwrap().unwrap(); - let second_session = SessionId::from_bytes([11; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - second_session, - ) - .unwrap(); - let restored = runtime - .takeover_restored( - proof, - replica, - authority.clone(), - stale, - crate::support::fencing::fence_session(&layout, first_session, second_session) - .await - .direct_takeover() - .unwrap(), - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - layout.clone(), - Limits::default(), - ), - directory.path().join("second.sqlite"), - Owner { - session: second_session, - endpoint: "https://second.internal:8081".into(), - }, - ) - .await - .unwrap(); - let restored_queue = QueueNamespace::::new( - CellClient::local(registry, restored.clone()), - target.tenant(), - target.application(), - ) - .unwrap(); - assert!( - restored_queue - .validate_claim(0, claimed.output.clone(), Some(claimed.receipt)) - .await - .unwrap() - .output - ); - let acked = restored_queue - .ack( - mutation_identity(10), - 0, - claimed.output[0].message_id, - claimed.output[0].token, - ) - .await - .unwrap(); - assert_eq!( - acked.output, - QueueLeaseOutcome::Applied { - state: QueueState::Acked, - lease_until_ms: None, - } - ); - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/queue/send.rs b/crates/crab-cell-runtime/tests/primitives/queue/send.rs deleted file mode 100644 index 6e2f04257..000000000 --- a/crates/crab-cell-runtime/tests/primitives/queue/send.rs +++ /dev/null @@ -1,66 +0,0 @@ -//! Producer send idempotency and delayed scheduling. - -use super::*; - -#[test] -fn producer_send_is_idempotent_and_conflicts_on_changed_payload_or_schedule() { - let mut connection = connection(); - let namespace = NamespaceId::from_bytes([3; 16]); - let request = send_request(1, b"payload", 20); - let transaction = connection.transaction().unwrap(); - let first = queue_send(&transaction, namespace, 10, &request).unwrap(); - let second = queue_send(&transaction, namespace, 10, &request).unwrap(); - assert_eq!(first, second); - assert_eq!( - queue_send( - &transaction, - namespace, - 10, - &send_request(1, b"changed", 20), - ) - .unwrap(), - QueueSendOutcome::ProducerConflict - ); - assert_eq!( - queue_send( - &transaction, - namespace, - 10, - &send_request(1, b"payload", 21), - ) - .unwrap(), - QueueSendOutcome::ProducerConflict - ); - transaction.commit().unwrap(); - assert_eq!( - connection - .query_row("SELECT count(*) FROM queue_messages", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 1 - ); -} -#[test] -fn delayed_send_with_past_schedule_becomes_immediately_ready() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - let outcome = queue_send( - &transaction, - QUEUE_NAMESPACE, - 20, - &send_request(2, b"delayed-effect", 10), - ) - .unwrap(); - let QueueSendOutcome::Sent { message_id } = outcome else { - panic!("first queue send must insert") - }; - let due_at_ms: i64 = transaction - .query_row( - "SELECT due_at_ms FROM queue_messages WHERE message_id = ?1", - [message_id.as_slice()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(due_at_ms, 20); - transaction.commit().unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/sql.rs b/crates/crab-cell-runtime/tests/primitives/sql.rs deleted file mode 100644 index 1d1ced4a8..000000000 --- a/crates/crab-cell-runtime/tests/primitives/sql.rs +++ /dev/null @@ -1,630 +0,0 @@ -use std::sync::Arc; - -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::catalog::{CatalogEntry, CellCatalog}; -use crab_cell_runtime::cell::schema::install_runtime_schema; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::client::{CellClient, InvocationError}; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{ - ApplicationId, CellId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::primitives::blob::install_blob_schema; -use crab_cell_runtime::primitives::cron::install_cron_schema; -use crab_cell_runtime::primitives::sql::{ - SqlBatch, SqlResultSet, SqlStatement, SqlValue, register_sql, sql_batch, sql_query_batch, -}; -use crab_cell_runtime::primitives::sql::{SqlCell, SqlModule}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, Command, CommandContext, CommandResult, ModuleDescriptor, - NamespaceDescriptor, RegistryBuilder, -}; -use crab_cell_runtime::registry::{MigrationDescriptor, OperationDescriptor}; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits, rusqlite::Connection}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use crate::support::fixtures::mutation_identity; - -const SQL_MODULE: &str = "sql-test"; -const SQL_NAMESPACE: NamespaceId = NamespaceId::from_bytes([7; 16]); -const SQL_MIGRATION: &str = - "CREATE TABLE app_items(id INTEGER PRIMARY KEY, name TEXT NOT NULL, payload BLOB);"; - -struct TestSql; - -impl SqlModule for TestSql { - const MODULE: &'static str = SQL_MODULE; - const BATCH_COMMAND_ID: u32 = 1; - const BATCH_QUERY_ID: u32 = 1; -} - -impl CellModule for TestSql { - const NAME: &'static str = SQL_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: std::sync::OnceLock = std::sync::OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: SQL_MODULE, - source_digest: Digest::from_bytes([8; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: SQL_MIGRATION, - digest: Digest::from_bytes(*blake3::hash(SQL_MIGRATION.as_bytes()).as_bytes()), - }])), - commands: &[ - OperationDescriptor { - id: 2, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 8, - output_limit: 8, - }, - OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 1024 * 1024, - output_limit: 1024 * 1024, - }, - ], - queries: &[OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 1024 * 1024, - output_limit: 1024 * 1024, - }], - workflow_definitions: &[], - activity_types: &[], - namespaces: &[NamespaceDescriptor { - id: SQL_NAMESPACE, - name: SQL_MODULE, - role: CatalogRole::Sql, - shards: 1, - effect_targets: &[], - dead_letter: None, - }], - }) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - register_sql::(registry)?; - registry.bind_command::() - } -} - -const BLOB_BYTES: usize = (1 << 20) + 17; -const BLOB_CHUNK: usize = 256 * 1024; - -struct WriteBlob; - -impl Command for WriteBlob { - const MODULE: &'static str = SQL_MODULE; - const ID: u32 = 2; - const CODEC_VERSION: u32 = 1; - type Input = u8; - type Output = (); - - fn execute( - context: &mut CommandContext<'_, '_>, - mode: u8, - ) -> crab_cell_runtime::Result> { - if mode == 0 { - context.sql(&SqlBatch { statements: vec![statement( - "INSERT INTO app_items(id, name, payload) VALUES (8, 'incremental', zeroblob(?1))", - vec![SqlValue::Integer(BLOB_BYTES as i64)], - )] })?; - for offset in (0..BLOB_BYTES).step_by(BLOB_CHUNK) { - let chunk = vec![(offset / BLOB_CHUNK) as u8; BLOB_CHUNK.min(BLOB_BYTES - offset)]; - context.write_sql_blob("app_items", "payload", 8, offset, &chunk)?; - } - } else if mode <= 2 { - // A release of staged storage and partial BLOB application must - // both roll back on rejection or a handler error. - context.sql(&SqlBatch { - statements: vec![statement("DELETE FROM app_items WHERE id = 7", vec![])], - })?; - context.write_sql_blob("app_items", "payload", 8, 0, &[99; 32])?; - if mode == 1 { - return Ok(CommandResult::Rejected(())); - } - return Err(crab_cell_runtime::Error::Command( - "injected after incremental write", - )); - } else { - for table in [ - "sys_meta", - "SyS_requests", - "kv_entries", - "queue_messages", - "workflow_runs", - "blob_objects", - "cron_schedules", - "CaPaCiTy_total", - "capacity_reservations", - "sqlite_schema", - ] { - assert!( - matches!( - context.write_sql_blob(table, "payload", 1, 0, &[1]), - Err(crab_cell_runtime::Error::Command( - "SQL blob targets a protected table" - )) - ), - "{table}" - ); - } - assert!( - context - .write_sql_blob("app_items", "payload", 8, 0, &vec![1; 1 << 20]) - .is_err() - ); - assert!( - context - .write_sql_blob("app_items", "payload", 8, BLOB_BYTES, &[1]) - .is_err() - ); - assert!( - context - .write_sql_blob("app_items", "payload", 8, usize::MAX, &[1]) - .is_err() - ); - assert!( - context - .write_sql_blob("app_items", "payload", 999, 0, &[1]) - .is_err() - ); - assert!( - context - .write_sql_blob("app_items", "id", 8, 0, &[1]) - .is_err() - ); - } - Ok(CommandResult::Success(())) - } -} - -async fn assert_blob(sql: &SqlCell) { - for offset in (0..BLOB_BYTES).step_by(BLOB_CHUNK) { - let result = sql.query(None, SqlBatch { statements: vec![statement( - "SELECT length(payload), substr(payload, ?1, ?2) FROM app_items WHERE id = 8", - vec![SqlValue::Integer(offset as i64 + 1), SqlValue::Integer(BLOB_CHUNK as i64)], - )] }).await.unwrap(); - assert_eq!( - result.output[0].rows, - vec![vec![ - SqlValue::Integer(BLOB_BYTES as i64), - SqlValue::Blob(vec![ - (offset / BLOB_CHUNK) as u8; - BLOB_CHUNK.min(BLOB_BYTES - offset) - ]), - ]] - ); - } -} - -fn sql_registry() -> Arc { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "sql-api-test".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }); - builder.register(TestSql).unwrap(); - Arc::new(builder.finish().unwrap()) -} - -fn install_sql_schema( - transaction: &crab_ltx::rusqlite::Transaction<'_>, -) -> crab_cell_runtime::Result<()> { - transaction.execute_batch(SQL_MIGRATION)?; - Ok(()) -} - -fn connection() -> Connection { - let mut connection = Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - CellId::from_bytes([1; 32]), - IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - connection - .execute_batch( - "CREATE TABLE app_items(id INTEGER PRIMARY KEY, name TEXT NOT NULL, payload BLOB);\n\ - CREATE VIEW app_runtime_metadata AS SELECT commit_sequence FROM sys_meta;\n\ - CREATE TRIGGER app_items_guard AFTER INSERT ON app_items BEGIN\n\ - UPDATE sys_meta SET logical_time_ms = logical_time_ms + 1 WHERE singleton = 1;\n\ - END;", - ) - .unwrap(); - connection -} - -fn statement(sql: &str, parameters: Vec) -> SqlStatement { - SqlStatement { - sql: sql.to_owned(), - parameters, - } -} - -#[test] -fn typed_batch_mutates_and_materializes_in_order() { - let mut connection = connection(); - connection - .execute_batch("DROP TRIGGER app_items_guard") - .unwrap(); - let transaction = connection.transaction().unwrap(); - let results = sql_batch( - &transaction, - &SqlBatch { - statements: vec![ - statement( - "INSERT INTO app_items(id, name, payload) VALUES (?1, ?2, ?3)", - vec![ - SqlValue::Integer(7), - SqlValue::Text("crab".into()), - SqlValue::Blob(vec![1, 2, 3]), - ], - ), - statement( - "SELECT id, name, payload, NULL, 1.5 FROM app_items WHERE id = ?1", - vec![SqlValue::Integer(7)], - ), - ], - }, - ) - .unwrap(); - assert_eq!( - results, - vec![ - SqlResultSet { - columns: Vec::new(), - rows: Vec::new(), - rows_affected: 1, - }, - SqlResultSet { - columns: vec![ - "id".into(), - "name".into(), - "payload".into(), - "NULL".into(), - "1.5".into() - ], - rows: vec![vec![ - SqlValue::Integer(7), - SqlValue::Text("crab".into()), - SqlValue::Blob(vec![1, 2, 3]), - SqlValue::Null, - SqlValue::Real(1.5), - ]], - rows_affected: 0, - }, - ] - ); - transaction.commit().unwrap(); -} - -#[test] -fn authorizer_blocks_runtime_tables_and_indirect_trigger_or_view_access() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_blob_schema(&transaction).unwrap(); - install_cron_schema(&transaction).unwrap(); - transaction - .execute_batch(crab_cell_runtime::primitives::capacity::SCHEMA) - .unwrap(); - for sql in [ - "SELECT commit_sequence FROM sys_meta", - "SELECT object_key FROM blob_objects", - "SELECT schedule_id FROM cron_schedules", - "DELETE FROM blob_objects", - "DELETE FROM cron_schedules", - "SELECT pages FROM capacity_total", - "SELECT pages FROM capacity_reservations", - "UPDATE capacity_total SET pages = 0", - "DELETE FROM capacity_reservations", - "SELECT commit_sequence FROM app_runtime_metadata", - "INSERT INTO app_items(id, name) VALUES (1, 'blocked by trigger')", - "PRAGMA user_version", - "ATTACH DATABASE ':memory:' AS other", - "SAVEPOINT nested", - "SELECT load_extension('missing')", - "CREATE TABLE app_other(value INTEGER)", - ] { - assert!( - sql_batch( - &transaction, - &SqlBatch { - statements: vec![statement(sql, Vec::new())], - }, - ) - .is_err(), - "statement unexpectedly authorized: {sql}" - ); - } - transaction.rollback().unwrap(); - - assert_eq!( - connection - .query_row("SELECT count(*) FROM sys_meta", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 1 - ); -} - -#[test] -fn read_boundary_rejects_mutation_and_mutation_returning() { - let mut connection = connection(); - connection - .execute_batch("DROP TRIGGER app_items_guard") - .unwrap(); - let update = SqlBatch { - statements: vec![statement( - "UPDATE app_items SET name = 'changed'", - Vec::new(), - )], - }; - assert!(sql_query_batch(&connection, &update).is_err()); - - let returning = SqlBatch { - statements: vec![statement( - "INSERT INTO app_items(id, name) VALUES (3, 'three') RETURNING id", - Vec::new(), - )], - }; - let transaction = connection.transaction().unwrap(); - assert!(sql_batch(&transaction, &returning).is_err()); - transaction.rollback().unwrap(); -} - -#[test] -fn batch_rejects_unbounded_or_ambiguous_inputs_and_outputs() { - let connection = connection(); - let too_many = SqlBatch { - statements: (0..129) - .map(|_| statement("SELECT 1", Vec::new())) - .collect(), - }; - assert!(sql_query_batch(&connection, &too_many).is_err()); - - let multiple = SqlBatch { - statements: vec![statement("SELECT ';'; SELECT 2", Vec::new())], - }; - assert!(sql_query_batch(&connection, &multiple).is_err()); - - let wrong_parameters = SqlBatch { - statements: vec![statement("SELECT ?1", Vec::new())], - }; - assert!(sql_query_batch(&connection, &wrong_parameters).is_err()); - - let too_many_rows = SqlBatch { - statements: vec![statement( - "WITH RECURSIVE values_(value) AS (SELECT 1 UNION ALL SELECT value + 1 FROM values_ WHERE value <= 1000) SELECT value FROM values_", - Vec::new(), - )], - }; - assert!(sql_query_batch(&connection, &too_many_rows).is_err()); - - let oversized = SqlBatch { - statements: vec![statement( - "SELECT ?1", - vec![SqlValue::Blob(vec![0; (1 << 20) + 1])], - )], - }; - assert!(sql_query_batch(&connection, &oversized).is_err()); - - let oversized_result = SqlBatch { - statements: vec![statement("SELECT zeroblob(1048576)", Vec::new())], - }; - assert!(sql_query_batch(&connection, &oversized_result).is_err()); -} - -#[test] -fn quoted_semicolons_and_comments_do_not_create_a_second_statement() { - let connection = connection(); - let results = sql_query_batch( - &connection, - &SqlBatch { - statements: vec![statement("SELECT ';' AS value /* ; */ -- ;\n", Vec::new())], - }, - ) - .unwrap(); - assert_eq!(results[0].rows, vec![vec![SqlValue::Text(";".into())]]); -} - -#[tokio::test] -async fn typed_sql_cell_publishes_enforces_read_only_queries_and_survives_restore() { - let registry = sql_registry(); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([3; 16]), - SQL_NAMESPACE, - b"repository-7", - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([2; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store, Path::from("runtime"), [3; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Sql, - registry.module_code(SQL_MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout); - let first_session = SessionId::from_bytes([4; 16]); - let observed = authority - .create_initial( - &proof, - incarnation, - Owner { - session: first_session, - endpoint: "https://first.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - first_session, - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - observed, - directory.path().join("first.sqlite"), - install_sql_schema, - ) - .await - .unwrap(); - let sql = SqlCell::::new( - CellClient::local(registry.clone(), handle.clone()), - target.clone(), - ) - .unwrap(); - let committed = sql - .batch( - mutation_identity(7), - SqlBatch { - statements: vec![ - statement( - "INSERT INTO app_items(id, name, payload) VALUES (?1, ?2, ?3)", - vec![ - SqlValue::Integer(7), - SqlValue::Text("crab".into()), - SqlValue::Blob(vec![1, 2, 3]), - ], - ), - statement( - "SELECT name FROM app_items WHERE id = ?1", - vec![SqlValue::Integer(7)], - ), - ], - }, - ) - .await - .unwrap(); - assert_eq!(committed.receipt.commit_sequence, 1); - assert_eq!( - committed.output[1].rows, - vec![vec![SqlValue::Text("crab".into())]] - ); - let mutation_query = sql - .query( - Some(committed.receipt), - SqlBatch { - statements: vec![statement( - "UPDATE app_items SET name = 'wrong' WHERE id = 7", - Vec::new(), - )], - }, - ) - .await; - assert!(matches!( - mutation_query, - Err(InvocationError::NotStarted(_)) - )); - let client = CellClient::local(registry.clone(), handle.clone()); - client - .command::(&target, mutation_identity(8), 0) - .await - .unwrap(); - assert!(matches!( - client - .command::(&target, mutation_identity(9), 1) - .await, - Err(InvocationError::Rejected(_)) - )); - assert!( - client - .command::(&target, mutation_identity(10), 2) - .await - .is_err() - ); - let latest = client - .command::(&target, mutation_identity(11), 3) - .await - .unwrap(); - assert_blob(&sql).await; - handle.drain().await.unwrap(); - - let idle = authority.load(cell).await.unwrap().unwrap(); - let second_session = SessionId::from_bytes([9; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - second_session, - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - replica, - authority, - idle, - directory.path().join("second.sqlite"), - Owner { - session: second_session, - endpoint: "https://second.internal:8081".into(), - }, - ) - .await - .unwrap(); - let restored_sql = - SqlCell::::new(CellClient::local(registry, restored.clone()), target).unwrap(); - let observed = restored_sql - .query( - Some(committed.receipt), - SqlBatch { - statements: vec![statement( - "SELECT id, name, payload FROM app_items WHERE id = ?1", - vec![SqlValue::Integer(7)], - )], - }, - ) - .await - .unwrap(); - assert_eq!(observed.receipt, latest.receipt); - assert_eq!( - observed.output[0].rows, - vec![vec![ - SqlValue::Integer(7), - SqlValue::Text("crab".into()), - SqlValue::Blob(vec![1, 2, 3]), - ]] - ); - assert_blob(&restored_sql).await; - restored.drain().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/workflow.rs b/crates/crab-cell-runtime/tests/primitives/workflow.rs deleted file mode 100644 index 46faa2cc3..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow.rs +++ /dev/null @@ -1,219 +0,0 @@ -use std::sync::Arc; - -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::catalog::{CatalogEntry, CellCatalog}; -use crab_cell_runtime::cell::executor::{HandlerOutcome, MutationIdentity}; -use crab_cell_runtime::cell::schema::install_runtime_schema; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::identity::{IncarnationId, RequestId}; -use crab_cell_runtime::peer::wire as peer_wire; -use crab_cell_runtime::primitives::effects::EffectCommandIntent; -use crab_cell_runtime::primitives::workflow::{ - ActivityCompletion, ActivityCompletionOutcome, ActivityLeaseOutcome, ActivitySupport, - ActivityTokenSource, WorkflowAction, WorkflowContext, WorkflowControl, WorkflowControlAction, - WorkflowDecision, WorkflowDefinition, WorkflowOutcome, WorkflowSignal, WorkflowStart, - WorkflowStatus, install_workflow_schema, workflow_cancel, workflow_claim_activities, - workflow_cleanup_terminal, workflow_complete_activity, workflow_control, - workflow_extend_activity, workflow_fire_timer, workflow_signal, workflow_start, - workflow_validate_activity_claim, -}; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; -use prost::Message; - -const WORKFLOW_NAMESPACE: NamespaceId = NamespaceId::from_bytes([3; 16]); -const EFFECT_NAMESPACE: NamespaceId = NamespaceId::from_bytes([8; 16]); -static EFFECT_TARGETS: [NamespaceId; 1] = [EFFECT_NAMESPACE]; - -#[derive(Clone, Copy)] -struct Definition { - digest: Digest, -} - -impl WorkflowDefinition for Definition { - fn digest(&self) -> Digest { - self.digest - } - - fn effect_targets(&self) -> &'static [NamespaceId] { - &EFFECT_TARGETS - } - - fn transition( - &self, - _state: &[u8], - event: &[u8], - context: WorkflowContext, - ) -> crab_cell_runtime::Result { - match event { - b"start" => { - let timer = context.action_id(0); - let activity = context.action_id(1); - let mut state = timer.to_vec(); - state.extend_from_slice(&activity); - Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state, - result: None, - actions: vec![ - WorkflowAction::Timer { due_at_ms: 20 }, - WorkflowAction::Activity { - activity_type: "email".into(), - input: b"payload".to_vec(), - due_at_ms: 10, - expires_at_ms: 20_000, - }, - ], - }) - } - value if value.starts_with(b"timer\0") => Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: b"done".to_vec(), - result: Some(b"timer-fired".to_vec()), - actions: Vec::new(), - }), - b"finish" => Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: b"done".to_vec(), - result: Some(b"signalled".to_vec()), - actions: Vec::new(), - }), - b"finish-with-effect" => { - effect_decision(&context, context.source().tenant(), EFFECT_NAMESPACE) - } - b"finish-with-undeclared-effect" => effect_decision( - &context, - context.source().tenant(), - NamespaceId::from_bytes([9; 16]), - ), - b"finish-with-cross-tenant-effect" => { - effect_decision(&context, TenantId::from_bytes([99; 16]), EFFECT_NAMESPACE) - } - event => Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state: event.to_vec(), - result: None, - actions: Vec::new(), - }), - } - } -} - -fn effect_decision( - context: &WorkflowContext, - tenant: TenantId, - namespace: NamespaceId, -) -> crab_cell_runtime::Result { - Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: b"done".to_vec(), - result: Some(b"effect-scheduled".to_vec()), - actions: vec![WorkflowAction::Effect { - intent: EffectCommandIntent { - target: CellTarget::new( - tenant, - context.source().application(), - namespace, - b"destination", - )?, - command_id: 7, - codec_version: 1, - input: b"canonical-destination-command".to_vec(), - expires_at_ms: 20_000, - }, - }], - }) -} - -struct InvalidDefinition; - -impl WorkflowDefinition for InvalidDefinition { - fn digest(&self) -> Digest { - Digest::from_bytes([12; 32]) - } - - fn transition( - &self, - _state: &[u8], - _event: &[u8], - _context: WorkflowContext, - ) -> crab_cell_runtime::Result { - Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: Vec::new(), - result: None, - actions: vec![WorkflowAction::Timer { due_at_ms: 20 }], - }) - } -} - -struct Tokens(u8); - -impl ActivityTokenSource for Tokens { - fn next_token(&mut self) -> crab_cell_runtime::Result<[u8; 16]> { - self.0 = self - .0 - .checked_add(1) - .ok_or(crab_cell_runtime::Error::Command( - "test activity token overflow", - ))?; - Ok([self.0; 16]) - } -} - -fn connection() -> crab_ltx::rusqlite::Connection { - let mut connection = crab_ltx::rusqlite::Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - source_target().cell_id(), - IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - let transaction = connection.transaction().unwrap(); - install_workflow_schema(&transaction).unwrap(); - transaction.commit().unwrap(); - connection -} - -fn source_target() -> CellTarget { - CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - WORKFLOW_NAMESPACE, - b"workflow", - ) - .unwrap() -} - -fn start(request_id: u8) -> WorkflowStart { - WorkflowStart { - workflow_id: b"build-42".to_vec(), - request_id: RequestId::from_bytes([request_id; 16]), - event: b"start".to_vec(), - } -} - -fn applied(outcome: WorkflowOutcome) -> ([u8; 16], WorkflowStatus, u64) { - match outcome { - WorkflowOutcome::Applied { - run_id, - status, - event_sequence, - } => (run_id, status, event_sequence), - other => panic!("expected applied workflow outcome, got {other:?}"), - } -} - -mod activity; -mod control; -mod decisions; -mod recovery; diff --git a/crates/crab-cell-runtime/tests/primitives/workflow/activity.rs b/crates/crab-cell-runtime/tests/primitives/workflow/activity.rs deleted file mode 100644 index 2753c7484..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow/activity.rs +++ /dev/null @@ -1,233 +0,0 @@ -//! Activity leases, timer firing, and retention cleanup. - -use super::*; - -#[test] -fn due_timer_fires_once_and_terminal_transition_cancels_sibling_activity() { - let mut connection = connection(); - let definition = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let transaction = connection.transaction().unwrap(); - let (run_id, _, _) = applied( - workflow_start(&transaction, &source_target(), 10, &start(5), &definition).unwrap(), - ); - let timer_id: [u8; 16] = transaction - .query_row("SELECT timer_id FROM workflow_timers", [], |row| { - row.get::<_, Vec>(0) - }) - .unwrap() - .try_into() - .unwrap(); - assert_eq!( - workflow_fire_timer( - &transaction, - &source_target(), - 19, - run_id, - timer_id, - &definition - ) - .unwrap(), - WorkflowOutcome::NotDue - ); - assert_eq!( - applied( - workflow_fire_timer( - &transaction, - &source_target(), - 20, - run_id, - timer_id, - &definition - ) - .unwrap(), - ), - (run_id, WorkflowStatus::Completed, 2) - ); - assert_eq!( - workflow_fire_timer( - &transaction, - &source_target(), - 21, - run_id, - timer_id, - &definition - ) - .unwrap(), - WorkflowOutcome::Duplicate { - run_id, - status: WorkflowStatus::Completed, - event_sequence: 2, - } - ); - let activity_state: i64 = transaction - .query_row("SELECT state FROM workflow_activities", [], |row| { - row.get(0) - }) - .unwrap(); - assert_eq!(activity_state, 4); - transaction.commit().unwrap(); -} -#[test] -fn activity_claim_retry_extension_and_completion_bind_exact_attempt() { - let mut connection = connection(); - let definition = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let support = ActivitySupport { - activity_type: "email".into(), - definition_digest: definition.digest(), - }; - let transaction = connection.transaction().unwrap(); - let (run_id, _, _) = applied( - workflow_start(&transaction, &source_target(), 10, &start(5), &definition).unwrap(), - ); - let mut tokens = Tokens(0); - assert!( - workflow_claim_activities( - &transaction, - 10, - 1, - 5_000, - &[ActivitySupport { - activity_type: "email".into(), - definition_digest: Digest::from_bytes([99; 32]), - }], - &mut tokens, - ) - .unwrap() - .is_empty() - ); - let first = workflow_claim_activities( - &transaction, - 10, - 1, - 5_000, - std::slice::from_ref(&support), - &mut tokens, - ) - .unwrap() - .remove(0); - assert_eq!(first.attempt, 1); - assert!( - workflow_validate_activity_claim(&transaction, 11, std::slice::from_ref(&first)).unwrap() - ); - assert!( - !workflow_validate_activity_claim(&transaction, 4_011, std::slice::from_ref(&first)) - .unwrap() - ); - assert_eq!( - workflow_extend_activity(&transaction, 100, &first, 10_000).unwrap(), - ActivityLeaseOutcome::Extended { - lease_until_ms: 10_100, - } - ); - - let failed = ActivityCompletion { - run_id, - activity_id: first.activity_id, - attempt: first.attempt, - lease_token: first.token, - completion_token: [7; 16], - result: b"transient".to_vec(), - failed: true, - retryable: true, - }; - assert_eq!( - workflow_complete_activity(&transaction, &source_target(), 101, &failed, &definition) - .unwrap(), - ActivityCompletionOutcome::Retrying { due_at_ms: 301 } - ); - assert_eq!( - workflow_complete_activity(&transaction, &source_target(), 102, &failed, &definition) - .unwrap(), - ActivityCompletionOutcome::Duplicate { - result: b"transient".to_vec(), - } - ); - let mut conflict = failed.clone(); - conflict.result = b"different".to_vec(); - assert_eq!( - workflow_complete_activity(&transaction, &source_target(), 102, &conflict, &definition) - .unwrap(), - ActivityCompletionOutcome::IdentityConflict - ); - - let second = workflow_claim_activities(&transaction, 301, 1, 5_000, &[support], &mut tokens) - .unwrap() - .remove(0); - assert_eq!(second.attempt, 2); - assert_eq!( - workflow_complete_activity(&transaction, &source_target(), 302, &failed, &definition) - .unwrap(), - ActivityCompletionOutcome::LeaseLost - ); - let completed = ActivityCompletion { - run_id, - activity_id: second.activity_id, - attempt: second.attempt, - lease_token: second.token, - completion_token: [8; 16], - result: b"sent".to_vec(), - failed: false, - retryable: false, - }; - assert_eq!( - workflow_complete_activity(&transaction, &source_target(), 302, &completed, &definition) - .unwrap(), - ActivityCompletionOutcome::Applied(WorkflowOutcome::Applied { - run_id, - status: WorkflowStatus::Running, - event_sequence: 2, - }) - ); - transaction.commit().unwrap(); -} -#[test] -fn terminal_cleanup_removes_children_only_after_retention() { - const RETENTION_MS: i64 = 30 * 24 * 60 * 60 * 1_000; - let mut connection = connection(); - let definition = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let transaction = connection.transaction().unwrap(); - let (run_id, _, _) = applied( - workflow_start(&transaction, &source_target(), 10, &start(5), &definition).unwrap(), - ); - let timer_id: [u8; 16] = transaction - .query_row("SELECT timer_id FROM workflow_timers", [], |row| { - row.get::<_, Vec>(0) - }) - .unwrap() - .try_into() - .unwrap(); - applied( - workflow_fire_timer( - &transaction, - &source_target(), - 20, - run_id, - timer_id, - &definition, - ) - .unwrap(), - ); - assert_eq!( - workflow_cleanup_terminal(&transaction, RETENTION_MS + 19).unwrap(), - 0 - ); - assert_eq!( - workflow_cleanup_terminal(&transaction, RETENTION_MS + 20).unwrap(), - 1 - ); - let rows: (i64, i64, i64, i64) = transaction - .query_row( - "SELECT (SELECT count(*) FROM workflow_runs), (SELECT count(*) FROM workflow_events), (SELECT count(*) FROM workflow_activities), (SELECT count(*) FROM workflow_timers)", - [], - |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?, row.get(3)?)), - ) - .unwrap(); - assert_eq!(rows, (0, 0, 0, 0)); - transaction.commit().unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/workflow/control.rs b/crates/crab-cell-runtime/tests/primitives/workflow/control.rs deleted file mode 100644 index de38bbfe7..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow/control.rs +++ /dev/null @@ -1,288 +0,0 @@ -//! Operator pause and restart control paths. - -use super::*; - -#[test] -fn operator_pause_waits_for_a_live_lease_and_freezes_scheduled_work() { - let mut connection = connection(); - let definition = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let support = ActivitySupport { - activity_type: "email".into(), - definition_digest: definition.digest(), - }; - let transaction = connection.transaction().unwrap(); - let (run_id, status, _) = applied( - workflow_start(&transaction, &source_target(), 10, &start(5), &definition).unwrap(), - ); - assert_eq!(status, WorkflowStatus::Running); - let timer_id: [u8; 16] = transaction - .query_row("SELECT timer_id FROM workflow_timers", [], |row| { - row.get::<_, Vec>(0) - }) - .unwrap() - .try_into() - .unwrap(); - let control = |now_ms: i64, action: WorkflowControlAction| { - workflow_control( - &transaction, - &source_target(), - now_ms, - &WorkflowControl { - workflow_id: b"build-42".to_vec(), - run_id, - action, - }, - &definition, - ) - .unwrap() - }; - - // A live activity lease blocks the operator pause. - let mut tokens = Tokens(0); - let claimed = workflow_claim_activities( - &transaction, - 10, - 1, - 5_000, - std::slice::from_ref(&support), - &mut tokens, - ) - .unwrap(); - assert_eq!( - claimed.len(), - 1, - "the start decision schedules one claimable activity" - ); - assert_eq!( - control(20, WorkflowControlAction::Pause), - WorkflowOutcome::Busy - ); - - let completion = ActivityCompletion { - run_id, - activity_id: claimed[0].activity_id, - attempt: claimed[0].attempt, - lease_token: claimed[0].token, - completion_token: [8; 16], - result: b"sent".to_vec(), - failed: false, - retryable: false, - }; - assert!(matches!( - workflow_complete_activity(&transaction, &source_target(), 30, &completion, &definition) - .unwrap(), - ActivityCompletionOutcome::Applied(WorkflowOutcome::Applied { .. }) - )); - // The completion is an event of its own, so the pause must preserve the - // sequence the run has now, not the one it started with. - let sequence_before_pause: u64 = transaction - .query_row("SELECT event_sequence FROM workflow_runs", [], |row| { - row.get::<_, i64>(0) - }) - .unwrap() - .try_into() - .unwrap(); - assert_eq!( - control(40, WorkflowControlAction::Pause), - WorkflowOutcome::Applied { - run_id, - status: WorkflowStatus::Paused, - event_sequence: sequence_before_pause, - }, - "pause keeps the history boundary" - ); - - // Durable work stays queued while paused, but neither class may run it. - assert_eq!( - workflow_fire_timer( - &transaction, - &source_target(), - 41, - run_id, - timer_id, - &definition - ) - .unwrap(), - WorkflowOutcome::NotRunning - ); - assert!( - workflow_claim_activities( - &transaction, - 41, - 1, - 5_000, - std::slice::from_ref(&support), - &mut tokens, - ) - .unwrap() - .is_empty() - ); - let state_while_paused: (i64, i64) = transaction - .query_row( - "SELECT (SELECT state FROM workflow_activities), (SELECT state FROM workflow_timers)", - [], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .unwrap(); - assert_eq!( - state_while_paused, - (2, 0), - "the completed activity stays completed and the due timer stays unstarted while paused" - ); - - assert_eq!( - control(42, WorkflowControlAction::Resume), - WorkflowOutcome::Applied { - run_id, - status: WorkflowStatus::Running, - event_sequence: sequence_before_pause, - }, - "resume returns the same run without synthesizing an event" - ); - assert_eq!( - applied( - workflow_fire_timer( - &transaction, - &source_target(), - 43, - run_id, - timer_id, - &definition - ) - .unwrap() - ), - (run_id, WorkflowStatus::Completed, sequence_before_pause + 1), - "the timer that came due while paused fires after resume" - ); - transaction.commit().unwrap(); -} -#[test] -fn operator_restart_clears_terminal_history_and_uses_the_current_definition() { - let mut connection = connection(); - let original = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let current = Definition { - digest: Digest::from_bytes([7; 32]), - }; - let transaction = connection.transaction().unwrap(); - let (run_id, _, _) = - applied(workflow_start(&transaction, &source_target(), 10, &start(5), &original).unwrap()); - let restart = |now_ms: i64, run_id: [u8; 16], definition: &Definition| { - workflow_control( - &transaction, - &source_target(), - now_ms, - &WorkflowControl { - workflow_id: b"build-42".to_vec(), - run_id, - action: WorkflowControlAction::Restart { - request_id: RequestId::from_bytes([9; 16]), - event: b"start".to_vec(), - }, - }, - definition, - ) - .unwrap() - }; - - // A live or paused run is not restartable. - assert_eq!(restart(10, run_id, ¤t), WorkflowOutcome::Busy); - assert!(matches!( - workflow_control( - &transaction, - &source_target(), - 12, - &WorkflowControl { - workflow_id: b"build-42".to_vec(), - run_id, - action: WorkflowControlAction::Pause, - }, - &original, - ) - .unwrap(), - WorkflowOutcome::Applied { - status: WorkflowStatus::Paused, - .. - } - )); - assert_eq!(restart(10, run_id, ¤t), WorkflowOutcome::Busy); - - assert!(matches!( - workflow_cancel( - &transaction, - 14, - &WorkflowSignal { - workflow_id: b"build-42".to_vec(), - run_id, - signal_id: [8; 16], - event: b"cancel".to_vec(), - }, - ) - .unwrap(), - WorkflowOutcome::Applied { - status: WorkflowStatus::Cancelled, - .. - } - )); - - let (restarted, status, sequence) = applied(restart(10, run_id, ¤t)); - assert_ne!(restarted, run_id, "restart starts a new run identity"); - assert_eq!( - (status, sequence), - (WorkflowStatus::Running, 1), - "the restarted run starts from its first event" - ); - let rows: (i64, i64, i64, i64) = transaction - .query_row( - "SELECT (SELECT count(*) FROM workflow_runs), \ - (SELECT count(*) FROM workflow_events), \ - (SELECT count(*) FROM workflow_activities), \ - (SELECT count(*) FROM workflow_timers)", - [], - |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?, row.get(3)?)), - ) - .unwrap(); - assert_eq!( - rows, - (1, 1, 1, 1), - "terminal history is deleted and the start decision schedules fresh work" - ); - let digest: Vec = transaction - .query_row("SELECT definition_digest FROM workflow_runs", [], |row| { - row.get(0) - }) - .unwrap(); - assert_eq!( - digest, - current.digest().as_bytes().to_vec(), - "the restarted run pins the definition the operator passed" - ); - // The same request may not own two runs: cancel the restarted run first so - // the identity rule, not the busy rule, decides the answer. - let cancelled = workflow_cancel( - &transaction, - 10, - &WorkflowSignal { - workflow_id: b"build-42".to_vec(), - run_id: restarted, - // The restart's request id may not double as a signal id: the two - // identity spaces derive from the same run identity. - signal_id: [10; 16], - event: b"cancel".to_vec(), - }, - ) - .unwrap(); - assert!( - matches!(cancelled, WorkflowOutcome::Applied { .. }), - "cancelling the restarted run returned {cancelled:?}" - ); - assert_eq!( - restart(10, restarted, ¤t), - WorkflowOutcome::IdentityConflict, - "the same request cannot restart the run it already started" - ); - transaction.commit().unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/workflow/decisions.rs b/crates/crab-cell-runtime/tests/primitives/workflow/decisions.rs deleted file mode 100644 index 206b2048a..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow/decisions.rs +++ /dev/null @@ -1,242 +0,0 @@ -//! Start, signal, terminal, effect, and cancellation decisions. - -use super::*; - -#[test] -fn workflow_event_counter_tracks_history_and_reports_drift() { - let mut connection = connection(); - let source = source_target(); - let definition = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let transaction = connection.transaction().unwrap(); - workflow_start(&transaction, &source, 10, &start(5), &definition).unwrap(); - crab_cell_runtime::primitives::workflow::verify_workflow_event_count(&transaction).unwrap(); - transaction - .execute( - "UPDATE workflow_control SET event_count = event_count + 1 WHERE singleton = 1", - [], - ) - .unwrap(); - assert!( - crab_cell_runtime::primitives::workflow::verify_workflow_event_count(&transaction).is_err() - ); -} -#[test] -fn start_allocates_stable_actions_and_signal_identity_is_conflict_safe() { - let mut connection = connection(); - let source = source_target(); - let definition = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let transaction = connection.transaction().unwrap(); - let (run_id, status, sequence) = - applied(workflow_start(&transaction, &source, 10, &start(5), &definition).unwrap()); - assert_eq!((status, sequence), (WorkflowStatus::Running, 1)); - let state: Vec = transaction - .query_row( - "SELECT state FROM workflow_runs WHERE run_id = ?1", - [run_id.as_slice()], - |row| row.get(0), - ) - .unwrap(); - let timer_id: Vec = transaction - .query_row( - "SELECT timer_id FROM workflow_timers WHERE run_id = ?1", - [run_id.as_slice()], - |row| row.get(0), - ) - .unwrap(); - let activity_id: Vec = transaction - .query_row( - "SELECT activity_id FROM workflow_activities WHERE run_id = ?1", - [run_id.as_slice()], - |row| row.get(0), - ) - .unwrap(); - assert_eq!(state, [timer_id, activity_id].concat()); - transaction.commit().unwrap(); - - let signal = WorkflowSignal { - workflow_id: b"build-42".to_vec(), - run_id, - signal_id: [6; 16], - event: b"continue".to_vec(), - }; - let transaction = connection.transaction().unwrap(); - let wrong = Definition { - digest: Digest::from_bytes([9; 32]), - }; - assert!(workflow_signal(&transaction, &source, 11, &signal, &wrong).is_err()); - assert_eq!( - applied(workflow_signal(&transaction, &source, 11, &signal, &definition).unwrap()), - (run_id, WorkflowStatus::Running, 2) - ); - assert_eq!( - workflow_signal(&transaction, &source, 12, &signal, &definition).unwrap(), - WorkflowOutcome::Duplicate { - run_id, - status: WorkflowStatus::Running, - event_sequence: 2, - } - ); - let mut changed = signal.clone(); - changed.event = b"changed".to_vec(); - assert_eq!( - workflow_signal(&transaction, &source, 12, &changed, &definition).unwrap(), - WorkflowOutcome::IdentityConflict - ); - transaction.commit().unwrap(); -} -#[test] -fn invalid_decision_is_rejected_before_any_workflow_rows_are_written() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - assert!( - workflow_start( - &transaction, - &source_target(), - 10, - &start(5), - &InvalidDefinition, - ) - .is_err() - ); - let rows: (i64, i64, i64) = transaction - .query_row( - "SELECT (SELECT count(*) FROM workflow_runs), (SELECT count(*) FROM workflow_events), (SELECT count(*) FROM workflow_timers)", - [], - |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), - ) - .unwrap(); - assert_eq!(rows, (0, 0, 0)); - transaction.commit().unwrap(); -} -#[test] -fn terminal_transition_inserts_effect_with_cell_command_identity() { - let mut connection = connection(); - let source = source_target(); - let definition = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let transaction = connection.transaction().unwrap(); - let (run_id, _, _) = - applied(workflow_start(&transaction, &source, 10, &start(5), &definition).unwrap()); - let outcome = workflow_signal( - &transaction, - &source, - 11, - &WorkflowSignal { - workflow_id: b"build-42".to_vec(), - run_id, - signal_id: [7; 16], - event: b"finish-with-effect".to_vec(), - }, - &definition, - ) - .unwrap(); - assert_eq!(applied(outcome), (run_id, WorkflowStatus::Completed, 2)); - let stored: (Vec, Vec, i64) = transaction - .query_row( - "SELECT effect_id, operation, expires_at_ms FROM sys_effects", - [], - |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), - ) - .unwrap(); - assert_eq!( - stored.0, - crab_cell_runtime::primitives::effects::effect_id( - source.cell_id(), - IncarnationId::from_bytes([2; 16]), - 1, - 0, - ) - .to_vec() - ); - assert_eq!(stored.2, 20_000); - let request = peer_wire::EffectRequest::decode(stored.1.as_slice()).unwrap(); - assert!(request.destination_incarnation.is_empty()); - let command = match request.operation { - Some(peer_wire::effect_request::Operation::CellCommand(command)) => command, - None => panic!("workflow effect must contain a typed Cell command"), - }; - assert_eq!((command.command_id, command.codec_version), (7, 1)); - assert_eq!(command.input, b"canonical-destination-command"); - transaction.commit().unwrap(); -} -#[test] -fn effect_transition_rejects_undeclared_or_cross_tenant_targets_before_writes() { - for event in [ - b"finish-with-undeclared-effect".as_slice(), - b"finish-with-cross-tenant-effect".as_slice(), - ] { - let mut connection = connection(); - let source = source_target(); - let definition = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let transaction = connection.transaction().unwrap(); - let (run_id, _, _) = - applied(workflow_start(&transaction, &source, 10, &start(5), &definition).unwrap()); - let result = workflow_signal( - &transaction, - &source, - 11, - &WorkflowSignal { - workflow_id: b"build-42".to_vec(), - run_id, - signal_id: [7; 16], - event: event.to_vec(), - }, - &definition, - ); - assert!(result.is_err()); - let unchanged: (i64, i64, i64) = transaction - .query_row( - "SELECT status, (SELECT count(*) FROM workflow_events), (SELECT count(*) FROM sys_effects) FROM workflow_runs", - [], - |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), - ) - .unwrap(); - assert_eq!(unchanged, (0, 1, 0)); - transaction.commit().unwrap(); - } -} -#[test] -fn cancellation_is_idempotent_and_clears_all_outstanding_work() { - let mut connection = connection(); - let definition = Definition { - digest: Digest::from_bytes([4; 32]), - }; - let transaction = connection.transaction().unwrap(); - let (run_id, _, _) = applied( - workflow_start(&transaction, &source_target(), 10, &start(5), &definition).unwrap(), - ); - let cancellation = WorkflowSignal { - workflow_id: b"build-42".to_vec(), - run_id, - signal_id: [7; 16], - event: b"cancelled by user".to_vec(), - }; - assert_eq!( - applied(workflow_cancel(&transaction, 11, &cancellation).unwrap()), - (run_id, WorkflowStatus::Cancelled, 2) - ); - assert_eq!( - workflow_cancel(&transaction, 12, &cancellation).unwrap(), - WorkflowOutcome::Duplicate { - run_id, - status: WorkflowStatus::Cancelled, - event_sequence: 2, - } - ); - let states: (i64, i64) = transaction - .query_row( - "SELECT (SELECT state FROM workflow_activities), (SELECT state FROM workflow_timers)", - [], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .unwrap(); - assert_eq!(states, (4, 2)); - transaction.commit().unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/workflow/recovery.rs b/crates/crab-cell-runtime/tests/primitives/workflow/recovery.rs deleted file mode 100644 index 10fff1a61..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow/recovery.rs +++ /dev/null @@ -1,253 +0,0 @@ -//! Exact-root restore of a published workflow on a new owner. - -use super::*; - -#[tokio::test] -async fn published_workflow_restores_from_exact_root_on_a_new_owner() { - let namespace = NamespaceId::from_bytes([6; 16]); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([3; 16]), - namespace, - &0_u32.to_be_bytes(), - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([2; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store, Path::from("runtime"), [3; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Workflow, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let first_session = SessionId::from_bytes([4; 16]); - let observed = authority - .create_initial( - &proof, - incarnation, - Owner { - session: first_session, - endpoint: "https://first.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - first_session, - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - observed, - directory.path().join("first.sqlite"), - install_workflow_schema, - ) - .await - .unwrap(); - let definition = Definition { - digest: Digest::from_bytes([5; 32]), - }; - let request = start(7); - let workflow_target = target.clone(); - handle - .execute( - MutationIdentity { - request_id: RequestId::from_bytes([8; 16]), - issued_at_ms: 10, - expires_at_ms: 10_000, - }, - Digest::from_bytes([9; 32]), - 10, - 128, - 128, - move |transaction| match workflow_start( - transaction, - &workflow_target, - 10, - &request, - &definition, - )? { - WorkflowOutcome::Applied { run_id, .. } => { - Ok(HandlerOutcome::Success(run_id.to_vec())) - } - _ => Ok(HandlerOutcome::Rejected(b"workflow exists".to_vec())), - }, - ) - .await - .unwrap(); - handle.drain().await.unwrap(); - - let idle = authority.load(cell).await.unwrap().unwrap(); - assert_eq!(idle.value().next_due_ms, Some(10)); - let second_session = SessionId::from_bytes([11; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - second_session, - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - replica, - authority, - idle, - directory.path().join("second.sqlite"), - Owner { - session: second_session, - endpoint: "https://second.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!( - restored - .query(64, 64, |connection| { - let count = connection.query_row( - "SELECT count(*) FROM workflow_runs WHERE status = 0 AND event_sequence = 1", - [], - |row| row.get::<_, i64>(0), - )?; - Ok(vec![count as u8]) - }) - .await - .unwrap(), - vec![1] - ); - let claimed = Arc::new(std::sync::Mutex::new(None)); - let observed = claimed.clone(); - restored - .execute( - MutationIdentity { - request_id: RequestId::from_bytes([12; 16]), - issued_at_ms: 11, - expires_at_ms: 10_000, - }, - Digest::from_bytes([13; 32]), - 11, - 128, - 512, - move |transaction| { - let mut tokens = Tokens(20); - let claim = workflow_claim_activities( - transaction, - 11, - 1, - 5_000, - &[ActivitySupport { - activity_type: "email".into(), - definition_digest: Digest::from_bytes([5; 32]), - }], - &mut tokens, - )? - .into_iter() - .next() - .ok_or(crab_cell_runtime::Error::Command( - "missing restored workflow activity", - ))?; - *observed.lock().unwrap() = Some(claim); - Ok(HandlerOutcome::Success(b"claimed".to_vec())) - }, - ) - .await - .unwrap(); - let claim = claimed.lock().unwrap().clone().unwrap(); - assert_eq!( - restored - .query(64, 64, { - let claim = claim.clone(); - move |connection| { - Ok(vec![u8::from(workflow_validate_activity_claim( - connection, - 12, - &[claim], - )?)]) - } - }) - .await - .unwrap(), - vec![1] - ); - let workflow_target = target.clone(); - restored - .execute( - MutationIdentity { - request_id: RequestId::from_bytes([14; 16]), - issued_at_ms: 13, - expires_at_ms: 10_000, - }, - Digest::from_bytes([15; 32]), - 13, - 128, - 128, - move |transaction| { - let completion = ActivityCompletion { - run_id: claim.run_id, - activity_id: claim.activity_id, - attempt: claim.attempt, - lease_token: claim.token, - completion_token: [16; 16], - result: b"sent".to_vec(), - failed: false, - retryable: false, - }; - match workflow_complete_activity( - transaction, - &workflow_target, - 13, - &completion, - &Definition { - digest: Digest::from_bytes([5; 32]), - }, - )? { - ActivityCompletionOutcome::Applied(_) => { - Ok(HandlerOutcome::Success(b"completed".to_vec())) - } - _ => Ok(HandlerOutcome::Rejected(b"completion rejected".to_vec())), - } - }, - ) - .await - .unwrap(); - assert_eq!( - restored - .query(64, 64, |connection| { - let state = - connection.query_row("SELECT state FROM workflow_activities", [], |row| { - row.get::<_, i64>(0) - })?; - let events = - connection.query_row("SELECT count(*) FROM workflow_events", [], |row| { - row.get::<_, i64>(0) - })?; - Ok(vec![state as u8, events as u8]) - }) - .await - .unwrap(), - vec![2, 2] - ); - restored.drain().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/workflow_activity_codec.rs b/crates/crab-cell-runtime/tests/primitives/workflow_activity_codec.rs deleted file mode 100644 index d0fdf36bf..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow_activity_codec.rs +++ /dev/null @@ -1,67 +0,0 @@ -use crab_cell_runtime::codec::{BoundedEncoder, CodecError, WireValue}; -use crab_cell_runtime::primitives::workflow::{ - ActivityClaim, ActivityCompletion, WorkflowActivityClaimRequest, WorkflowActivityExtendRequest, - WorkflowActivityValidateRequest, -}; -use crab_cell_runtime::primitives::workflow::{ActivityCompletionOutcome, ActivityLeaseOutcome}; - -use crate::support::fixtures::codec_roundtrip; - -#[test] -fn activity_codecs_roundtrip_claim_lease_completion_and_validation() { - let claim = ActivityClaim { - run_id: [1; 16], - activity_id: [2; 16], - activity_type: "echo".into(), - input: b"payload".to_vec(), - definition_digest: crab_cell_runtime::Digest::from_bytes([3; 32]), - attempt: 1, - token: [4; 16], - lease_until_ms: 10_000, - }; - codec_roundtrip(WorkflowActivityClaimRequest { - limit: 1, - lease_ms: 5_000, - }); - codec_roundtrip(vec![claim.clone()]); - codec_roundtrip(WorkflowActivityExtendRequest { - claim: claim.clone(), - extension_ms: 5_000, - }); - codec_roundtrip(ActivityLeaseOutcome::Extended { - lease_until_ms: 15_000, - }); - codec_roundtrip(ActivityCompletion { - run_id: claim.run_id, - activity_id: claim.activity_id, - attempt: claim.attempt, - lease_token: claim.token, - completion_token: [5; 16], - result: b"done".to_vec(), - failed: false, - retryable: false, - }); - codec_roundtrip(ActivityCompletionOutcome::Retrying { due_at_ms: 20_000 }); - codec_roundtrip(WorkflowActivityValidateRequest { - claimed: vec![claim], - }); -} - -#[test] -fn activity_decoder_rejects_invalid_claim_shape() { - let claim = ActivityClaim { - run_id: [1; 16], - activity_id: [2; 16], - activity_type: "echo".into(), - input: Vec::new(), - definition_digest: crab_cell_runtime::Digest::from_bytes([3; 32]), - attempt: 0, - token: [4; 16], - lease_until_ms: 10_000, - }; - let mut encoder = BoundedEncoder::new(256).unwrap(); - assert!(matches!( - claim.encode(&mut encoder), - Err(CodecError::Invalid("invalid activity claim")) - )); -} diff --git a/crates/crab-cell-runtime/tests/primitives/workflow_api.rs b/crates/crab-cell-runtime/tests/primitives/workflow_api.rs deleted file mode 100644 index b9221262c..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow_api.rs +++ /dev/null @@ -1,510 +0,0 @@ -use std::{ - future::Future, - io::Write, - pin::Pin, - sync::{ - Arc, - atomic::{AtomicBool, AtomicUsize, Ordering}, - }, - time::Duration, -}; - -use crab_cell_runtime::Error; -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::catalog::{CatalogEntry, CellCatalog}; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::client::{CellClient, InvocationError}; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::fleet::scheduler::DueCellScan; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::primitives::activity_pool::BlockingActivityPool; -use crab_cell_runtime::primitives::maintenance::{ - MaintenanceModule, MaintenanceTickCommand, MaintenanceTickOutcome, MaintenanceTickRequest, - register_maintenance, -}; -use crab_cell_runtime::primitives::workflow::{ - ActivityContext, ActivityExecution, ActivityHandler, ActivityRunOutcome, - BlockingActivityHandler, WorkflowAction, WorkflowActivityClaimCommand, - WorkflowActivityCompleteCommand, WorkflowActivityExtendCommand, WorkflowActivityValidateQuery, - WorkflowCancelCommand, WorkflowContext, WorkflowControlCommand, WorkflowDecision, - WorkflowDefinition, WorkflowGetQuery, WorkflowOutcome, WorkflowSignal, WorkflowSignalCommand, - WorkflowStartCommand, WorkflowStatus, install_workflow_schema, register_activity, - register_blocking_activity, register_workflow, register_workflow_activities, -}; -use crab_cell_runtime::primitives::workflow::{ - WorkflowActivityModule, WorkflowModule, WorkflowNamespace, -}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, ModuleDescriptor, NamespaceDescriptor, RegistryBuilder, -}; -use crab_cell_runtime::registry::{MigrationDescriptor, OperationDescriptor}; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use crate::support::fencing::fence_session; -use crate::support::fixtures::mutation_identity; - -const WORKFLOW_MODULE: &str = "workflow-api-test"; -const WORKFLOW_NAMESPACE: NamespaceId = NamespaceId::from_bytes([8; 16]); -const EFFECT_NAMESPACE: NamespaceId = NamespaceId::from_bytes([9; 16]); -static EFFECT_TARGETS: [NamespaceId; 1] = [EFFECT_NAMESPACE]; -static WORKFLOW_NAMESPACES: [NamespaceDescriptor; 2] = [ - NamespaceDescriptor { - id: WORKFLOW_NAMESPACE, - name: WORKFLOW_MODULE, - role: CatalogRole::Workflow, - shards: 1, - effect_targets: &EFFECT_TARGETS, - dead_letter: None, - }, - NamespaceDescriptor { - id: EFFECT_NAMESPACE, - name: "workflow-effect-target", - role: CatalogRole::Repository, - shards: 1, - effect_targets: &[], - dead_letter: None, - }, -]; -static DRIFT_NAMESPACES: [NamespaceDescriptor; 2] = [ - NamespaceDescriptor { - id: WORKFLOW_NAMESPACE, - name: WORKFLOW_MODULE, - role: CatalogRole::Workflow, - shards: 1, - effect_targets: &[], - dead_letter: None, - }, - NamespaceDescriptor { - id: EFFECT_NAMESPACE, - name: "workflow-effect-target", - role: CatalogRole::Repository, - shards: 1, - effect_targets: &[], - dead_letter: None, - }, -]; -const WORKFLOW_MIGRATION: &str = include_str!("../../src/migrations/workflow.sql"); -const DEFINITION_DIGEST: Digest = Digest::from_bytes([6; 32]); -const LEGACY_DEFINITION_DIGEST: Digest = Digest::from_bytes([7; 32]); -const COMMANDS: &[OperationDescriptor] = &[ - operation(1, 1024 * 1024, 64), - operation(2, 1024 * 1024, 64), - operation(3, 1024 * 1024, 64), - operation(4, 1024 * 1024, 1024 * 1024), - operation(5, 1024 * 1024, 1024 * 1024), - operation(6, 1024 * 1024, 64), - operation(7, 8, 5), - operation(8, 1024 * 1024, 64), -]; -const QUERIES: &[OperationDescriptor] = &[ - operation(1, 2048, 1024 * 1024), - operation(2, 1024 * 1024, 1), -]; -static HEARTBEAT_OBSERVED: AtomicBool = AtomicBool::new(false); -static FAILOVER_ACTIVITY_ENTERED: AtomicBool = AtomicBool::new(false); -static FAILOVER_ACTIVITY_BLOCKED: AtomicBool = AtomicBool::new(false); -static FAILOVER_ACTIVITY_ATTEMPTS: AtomicUsize = AtomicUsize::new(0); - -struct Definition; - -static DEFINITION: Definition = Definition; -static LEGACY_DEFINITION: LegacyDefinition = LegacyDefinition; -static DEFINITIONS: [&dyn WorkflowDefinition; 2] = [&LEGACY_DEFINITION, &DEFINITION]; - -impl WorkflowDefinition for Definition { - fn digest(&self) -> Digest { - DEFINITION_DIGEST - } - - fn effect_targets(&self) -> &'static [NamespaceId] { - &EFFECT_TARGETS - } - - fn transition( - &self, - _state: &[u8], - event: &[u8], - context: WorkflowContext, - ) -> crab_cell_runtime::Result { - if matches!( - event, - b"activity" | b"activity-retry" | b"activity-blocking" | b"activity-failover" - ) { - return Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state: b"waiting".to_vec(), - result: None, - actions: vec![WorkflowAction::Activity { - activity_type: if event == b"activity-blocking" { - "blocking-echo".into() - } else { - "echo".into() - }, - input: match event { - b"activity-retry" => b"retry".to_vec(), - b"activity-failover" => b"failover".to_vec(), - _ => b"payload".to_vec(), - }, - due_at_ms: context.now_ms(), - expires_at_ms: context.now_ms() + 60_000, - }], - }); - } - if let Some(path) = event.strip_prefix(b"publish-report\0") { - return Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state: b"report-pending".to_vec(), - result: None, - actions: vec![WorkflowAction::Activity { - activity_type: "publish-report".into(), - input: path.to_vec(), - due_at_ms: context.now_ms(), - expires_at_ms: context.now_ms() + 60_000, - }], - }); - } - if event.starts_with(b"activity\0") { - return Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: b"activity-complete".to_vec(), - result: Some(event.to_vec()), - actions: Vec::new(), - }); - } - if event.starts_with(b"timer\0") { - return Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: b"timer-complete".to_vec(), - result: Some(event.to_vec()), - actions: Vec::new(), - }); - } - if event == b"timer" { - return Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state: b"timer-waiting".to_vec(), - result: None, - actions: vec![WorkflowAction::Timer { - due_at_ms: context.now_ms(), - }], - }); - } - if event == b"finish" { - return Ok(WorkflowDecision { - status: WorkflowStatus::Completed, - state: b"done".to_vec(), - result: Some(b"finished".to_vec()), - actions: Vec::new(), - }); - } - Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state: event.to_vec(), - result: None, - actions: Vec::new(), - }) - } -} - -struct LegacyDefinition; - -impl WorkflowDefinition for LegacyDefinition { - fn digest(&self) -> Digest { - LEGACY_DEFINITION_DIGEST - } - - fn transition( - &self, - _state: &[u8], - event: &[u8], - _context: WorkflowContext, - ) -> crab_cell_runtime::Result { - let mut state = b"legacy:".to_vec(); - state.extend_from_slice(event); - Ok(WorkflowDecision { - status: WorkflowStatus::Running, - state, - result: None, - actions: Vec::new(), - }) - } -} - -struct TestWorkflow; - -impl WorkflowModule for TestWorkflow { - const MODULE: &'static str = WORKFLOW_MODULE; - const NAMESPACE: NamespaceId = WORKFLOW_NAMESPACE; - const CURRENT_DEFINITION: &'static dyn WorkflowDefinition = &DEFINITION; - const DEFINITIONS: &'static [&'static dyn WorkflowDefinition] = &DEFINITIONS; - const START_COMMAND_ID: u32 = 1; - const SIGNAL_COMMAND_ID: u32 = 2; - const CANCEL_COMMAND_ID: u32 = 3; - const CONTROL_COMMAND_ID: u32 = 8; - const GET_QUERY_ID: u32 = 1; -} - -impl WorkflowActivityModule for TestWorkflow { - const ACTIVITY_TYPES: &'static [&'static str] = &["blocking-echo", "echo", "publish-report"]; - const ACTIVITY_CLAIM_COMMAND_ID: u32 = 4; - const ACTIVITY_COMPLETE_COMMAND_ID: u32 = 5; - const ACTIVITY_EXTEND_COMMAND_ID: u32 = 6; - const ACTIVITY_VALIDATE_QUERY_ID: u32 = 2; -} - -impl MaintenanceModule for TestWorkflow { - const MODULE: &'static str = WORKFLOW_MODULE; - const TICK_COMMAND_ID: u32 = 7; - const WORKFLOW_DEFINITIONS: &'static [&'static dyn WorkflowDefinition] = &DEFINITIONS; -} - -struct EchoActivity; - -impl ActivityHandler for EchoActivity { - const TYPE: &'static str = "echo"; - - fn execute( - context: ActivityContext, - input: Vec, - ) -> Pin + Send + 'static>> { - Box::pin(async move { - if input == b"retry" { - return ActivityExecution::Failed { - details: b"temporary".to_vec(), - retryable: true, - }; - } - if input == b"failover" { - FAILOVER_ACTIVITY_ATTEMPTS.fetch_add(1, Ordering::AcqRel); - FAILOVER_ACTIVITY_ENTERED.store(true, Ordering::Release); - while FAILOVER_ACTIVITY_BLOCKED.load(Ordering::Acquire) { - tokio::time::sleep(Duration::from_millis(10)).await; - } - } - let initial_deadline = context.lease_until_ms(); - assert_ne!(context.idempotency_key(), [0; 32]); - assert_ne!(context.lease_token(), [0; 16]); - tokio::time::sleep(Duration::from_millis(1_800)).await; - HEARTBEAT_OBSERVED.store( - context.lease_until_ms() > initial_deadline - && !context.cancellation().is_cancelled(), - Ordering::Release, - ); - let mut result = input; - result.extend_from_slice(b"-complete"); - ActivityExecution::Completed(result) - }) - } -} - -struct BlockingEchoActivity; - -impl BlockingActivityHandler for BlockingEchoActivity { - const TYPE: &'static str = "blocking-echo"; - - fn execute(_context: ActivityContext, mut input: Vec) -> ActivityExecution { - input.extend_from_slice(b"-blocking"); - ActivityExecution::Completed(input) - } -} - -struct PublishReportActivity; - -impl ActivityHandler for PublishReportActivity { - const TYPE: &'static str = "publish-report"; - - fn execute( - context: ActivityContext, - input: Vec, - ) -> Pin + Send + 'static>> { - Box::pin(async move { - let key = context.idempotency_key(); - let report = format!("build report\nrequest={key:02x?}\n").into_bytes(); - let written = tokio::task::spawn_blocking(move || { - let path = std::path::PathBuf::from( - String::from_utf8(input).map_err(|error| error.to_string())?, - ); - match std::fs::OpenOptions::new() - .write(true) - .create_new(true) - .open(&path) - { - Ok(mut file) => file.write_all(&report).map_err(|error| error.to_string()), - Err(error) if error.kind() == std::io::ErrorKind::AlreadyExists => { - let existing = std::fs::read(path).map_err(|error| error.to_string())?; - if existing == report { - Ok(()) - } else { - Err("report identity conflict".into()) - } - } - Err(error) => Err(error.to_string()), - } - }) - .await; - match written { - Ok(Ok(())) if context.attempt() == 1 => ActivityExecution::Failed { - details: b"response lost after report publication".to_vec(), - retryable: true, - }, - Ok(Ok(())) => ActivityExecution::Completed(key.to_vec()), - Ok(Err(error)) => ActivityExecution::Failed { - details: error.into_bytes(), - retryable: false, - }, - Err(error) => ActivityExecution::Failed { - details: error.to_string().into_bytes(), - retryable: true, - }, - } - }) - } -} - -impl CellModule for TestWorkflow { - const NAME: &'static str = WORKFLOW_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: std::sync::OnceLock = std::sync::OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: WORKFLOW_MODULE, - source_digest: Digest::from_bytes([4; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: WORKFLOW_MIGRATION, - digest: Digest::from_bytes(*blake3::hash(WORKFLOW_MIGRATION.as_bytes()).as_bytes()), - }])), - commands: COMMANDS, - queries: QUERIES, - workflow_definitions: &[LEGACY_DEFINITION_DIGEST, DEFINITION_DIGEST], - activity_types: &["blocking-echo", "echo", "publish-report"], - namespaces: &WORKFLOW_NAMESPACES, - }) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - register_workflow::(registry)?; - register_workflow_activities::(registry)?; - register_activity::(registry)?; - register_activity::(registry)?; - register_blocking_activity::(registry)?; - register_maintenance::(registry) - } -} - -struct EffectTargetDrift; - -impl CellModule for EffectTargetDrift { - const NAME: &'static str = WORKFLOW_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - Box::leak(Box::new(ModuleDescriptor { - name: WORKFLOW_MODULE, - source_digest: Digest::from_bytes([4; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: TestWorkflow.descriptor().migrations, - commands: COMMANDS, - queries: QUERIES, - workflow_definitions: &[LEGACY_DEFINITION_DIGEST, DEFINITION_DIGEST], - activity_types: &["blocking-echo", "echo", "publish-report"], - namespaces: &DRIFT_NAMESPACES, - })) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - TestWorkflow.register(registry) - } -} - -struct MissingDefinitionBinding; - -impl CellModule for MissingDefinitionBinding { - const NAME: &'static str = WORKFLOW_MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - TestWorkflow.descriptor() - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - registry.bind_workflow_definition(WORKFLOW_MODULE, &DEFINITION)?; - registry.bind_workflow_definition(WORKFLOW_MODULE, &LEGACY_DEFINITION)?; - registry.bind_activity_inventory(WORKFLOW_MODULE, DEFINITION_DIGEST, &["echo"])?; - registry.bind_activity_inventory(WORKFLOW_MODULE, LEGACY_DEFINITION_DIGEST, &["echo"])?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_command::>()?; - registry.bind_query::>()?; - registry.bind_query::>()?; - registry.bind_command::>()?; - registry.bind_activity::(WORKFLOW_MODULE, DEFINITION_DIGEST) - } -} - -const fn operation(id: u32, input_limit: u32, output_limit: u32) -> OperationDescriptor { - OperationDescriptor { - id, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit, - output_limit, - } -} - -fn registry() -> Arc { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "workflow-api-test".into(), - cargo_lock_digest: Digest::from_bytes([5; 32]), - }); - builder.register(TestWorkflow).unwrap(); - Arc::new(builder.finish().unwrap()) -} - -#[test] -fn registry_rejects_a_declared_activity_without_its_native_binding() { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "workflow-api-test".into(), - cargo_lock_digest: Digest::from_bytes([5; 32]), - }); - builder.register(MissingDefinitionBinding).unwrap(); - assert!(matches!( - builder.finish(), - Err(Error::Registry("descriptor and activity bindings differ")) - )); -} - -#[test] -fn registry_rejects_workflow_effect_target_drift() { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "workflow-effect-target-test".into(), - cargo_lock_digest: Digest::from_bytes([5; 32]), - }); - builder.register(EffectTargetDrift).unwrap(); - assert!(matches!( - builder.finish(), - Err(Error::Registry( - "Workflow effect targets and compiled definitions differ" - )) - )); -} - -mod activity; -mod namespace; -mod retry; diff --git a/crates/crab-cell-runtime/tests/primitives/workflow_api/activity.rs b/crates/crab-cell-runtime/tests/primitives/workflow_api/activity.rs deleted file mode 100644 index 31f2d996c..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow_api/activity.rs +++ /dev/null @@ -1,384 +0,0 @@ -//! Native activity heartbeats and node-loss recovery. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn native_activity_heartbeats_and_recovers_after_node_loss() { - HEARTBEAT_OBSERVED.store(false, Ordering::Release); - FAILOVER_ACTIVITY_ENTERED.store(false, Ordering::Release); - FAILOVER_ACTIVITY_BLOCKED.store(true, Ordering::Release); - FAILOVER_ACTIVITY_ATTEMPTS.store(0, Ordering::Release); - let registry = registry(); - assert!(registry.has_blocking_activities()); - assert!(registry.requires_blocking_activity(WORKFLOW_NAMESPACE)); - assert_eq!( - registry.internal_command_action(WORKFLOW_NAMESPACE, 7, 1), - Some("cell.scheduler.tick") - ); - assert_eq!( - registry.internal_command_action(WORKFLOW_NAMESPACE, 4, 1), - Some("cell.activity.source") - ); - assert_eq!( - registry.internal_query_action(WORKFLOW_NAMESPACE, 2, 1), - Some("cell.activity.source") - ); - assert_eq!( - registry.internal_command_action(WORKFLOW_NAMESPACE, 1, 1), - None - ); - let target = CellTarget::new( - TenantId::from_bytes([21; 16]), - ApplicationId::from_bytes([22; 16]), - WORKFLOW_NAMESPACE, - &0_u32.to_be_bytes(), - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([23; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store, Path::from("activity-runtime"), [22; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Workflow, - registry.module_code(WORKFLOW_MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let first_session = SessionId::from_bytes([25; 16]); - let control = authority - .create_initial( - &proof, - incarnation, - Owner { - session: first_session, - endpoint: "https://activity-first.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - first_session, - ) - .unwrap(); - let first_database = directory.path().join("activity-first.sqlite"); - let handle = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - control, - first_database.clone(), - install_workflow_schema, - ) - .await - .unwrap(); - let client = CellClient::local(registry.clone(), handle.clone()); - let workflows = WorkflowNamespace::::new( - client.clone(), - target.tenant(), - target.application(), - ) - .unwrap(); - workflows - .start( - mutation_identity(26), - b"activity-build".to_vec(), - b"activity".to_vec(), - ) - .await - .unwrap(); - assert!(registry.has_activity_runner(target.namespace())); - let blocking_pool = BlockingActivityPool::new(1).unwrap(); - assert!(matches!( - registry - .run_activity_once(client.clone(), &target, 5_000, None) - .await, - Err( - crab_cell_runtime::primitives::workflow::ActivitySupervisorError::Runtime( - Error::Capacity("blocking activity slot was not reserved") - ) - ) - )); - let completed = registry - .run_activity_once(client, &target, 5_000, blocking_pool.try_reserve().unwrap()) - .await - .unwrap(); - let ActivityRunOutcome::Completed { workflow, receipt } = completed else { - panic!("native activity was not completed"); - }; - assert!(matches!( - workflow, - WorkflowOutcome::Applied { - status: WorkflowStatus::Completed, - event_sequence: 2, - .. - } - )); - assert!(HEARTBEAT_OBSERVED.load(Ordering::Acquire)); - let state = workflows - .state(b"activity-build".to_vec(), Some(receipt)) - .await - .unwrap() - .output - .unwrap(); - assert_eq!(state.state, b"activity-complete"); - assert!(state.result.unwrap().ends_with(b"payload-complete")); - workflows - .start( - mutation_identity(28), - b"retry-build".to_vec(), - b"activity-retry".to_vec(), - ) - .await - .unwrap(); - assert!(matches!( - registry - .run_activity_once( - CellClient::local(registry.clone(), handle.clone()), - &target, - 5_000, - blocking_pool.try_reserve().unwrap(), - ) - .await - .unwrap(), - ActivityRunOutcome::Retrying { .. } - )); - workflows - .start( - mutation_identity(29), - b"blocking-build".to_vec(), - b"activity-blocking".to_vec(), - ) - .await - .unwrap(); - let blocking = blocking_pool.try_reserve().unwrap().unwrap(); - assert!(matches!( - registry - .run_activity_once( - CellClient::local(registry.clone(), handle.clone()), - &target, - 5_000, - Some(blocking), - ) - .await - .unwrap(), - ActivityRunOutcome::Completed { .. } - )); - assert!( - workflows - .state(b"blocking-build".to_vec(), None) - .await - .unwrap() - .output - .unwrap() - .result - .unwrap() - .ends_with(b"payload-blocking") - ); - workflows - .start( - mutation_identity(30), - b"failover-build".to_vec(), - b"activity-failover".to_vec(), - ) - .await - .unwrap(); - let failover_registry = registry.clone(); - let failover_client = CellClient::local(registry.clone(), handle.clone()); - let failover_target = target.clone(); - let failover_pool = blocking_pool.clone(); - let first_attempt = tokio::spawn(async move { - // The retry activity may become due before the newly started failover activity. - // Consume that retry first so this attempt deterministically owns the failover lease. - loop { - let reservation = loop { - if let Some(reservation) = failover_pool.try_reserve().unwrap() { - break reservation; - } - tokio::task::yield_now().await; - }; - let outcome = match failover_registry - .run_activity_once( - failover_client.clone(), - &failover_target, - 5_000, - Some(reservation), - ) - .await - { - Ok(outcome) => outcome, - Err(error) => { - return Err::< - ActivityRunOutcome, - crab_cell_runtime::primitives::workflow::ActivitySupervisorError, - >(error); - } - }; - if matches!( - outcome, - ActivityRunOutcome::Retrying { .. } | ActivityRunOutcome::Idle { .. } - ) { - continue; - } - return Ok(outcome); - } - }); - // The activity claim crosses the SQL worker and the node-owned callback - // boundary. Allow one lease interval for a busy multi-crate test runner; - // the short sleep keeps this readiness wait from monopolizing the runtime. - tokio::time::timeout(Duration::from_secs(5), async { - while !FAILOVER_ACTIVITY_ENTERED.load(Ordering::Acquire) { - tokio::time::sleep(Duration::from_millis(1)).await; - } - }) - .await - .unwrap(); - first_attempt.abort(); - assert!(first_attempt.await.unwrap_err().is_cancelled()); - assert_eq!(FAILOVER_ACTIVITY_ATTEMPTS.load(Ordering::Acquire), 1); - blocking_pool.shutdown().await.unwrap(); - drop(workflows); - drop(handle); - drop(runtime); - tokio::time::timeout(Duration::from_secs(2), async { - loop { - match std::fs::remove_file(&first_database) { - Ok(()) => break, - Err(error) if error.kind() == std::io::ErrorKind::NotFound => break, - Err(_) => tokio::time::sleep(Duration::from_millis(10)).await, - } - } - }) - .await - .expect("dropped node did not release its local SQLite file"); - - let stale_owner = authority.load(cell).await.unwrap().unwrap(); - assert_eq!( - stale_owner.value().owner.as_ref().unwrap().session, - first_session - ); - assert!(stale_owner.value().root.is_some()); - let fenced = fence_session(&layout, first_session, SessionId::from_bytes([27; 16])).await; - let takeover = fenced.direct_takeover().unwrap(); - let second_session = SessionId::from_bytes([27; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - second_session, - ) - .unwrap(); - let restored = runtime - .takeover_restored( - proof, - replica, - authority.clone(), - stale_owner, - takeover, - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - layout.clone(), - Limits::default(), - ), - directory.path().join("activity-second.sqlite"), - Owner { - session: second_session, - endpoint: "https://activity-second.internal:8081".into(), - }, - ) - .await - .unwrap(); - let restored_workflows = WorkflowNamespace::::new( - CellClient::local(registry.clone(), restored.clone()), - target.tenant(), - target.application(), - ) - .unwrap(); - assert_eq!( - restored_workflows - .state(b"activity-build".to_vec(), Some(receipt)) - .await - .unwrap() - .output - .unwrap() - .state, - b"activity-complete" - ); - // Node-session fencing makes Cell takeover immediate, but activity leases - // remain valid until their published deadline. - tokio::time::sleep(Duration::from_millis(5_100)).await; - let current = authority.load(cell).await.unwrap().unwrap(); - let current_sequence = current.value().root.as_ref().unwrap().commit_sequence; - let restored_client = CellClient::local(registry.clone(), restored.clone()); - let reclaimed = registry - .run_maintenance_once( - restored_client.clone(), - target.clone(), - mutation_identity(31), - MaintenanceTickRequest { - expected_commit_sequence: current_sequence, - }, - ) - .await - .unwrap(); - assert_eq!( - reclaimed.output, - MaintenanceTickOutcome::Applied { processed: 1 } - ); - FAILOVER_ACTIVITY_BLOCKED.store(false, Ordering::Release); - let restored_blocking_pool = BlockingActivityPool::new(1).unwrap(); - let older_retry = registry - .run_activity_once( - restored_client.clone(), - &target, - 5_000, - restored_blocking_pool.try_reserve().unwrap(), - ) - .await - .unwrap(); - assert!(matches!(older_retry, ActivityRunOutcome::Retrying { .. })); - let failover = registry - .run_activity_once( - restored_client, - &target, - 5_000, - restored_blocking_pool.try_reserve().unwrap(), - ) - .await - .unwrap(); - assert!( - matches!(failover, ActivityRunOutcome::Completed { .. }), - "unexpected failover outcome: {failover:?}" - ); - assert_eq!(FAILOVER_ACTIVITY_ATTEMPTS.load(Ordering::Acquire), 2); - assert!( - restored_workflows - .state(b"failover-build".to_vec(), None) - .await - .unwrap() - .output - .unwrap() - .result - .unwrap() - .ends_with(b"failover-complete") - ); - restored_blocking_pool.shutdown().await.unwrap(); - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/workflow_api/namespace.rs b/crates/crab-cell-runtime/tests/primitives/workflow_api/namespace.rs deleted file mode 100644 index 761aeb5b5..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow_api/namespace.rs +++ /dev/null @@ -1,259 +0,0 @@ -//! Typed workflow namespace publication, rejection, reads, and restore. - -use super::*; - -#[tokio::test] -async fn typed_workflow_namespace_publishes_rejects_reads_and_survives_restore() { - let registry = registry(); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([3; 16]), - WORKFLOW_NAMESPACE, - &0_u32.to_be_bytes(), - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([2; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store, Path::from("runtime"), [3; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Workflow, - registry.module_code(WORKFLOW_MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let first_session = SessionId::from_bytes([4; 16]); - let control = authority - .create_initial( - &proof, - incarnation, - Owner { - session: first_session, - endpoint: "https://first.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - first_session, - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - control, - directory.path().join("first.sqlite"), - install_workflow_schema, - ) - .await - .unwrap(); - let client = CellClient::local(registry.clone(), handle.clone()); - let workflows = WorkflowNamespace::::new( - client.clone(), - target.tenant(), - target.application(), - ) - .unwrap(); - let started = workflows - .start( - mutation_identity(7), - b"build-42".to_vec(), - b"start".to_vec(), - ) - .await - .unwrap(); - let WorkflowOutcome::Applied { run_id, .. } = started.output else { - panic!("workflow start was not applied"); - }; - let signal = WorkflowSignal { - workflow_id: b"build-42".to_vec(), - run_id, - signal_id: [8; 16], - event: b"continue".to_vec(), - }; - let signalled = workflows - .signal(mutation_identity(9), signal.clone()) - .await - .unwrap(); - assert!(matches!( - signalled.output, - WorkflowOutcome::Applied { - event_sequence: 2, - .. - } - )); - assert!(matches!( - workflows - .signal(mutation_identity(10), signal.clone()) - .await - .unwrap() - .output, - WorkflowOutcome::Duplicate { - event_sequence: 2, - .. - } - )); - let mut conflicting = signal; - conflicting.event = b"different".to_vec(); - let conflict = workflows.signal(mutation_identity(11), conflicting).await; - assert!(matches!( - conflict, - Err(InvocationError::Rejected(outcome)) - if outcome.output == WorkflowOutcome::IdentityConflict - )); - let state = workflows - .state(b"build-42".to_vec(), Some(signalled.receipt)) - .await - .unwrap() - .output - .unwrap(); - assert_eq!(state.state, b"continue"); - assert_eq!(state.event_sequence, 2); - let timer = workflows - .start( - mutation_identity(16), - b"timer-build".to_vec(), - b"timer".to_vec(), - ) - .await - .unwrap(); - let mut due = DueCellScan::new(&catalog, authority.clone(), target.cell_id().as_bytes()[0]) - .await - .unwrap(); - let due = due.next_batch(i64::MAX).await.unwrap().unwrap(); - assert_eq!(due.len(), 1); - assert_eq!( - due[0] - .control() - .value() - .root - .as_ref() - .unwrap() - .commit_sequence, - timer.receipt.commit_sequence - ); - let routed = runtime - .local_handle(due[0].catalog().clone(), due[0].control()) - .await - .unwrap() - .unwrap(); - let scheduler_client = CellClient::local(registry.clone(), routed); - let tick = registry - .run_maintenance_once( - scheduler_client.clone(), - target.clone(), - mutation_identity(17), - MaintenanceTickRequest { - expected_commit_sequence: timer.receipt.commit_sequence, - }, - ) - .await - .unwrap(); - assert_eq!( - tick.output, - MaintenanceTickOutcome::Applied { processed: 1 } - ); - assert_eq!( - workflows - .state(b"timer-build".to_vec(), Some(tick.receipt)) - .await - .unwrap() - .output - .unwrap() - .state, - b"timer-complete" - ); - let stale = registry - .run_maintenance_once( - scheduler_client, - target.clone(), - mutation_identity(18), - MaintenanceTickRequest { - expected_commit_sequence: timer.receipt.commit_sequence, - }, - ) - .await - .unwrap(); - assert_eq!(stale.output, MaintenanceTickOutcome::Stale); - handle.drain().await.unwrap(); - - let idle = authority.load(cell).await.unwrap().unwrap(); - let second_session = SessionId::from_bytes([12; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - second_session, - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - replica, - authority, - idle, - directory.path().join("second.sqlite"), - Owner { - session: second_session, - endpoint: "https://second.internal:8081".into(), - }, - ) - .await - .unwrap(); - let restored_workflows = WorkflowNamespace::::new( - CellClient::local(registry, restored.clone()), - target.tenant(), - target.application(), - ) - .unwrap(); - assert_eq!( - restored_workflows - .state(b"build-42".to_vec(), Some(signalled.receipt)) - .await - .unwrap() - .output - .unwrap() - .state, - b"continue" - ); - let cancelled = restored_workflows - .cancel( - mutation_identity(13), - WorkflowSignal { - workflow_id: b"build-42".to_vec(), - run_id, - signal_id: [14; 16], - event: b"cancel".to_vec(), - }, - ) - .await - .unwrap(); - assert!(matches!( - cancelled.output, - WorkflowOutcome::Applied { - status: WorkflowStatus::Cancelled, - event_sequence: 3, - .. - } - )); - restored.drain().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/primitives/workflow_api/retry.rs b/crates/crab-cell-runtime/tests/primitives/workflow_api/retry.rs deleted file mode 100644 index ff70fb6cf..000000000 --- a/crates/crab-cell-runtime/tests/primitives/workflow_api/retry.rs +++ /dev/null @@ -1,132 +0,0 @@ -//! Published-report retry after a lost response. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn published_report_survives_a_lost_response_and_completes_on_retry() { - let registry = registry(); - let target = CellTarget::new( - TenantId::from_bytes([51; 16]), - ApplicationId::from_bytes([52; 16]), - WORKFLOW_NAMESPACE, - &0_u32.to_be_bytes(), - ) - .unwrap(); - let incarnation = IncarnationId::from_bytes([53; 16]); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("report-publication"), - [52; 16], - ); - let replica = CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Workflow, - registry.module_code(WORKFLOW_MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout); - let session = SessionId::from_bytes([54; 16]); - let control = authority - .create_initial( - &proof, - incarnation, - Owner { - session, - endpoint: "https://report-worker.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let report_path = directory.path().join("build-report.txt"); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - session, - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof, - replica, - authority, - control, - directory.path().join("workflow.sqlite"), - install_workflow_schema, - ) - .await - .unwrap(); - let client = CellClient::local(registry.clone(), handle.clone()); - let workflows = WorkflowNamespace::::new( - client.clone(), - target.tenant(), - target.application(), - ) - .unwrap(); - let mut event = b"publish-report\0".to_vec(); - event.extend_from_slice(report_path.to_str().unwrap().as_bytes()); - workflows - .start(mutation_identity(55), b"build-report".to_vec(), event) - .await - .unwrap(); - let pool = BlockingActivityPool::new(1).unwrap(); - let first = registry - .run_activity_once(client.clone(), &target, 5_000, pool.try_reserve().unwrap()) - .await - .unwrap(); - assert!(matches!(first, ActivityRunOutcome::Retrying { .. })); - let published = std::fs::read(&report_path).unwrap(); - assert!(published.starts_with(b"build report\nrequest=")); - assert_eq!( - workflows - .state(b"build-report".to_vec(), None) - .await - .unwrap() - .output - .unwrap() - .status, - WorkflowStatus::Running - ); - - tokio::time::sleep(Duration::from_millis(250)).await; - let second = registry - .run_activity_once(client, &target, 5_000, pool.try_reserve().unwrap()) - .await - .unwrap(); - let ActivityRunOutcome::Completed { receipt, .. } = second else { - panic!("report activity did not complete after retry: {second:?}"); - }; - let state = workflows - .state(b"build-report".to_vec(), Some(receipt)) - .await - .unwrap() - .output - .unwrap(); - assert_eq!(state.status, WorkflowStatus::Completed); - let result = state.result.unwrap(); - assert_eq!(result.len(), 62); - assert_eq!(&result[..9], b"activity\0"); - assert_eq!(result[9], 0); - assert_eq!( - published, - format!("build report\nrequest={:02x?}\n", &result[30..]).into_bytes() - ); - assert_eq!(std::fs::read(report_path).unwrap(), published); - pool.shutdown().await.unwrap(); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/protocol.rs b/crates/crab-cell-runtime/tests/protocol.rs deleted file mode 100644 index c8d272c18..000000000 --- a/crates/crab-cell-runtime/tests/protocol.rs +++ /dev/null @@ -1,7 +0,0 @@ -//! Typed client and peer-protocol integration tests. - -mod support; - -mod protocol { - pub mod client; -} diff --git a/crates/crab-cell-runtime/tests/protocol/client.rs b/crates/crab-cell-runtime/tests/protocol/client.rs deleted file mode 100644 index b7e10b448..000000000 --- a/crates/crab-cell-runtime/tests/protocol/client.rs +++ /dev/null @@ -1,610 +0,0 @@ -use std::{ - future::Future, - pin::Pin, - sync::{Arc, Mutex}, - time::{Duration, UNIX_EPOCH}, -}; - -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::catalog::{CatalogEntry, CatalogProof}; -use crab_cell_runtime::cell::executor::MutationIdentity; -use crab_cell_runtime::cell::executor::Resolution; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::client::{CellClient, InvocationError}; -use crab_cell_runtime::client::{CellDescription, Receipt, command_operation_digest}; -use crab_cell_runtime::codec::{BoundedDecoder, BoundedEncoder, CodecError, WireValue}; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::fleet::telemetry::{ - CellTelemetry, PrimitiveOperationKind, PrimitiveOperationOutcome, -}; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::identity::{IncarnationId, RequestId}; -use crab_cell_runtime::peer::wire; -use crab_cell_runtime::peer::{ - EffectPeerClient, PeerAuthorizer, PeerCellResolver, PeerDispatcher, PeerPrincipal, - PeerRoundTrip, PeerSigner, PeerVerifier, VerifiedPeerRequest, -}; -use crab_cell_runtime::primitives::effects::{ - EffectClaim, EffectClaimRequest, EffectCommandIntent, EffectLeaseOutcome, EffectRunOutcome, - effect_id, effect_operation_digest, register_effect_delivery, -}; -use crab_cell_runtime::primitives::effects::{EffectModule, EffectSource}; -use crab_cell_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, Command, ModuleDescriptor, NamespaceDescriptor, Query, Registry, - RegistryBuilder, -}; -use crab_cell_runtime::registry::{ - CommandContext, CommandResult, MigrationDescriptor, OperationDescriptor, QueryContext, -}; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use crate::support::fixtures::mutation_identity; - -const MODULE: &str = "repository"; -const NAMESPACE: NamespaceId = NamespaceId::from_bytes([6; 16]); -const MIGRATION: &str = - "CREATE TABLE comments(body BLOB NOT NULL); CREATE TABLE audit(value INTEGER NOT NULL)"; -const COMMANDS: &[OperationDescriptor] = &[ - OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 64, - output_limit: 64, - }, - OperationDescriptor { - id: 2, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 64, - output_limit: 64, - }, - OperationDescriptor { - id: 3, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 64, - output_limit: 1, - }, - OperationDescriptor { - id: 4, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 8, - output_limit: 1 << 20, - }, - OperationDescriptor { - id: 5, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 1 << 20, - output_limit: 16, - }, - OperationDescriptor { - id: 6, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 64, - output_limit: 64, - }, - OperationDescriptor { - id: 7, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 64, - output_limit: 64, - }, - OperationDescriptor { - id: 8, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 8, - output_limit: 1, - }, -]; -const QUERIES: &[OperationDescriptor] = &[ - OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 1, - output_limit: 8, - }, - OperationDescriptor { - id: 2, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 1 << 20, - output_limit: 1, - }, - OperationDescriptor { - id: 3, - codec_version: 1, - schema_min: 1, - schema_max: 1, - input_limit: 36, - output_limit: 1 << 20, - }, -]; - -struct RepositoryModule; - -impl CellModule for RepositoryModule { - const NAME: &'static str = MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - descriptor() - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_command::()?; - registry.bind_query::()?; - register_effect_delivery::(registry)?; - Ok(()) - } -} - -impl EffectModule for RepositoryModule { - const MODULE: &'static str = MODULE; - const CLAIM_COMMAND_ID: u32 = 4; - const LEASE_COMMAND_ID: u32 = 5; - const VALIDATE_QUERY_ID: u32 = 2; - const STATUS_QUERY_ID: u32 = 3; -} - -struct CreateComment; - -impl Command for CreateComment { - const MODULE: &'static str = MODULE; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - type Input = Vec; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crab_cell_runtime::Result> { - context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "INSERT INTO comments(body) VALUES (?)".into(), - parameters: vec![SqlValue::Blob(input.clone())], - }], - })?; - Ok(CommandResult::Success(input)) - } -} - -struct AllocateComment; - -impl Command for AllocateComment { - const MODULE: &'static str = MODULE; - const ID: u32 = 8; - const CODEC_VERSION: u32 = 1; - type Input = u64; - type Output = (); - - fn execute( - context: &mut CommandContext<'_, '_>, - bytes: Self::Input, - ) -> crab_cell_runtime::Result> { - context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "INSERT INTO comments(body) VALUES (zeroblob(?))".into(), - parameters: vec![SqlValue::Integer(i64::try_from(bytes).unwrap())], - }], - })?; - Ok(CommandResult::Success(())) - } -} - -struct RejectComment; - -impl Command for RejectComment { - const MODULE: &'static str = MODULE; - const ID: u32 = 2; - const CODEC_VERSION: u32 = 1; - type Input = Vec; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crab_cell_runtime::Result> { - context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "INSERT INTO comments(body) VALUES (?)".into(), - parameters: vec![SqlValue::Blob(input)], - }], - })?; - Ok(CommandResult::Rejected(b"moderated".to_vec())) - } -} - -struct EmitEffectComment; - -impl Command for EmitEffectComment { - const MODULE: &'static str = MODULE; - const ID: u32 = 6; - const CODEC_VERSION: u32 = 1; - type Input = Vec; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crab_cell_runtime::Result> { - let expires_at_ms = context - .now_ms() - .checked_add(60_000) - .ok_or(crab_cell_runtime::Error::Command("effect expiry overflow"))?; - let effect_id = context.emit_effect(&EffectCommandIntent { - target: context.target().clone(), - command_id: CreateComment::ID, - codec_version: CreateComment::CODEC_VERSION, - input, - expires_at_ms, - })?; - Ok(CommandResult::Success(effect_id.to_vec())) - } -} - -struct EmitUndeclaredEffect; - -impl Command for EmitUndeclaredEffect { - const MODULE: &'static str = MODULE; - const ID: u32 = 7; - const CODEC_VERSION: u32 = 1; - type Input = Vec; - type Output = Vec; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crab_cell_runtime::Result> { - let target = CellTarget::new( - context.target().tenant(), - context.target().application(), - NamespaceId::from_bytes([99; 16]), - context.target().partition(), - )?; - context.emit_effect(&EffectCommandIntent { - target, - command_id: CreateComment::ID, - codec_version: CreateComment::CODEC_VERSION, - input, - expires_at_ms: context - .now_ms() - .checked_add(60_000) - .ok_or(crab_cell_runtime::Error::Command("effect expiry overflow"))?, - })?; - Ok(CommandResult::Success(Vec::new())) - } -} - -struct InvalidOutput; - -impl WireValue for InvalidOutput { - fn encode(&self, encoder: &mut BoundedEncoder) -> Result<(), CodecError> { - encoder.write_u8(2) - } - - fn decode(decoder: &mut BoundedDecoder<'_>) -> Result { - decoder.read_bool()?; - Ok(Self) - } -} - -struct InvalidResultComment; - -impl Command for InvalidResultComment { - const MODULE: &'static str = MODULE; - const ID: u32 = 3; - const CODEC_VERSION: u32 = 1; - type Input = Vec; - type Output = InvalidOutput; - - fn execute( - context: &mut CommandContext<'_, '_>, - input: Self::Input, - ) -> crab_cell_runtime::Result> { - context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "INSERT INTO comments(body) VALUES (?)".into(), - parameters: vec![SqlValue::Blob(input)], - }], - })?; - Ok(CommandResult::Success(InvalidOutput)) - } -} - -struct CountComments; - -impl Query for CountComments { - const MODULE: &'static str = MODULE; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - type Input = (); - type Output = u64; - - fn execute( - context: &mut QueryContext<'_>, - _input: Self::Input, - ) -> crab_cell_runtime::Result { - let results = context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT COUNT(*) FROM comments".into(), - parameters: Vec::new(), - }], - })?; - match results[0].rows.first().and_then(|row| row.first()) { - Some(SqlValue::Integer(count)) => u64::try_from(*count) - .map_err(|_| crab_cell_runtime::Error::Command("negative comment count")), - _ => Err(crab_cell_runtime::Error::Command( - "comment count query returned no integer", - )), - } - } -} - -struct WrongModuleCommand; - -impl Command for WrongModuleCommand { - const MODULE: &'static str = "other"; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - type Input = (); - type Output = (); - - fn execute( - _context: &mut CommandContext<'_, '_>, - _input: Self::Input, - ) -> crab_cell_runtime::Result> { - Ok(CommandResult::Success(())) - } -} - -fn descriptor() -> &'static ModuleDescriptor { - static DESCRIPTOR: std::sync::OnceLock = std::sync::OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: MODULE, - source_digest: Digest::from_bytes([3; 32]), - retained_codes: &[], - schema_min: 1, - schema_max: 1, - migrations: Box::leak(Box::new([MigrationDescriptor { - version: 1, - sql: MIGRATION, - digest: Digest::from_bytes(*blake3::hash(MIGRATION.as_bytes()).as_bytes()), - }])), - commands: COMMANDS, - queries: QUERIES, - workflow_definitions: &[], - activity_types: &[], - namespaces: &[NamespaceDescriptor { - id: NAMESPACE, - name: MODULE, - role: CatalogRole::Repository, - shards: 1, - effect_targets: &[NAMESPACE], - dead_letter: None, - }], - }) -} - -fn registry() -> Arc { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "client-test".into(), - cargo_lock_digest: Digest::from_bytes([9; 32]), - }); - builder.register(RepositoryModule).unwrap(); - Arc::new(builder.finish().unwrap()) -} - -struct Fixture { - _directory: tempfile::TempDir, - layout: CellStorageLayout, - replica: CellReplica, - authority: CellAuthority, - proof: CatalogProof, - runtime: Option, - session: SessionId, - incarnation: IncarnationId, - target: CellTarget, - handle: Option, - registry: Arc, -} - -impl Fixture { - fn handle(&self) -> &crab_cell_runtime::cell::actor::CellHandle { - self.handle.as_ref().expect("fixture handle is present") - } - - fn take_handle(&mut self) -> crab_cell_runtime::cell::actor::CellHandle { - self.handle.take().expect("fixture handle is present") - } -} - -async fn fixture() -> Fixture { - fixture_with_limits(Limits::default()).await -} - -async fn fixture_with_limits(limits: Limits) -> Fixture { - let registry = registry(); - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NAMESPACE, - b"repository-42", - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([4; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store, Path::from("client"), [2; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - limits, - ) - .unwrap(); - let catalog = - crab_cell_runtime::cell::catalog::CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Repository, - registry.module_code(MODULE).unwrap(), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let session = SessionId::from_bytes([5; 16]); - let authority = CellAuthority::new(layout.clone()); - let observed = authority - .create_initial( - &proof, - incarnation, - Owner { - session, - endpoint: "https://local.internal:8081".into(), - }, - ) - .await - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 4).unwrap(), 4 * 1024 * 1024, session).unwrap(); - let handle = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - observed, - directory.path().join("repository.sqlite"), - |transaction| { - transaction.execute_batch(MIGRATION)?; - Ok(()) - }, - ) - .await - .unwrap(); - Fixture { - _directory: directory, - layout, - replica, - authority, - proof, - runtime: Some(runtime), - session, - incarnation, - target, - handle: Some(handle), - registry, - } -} - -struct LocalResolver { - target: CellTarget, - handle: crab_cell_runtime::cell::actor::CellHandle, -} - -impl PeerCellResolver for LocalResolver { - fn resolve( - &self, - target: CellTarget, - ) -> Pin< - Box< - dyn Future< - Output = crab_cell_runtime::Result, - > + Send - + 'static, - >, - > { - let matches = target == self.target; - let handle = self.handle.clone(); - Box::pin(async move { - if matches { - Ok(handle) - } else { - Err(crab_cell_runtime::Error::CellNotActive) - } - }) - } -} - -struct RepositoryAuthorizer; - -impl PeerAuthorizer for RepositoryAuthorizer { - fn authorize(&self, request: &VerifiedPeerRequest) -> crab_cell_runtime::Result<()> { - if request.permits("repository.issue.create") { - Ok(()) - } else { - Err(crab_cell_runtime::Error::PeerAuthorization( - "missing repository action", - )) - } - } -} - -struct LoopbackRoundTrip { - verifier: Arc, - dispatcher: Arc, -} - -impl PeerRoundTrip for LoopbackRoundTrip { - fn send( - &self, - target: CellTarget, - request: Vec, - _remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let verifier = Arc::clone(&self.verifier); - let dispatcher = Arc::clone(&self.dispatcher); - Box::pin(async move { - let now_ms = i64::try_from( - std::time::SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_err(|_| crab_cell_runtime::Error::Peer("test clock"))? - .as_millis(), - ) - .map_err(|_| crab_cell_runtime::Error::Peer("test clock overflow"))?; - let verified = verifier.verify(&request, now_ms)?; - if verified.target() != &target { - return Err(crab_cell_runtime::Error::Peer("round trip target changed")); - } - dispatcher.dispatch_bytes(&verified, now_ms).await - }) - } -} - -mod effects; -mod telemetry; -mod typed; diff --git a/crates/crab-cell-runtime/tests/protocol/client/effects.rs b/crates/crab-cell-runtime/tests/protocol/client/effects.rs deleted file mode 100644 index e661c1ebf..000000000 --- a/crates/crab-cell-runtime/tests/protocol/client/effects.rs +++ /dev/null @@ -1,307 +0,0 @@ -//! Authenticated effect delivery, source leases, and supervision. - -use super::*; - -#[tokio::test] -async fn authenticated_effect_delivery_publishes_once_and_resolves_from_inbox() { - let fixture = fixture().await; - let signer = PeerSigner::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - ed25519_dalek::SigningKey::from_bytes(&[13; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - signer.verifying_key(), - )); - let dispatcher = Arc::new(PeerDispatcher::new( - Arc::clone(&fixture.registry), - Arc::new(LocalResolver { - target: fixture.target.clone(), - handle: fixture.handle().clone(), - }), - Arc::new(RepositoryAuthorizer), - )); - let round_trip = Arc::new(LoopbackRoundTrip { - verifier, - dispatcher, - }); - let principal = PeerPrincipal { - issuer: "crab-runtime:test".into(), - subject: "source-session".into(), - actions: vec!["repository.issue.create".into()], - }; - let now_ms = i64::try_from( - std::time::SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap(); - let source_cell = crab_cell_runtime::CellId::from_bytes([21; 32]); - let source_incarnation = IncarnationId::from_bytes([22; 16]); - let source_sequence = 9; - let ordinal = 3; - let effect_id = effect_id(source_cell, source_incarnation, source_sequence, ordinal); - let identity = wire::EffectIdentity { - effect_id: effect_id.to_vec(), - source_cell: source_cell.as_bytes().to_vec(), - source_incarnation: source_incarnation.as_bytes().to_vec(), - source_sequence, - ordinal, - expires_at_ms: now_ms + 60_000, - }; - let mut encoder = BoundedEncoder::new(64).unwrap(); - b"effect".to_vec().encode(&mut encoder).unwrap(); - let request = wire::EffectRequest { - target: Some(wire::Target { - tenant_id: fixture.target.tenant().as_bytes().to_vec(), - application_id: fixture.target.application().as_bytes().to_vec(), - namespace_id: fixture.target.namespace().as_bytes().to_vec(), - partition: fixture.target.partition().to_vec(), - }), - destination_incarnation: Vec::new(), - identity: Some(identity.clone()), - operation: Some(wire::effect_request::Operation::CellCommand( - wire::CellCommand { - command_id: CreateComment::ID, - codec_version: CreateComment::CODEC_VERSION, - input: encoder.finish(), - }, - )), - }; - let operation_digest = effect_operation_digest( - fixture.target.cell_id(), - effect_id, - &prost::Message::encode_to_vec(&request), - ); - let claim = EffectClaim { - effect_id, - destination: fixture.target.cell_id(), - operation: prost::Message::encode_to_vec(&request), - operation_digest, - attempt: 1, - token: [23; 16], - lease_until_ms: now_ms + 30_000, - expires_at_ms: identity.expires_at_ms, - created_sequence: source_sequence, - }; - let client = EffectPeerClient::new(Arc::new(signer), principal, round_trip); - let delivered = client.deliver(&claim, now_ms).await.unwrap(); - assert_eq!(delivered.commit_sequence(), 1); - assert_eq!(client.deliver(&claim, now_ms + 1).await.unwrap(), delivered); - assert_eq!( - client.resolve(&claim, now_ms + 2).await.unwrap(), - Resolution::Committed(delivered) - ); - - assert_eq!( - CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()) - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output, - 1 - ); - fixture.handle().drain().await.unwrap(); -} -#[tokio::test] -async fn typed_effect_source_publishes_claim_validation_ack_and_lost_lease() { - let mut fixture = fixture().await; - let source_cell = fixture.target.cell_id(); - let source_incarnation = fixture.incarnation; - let source_sequence = 1; - let ordinal = 0; - let identity = mutation_identity(30); - let expected_effect_id = effect_id(source_cell, source_incarnation, source_sequence, ordinal); - let expires_at_ms = identity.issued_at_ms + 60_000; - let mut encoder = BoundedEncoder::new(64).unwrap(); - b"effect-source".to_vec().encode(&mut encoder).unwrap(); - let input = encoder.finish(); - let client = CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()); - let committed = client - .command::(&fixture.target, identity, input) - .await - .unwrap(); - let effect_id: [u8; 32] = committed.output.try_into().unwrap(); - assert_eq!(effect_id, expected_effect_id); - assert_eq!(source_sequence, committed.receipt.commit_sequence); - assert_eq!(expires_at_ms, identity.issued_at_ms + 60_000); - - let first_handle = fixture.take_handle(); - drop(first_handle); - drop(fixture.runtime.take().expect("fixture runtime is present")); - let stale = fixture.authority.load(source_cell).await.unwrap().unwrap(); - let successor = SessionId::from_bytes([15; 16]); - let fenced = - crate::support::fencing::fence_session(&fixture.layout, fixture.session, successor).await; - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 4).unwrap(), - 4 * 1024 * 1024, - successor, - ) - .unwrap(); - let restored = runtime - .takeover_restored( - fixture.proof.clone(), - fixture.replica.clone(), - fixture.authority.clone(), - stale, - fenced.direct_takeover().unwrap(), - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - fixture.layout.clone(), - Limits::default(), - ), - fixture._directory.path().join("effect-successor.sqlite"), - Owner { - session: successor, - endpoint: "https://effect-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - - let source = EffectSource::::new( - CellClient::local(Arc::clone(&fixture.registry), restored.clone()), - fixture.target.clone(), - ); - let claimed = source - .claim( - mutation_identity(32), - EffectClaimRequest { - limit: 1, - lease_ms: 5_000, - }, - ) - .await - .unwrap(); - assert_eq!(claimed.receipt.commit_sequence, 2); - assert_eq!(claimed.output.len(), 1); - assert_eq!(claimed.output[0].effect_id, effect_id); - - let claim = claimed.output[0].clone(); - let validated = source - .validate(vec![claim.clone()], claimed.receipt) - .await - .unwrap(); - assert!(validated.output); - assert_eq!(validated.receipt.commit_sequence, 2); - - let acknowledged = source - .ack( - mutation_identity(33), - claim.clone(), - b"destination result".to_vec(), - ) - .await - .unwrap(); - assert_eq!(acknowledged.output, EffectLeaseOutcome::Delivered); - assert_eq!(acknowledged.receipt.commit_sequence, 3); - - let lost = source - .retry(mutation_identity(34), claim) - .await - .unwrap_err(); - assert!(matches!( - lost, - InvocationError::Rejected(ref outcome) - if outcome.output == EffectLeaseOutcome::LeaseLost - && outcome.receipt.commit_sequence == 4 - )); - - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} -#[tokio::test] -async fn effect_supervisor_delivers_to_inbox_and_acknowledges_source() { - let fixture = fixture().await; - let source_sequence = 1; - let identity = mutation_identity(40); - let mut encoder = BoundedEncoder::new(64).unwrap(); - b"supervised".to_vec().encode(&mut encoder).unwrap(); - let input = encoder.finish(); - let client = CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()); - let committed = client - .command::(&fixture.target, identity, input) - .await - .unwrap(); - assert_eq!(source_sequence, committed.receipt.commit_sequence); - - let signer = PeerSigner::new( - SessionId::from_bytes([42; 16]), - fixture.registry.release_digest(), - ed25519_dalek::SigningKey::from_bytes(&[43; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - SessionId::from_bytes([42; 16]), - fixture.registry.release_digest(), - signer.verifying_key(), - )); - let dispatcher = Arc::new(PeerDispatcher::new( - Arc::clone(&fixture.registry), - Arc::new(LocalResolver { - target: fixture.target.clone(), - handle: fixture.handle().clone(), - }), - Arc::new(RepositoryAuthorizer), - )); - let peer = EffectPeerClient::new( - Arc::new(signer), - PeerPrincipal { - issuer: "crab-runtime:test".into(), - subject: "source-session".into(), - actions: vec!["repository.issue.create".into()], - }, - Arc::new(LoopbackRoundTrip { - verifier, - dispatcher, - }), - ); - assert!( - fixture - .registry - .has_effect_runner(fixture.target.namespace()) - ); - let outcome = fixture - .registry - .run_effect_once( - CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()), - fixture.target.clone(), - peer, - 5_000, - ) - .await - .unwrap(); - assert!( - matches!( - &outcome, - EffectRunOutcome::Delivered { - destination: crab_cell_runtime::cell::executor::StoredOutcome::Success { - result, - commit_sequence: 3, - }, - receipt: Receipt { - commit_sequence: 4, - .. - }, - } if { - let mut decoder = BoundedDecoder::new(result, 64).unwrap(); - let decoded = Vec::::decode(&mut decoder).unwrap(); - decoder.finish().unwrap(); - decoded == b"supervised" - } - ), - "{outcome:?}" - ); - assert_eq!( - CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()) - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output, - 1 - ); - - fixture.handle().drain().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/protocol/client/telemetry.rs b/crates/crab-cell-runtime/tests/protocol/client/telemetry.rs deleted file mode 100644 index 7d56dde56..000000000 --- a/crates/crab-cell-runtime/tests/protocol/client/telemetry.rs +++ /dev/null @@ -1,151 +0,0 @@ -//! Primitive operation telemetry as observed through the typed client. - -use super::*; - -struct RecordingPrimitiveTelemetry { - operations: Mutex< - Vec<( - &'static str, - PrimitiveOperationKind, - PrimitiveOperationOutcome, - )>, - >, -} -impl Default for RecordingPrimitiveTelemetry { - fn default() -> Self { - Self { - operations: Mutex::new(Vec::new()), - } - } -} -impl CellTelemetry for RecordingPrimitiveTelemetry { - fn primitive_operation( - &self, - module: &'static str, - kind: PrimitiveOperationKind, - outcome: PrimitiveOperationOutcome, - _elapsed: Duration, - ) { - self.operations - .lock() - .unwrap() - .push((module, kind, outcome)); - } -} -#[tokio::test] -async fn primitive_operations_are_reported_for_local_and_peer_execution() { - let fixture = fixture().await; - let recording = Arc::new(RecordingPrimitiveTelemetry::default()); - fixture - .runtime - .as_ref() - .expect("fixture runtime") - .install_telemetry(recording.clone()) - .unwrap(); - let telemetry = fixture - .runtime - .as_ref() - .expect("fixture runtime") - .telemetry_handle(); - - let client = CellClient::local_with_telemetry( - Arc::clone(&fixture.registry), - fixture.handle().clone(), - telemetry.clone(), - ); - client - .command::(&fixture.target, mutation_identity(31), b"reported".to_vec()) - .await - .unwrap(); - assert!( - client - .command::( - &fixture.target, - mutation_identity(32), - b"moderated".to_vec() - ) - .await - .is_err() - ); - client - .query::(&fixture.target, None, ()) - .await - .unwrap(); - - let signer = PeerSigner::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - ed25519_dalek::SigningKey::from_bytes(&[13; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - signer.verifying_key(), - )); - let dispatcher = Arc::new( - PeerDispatcher::new( - Arc::clone(&fixture.registry), - Arc::new(LocalResolver { - target: fixture.target.clone(), - handle: fixture.handle().clone(), - }), - Arc::new(RepositoryAuthorizer), - ) - .with_telemetry(telemetry), - ); - let peer = CellClient::peer( - Arc::clone(&fixture.registry), - Arc::new(signer), - PeerPrincipal { - issuer: "https://identity.example".into(), - subject: "alice".into(), - actions: vec!["repository.issue.create".into()], - }, - Arc::new(LoopbackRoundTrip { - verifier, - dispatcher, - }), - ); - peer.command::(&fixture.target, mutation_identity(33), b"peer".to_vec()) - .await - .unwrap(); - - let reported = recording.operations.lock().unwrap().clone(); - assert!( - reported.contains(&( - MODULE, - PrimitiveOperationKind::Command, - PrimitiveOperationOutcome::Success - )), - "local command outcome is reported: {reported:?}" - ); - assert!( - reported.contains(&( - MODULE, - PrimitiveOperationKind::Command, - PrimitiveOperationOutcome::Rejected - )), - "rejected command outcome is reported: {reported:?}" - ); - assert!( - reported.contains(&( - MODULE, - PrimitiveOperationKind::Query, - PrimitiveOperationOutcome::Success - )), - "query outcome is reported: {reported:?}" - ); - assert!( - reported - .iter() - .filter( - |(_, kind, outcome)| *kind == PrimitiveOperationKind::Command - && *outcome == PrimitiveOperationOutcome::Success - ) - .count() - >= 2, - "peer-routed command is reported beside the local one: {reported:?}" - ); - - fixture.handle().drain().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/protocol/client/typed.rs b/crates/crab-cell-runtime/tests/protocol/client/typed.rs deleted file mode 100644 index 06792802b..000000000 --- a/crates/crab-cell-runtime/tests/protocol/client/typed.rs +++ /dev/null @@ -1,746 +0,0 @@ -//! Typed client command, query, digest, and conflict semantics. - -use super::*; -use crab_cell_runtime::cell::actor::CellHandle; - -#[tokio::test] -async fn local_resolver_refusal_never_dispatches_to_the_underlying_owner() { - struct Refused; - impl crab_cell_runtime::client::LocalCellResolver for Refused { - fn resolve( - &self, - _target: CellTarget, - ) -> Pin< - Box< - dyn Future>> + Send + 'static, - >, - > { - Box::pin(async { Err(crab_cell_runtime::Error::Capacity("owner admission")) }) - } - } - let fixture = fixture().await; - let owner = CellClient::local(fixture.registry.clone(), fixture.handle().clone()); - let client = owner.clone().with_local_resolver(Arc::new(Refused)); - let refused = client - .command::(&fixture.target, mutation_identity(73), b"rejected".to_vec()) - .await - .unwrap_err(); - assert!(matches!( - refused, - InvocationError::NotStarted(crab_cell_runtime::Error::Capacity("owner admission")) - )); - assert_eq!( - owner - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output, - 0 - ); - fixture.handle().drain().await.unwrap(); -} - -fn verified_description(target: &CellTarget) -> VerifiedPeerRequest { - let session = SessionId::from_bytes([12; 16]); - let release = Digest::from_bytes([13; 32]); - let signer = PeerSigner::new( - session, - release, - ed25519_dalek::SigningKey::from_bytes(&[14; 32]), - ); - let signed = signer - .sign( - PeerPrincipal { - issuer: "https://identity.example".into(), - subject: "alice".into(), - actions: vec!["repository.issue.create".into()], - }, - 1_000, - 61_000, - 30_000, - crab_cell_runtime::peer::PeerOperation::Read(wire::ReadRequest { - expected: None, - target: Some(wire::Target { - tenant_id: target.tenant().as_bytes().to_vec(), - application_id: target.application().as_bytes().to_vec(), - namespace_id: target.namespace().as_bytes().to_vec(), - partition: target.partition().to_vec(), - }), - timeout_ms: 30_000, - minimum: None, - operation: Some(wire::read_request::Operation::Describe(true)), - }), - ) - .unwrap(); - PeerVerifier::new(session, release, signer.verifying_key()) - .verify(&signed, 1_000) - .unwrap() -} - -#[tokio::test] -async fn resolved_peer_dispatch_uses_the_receivers_handle() { - let fixture = fixture().await; - let other = CellTarget::new( - fixture.target.tenant(), - fixture.target.application(), - fixture.target.namespace(), - b"another-repository", - ) - .unwrap(); - let dispatcher = PeerDispatcher::new( - Arc::clone(&fixture.registry), - Arc::new(LocalResolver { - target: other, - handle: fixture.handle().clone(), - }), - Arc::new(RepositoryAuthorizer), - ); - let request = verified_description(&fixture.target); - let reply = dispatcher - .dispatch_resolved(&request, 1_000, Ok(fixture.handle().clone())) - .await; - assert!(matches!( - reply.outcome, - Some(wire::peer_reply::Outcome::Read(_)) - )); - fixture.handle().drain().await.unwrap(); -} - -#[tokio::test] -async fn resolved_peer_dispatch_rejects_a_handle_for_another_target() { - let fixture = fixture().await; - let other = CellTarget::new( - fixture.target.tenant(), - fixture.target.application(), - fixture.target.namespace(), - b"another-repository", - ) - .unwrap(); - let dispatcher = PeerDispatcher::new( - Arc::clone(&fixture.registry), - Arc::new(LocalResolver { - target: fixture.target.clone(), - handle: fixture.handle().clone(), - }), - Arc::new(RepositoryAuthorizer), - ); - let request = verified_description(&other); - let reply = dispatcher - .dispatch_resolved(&request, 1_000, Ok(fixture.handle().clone())) - .await; - assert!(matches!( - reply.outcome, - Some(wire::peer_reply::Outcome::Error(_)) - )); - fixture.handle().drain().await.unwrap(); -} - -#[tokio::test] -async fn peer_checks_observed_contract_before_commands_queries_and_resolution() { - let fixture = fixture().await; - let handle = fixture.handle(); - let expected = wire::CellDescription { - cell_id: handle.cell_id().as_bytes().to_vec(), - incarnation: handle.incarnation().as_bytes().to_vec(), - code: handle.code().as_bytes().to_vec(), - schema: handle.schema(), - }; - let target = wire::Target { - tenant_id: fixture.target.tenant().as_bytes().to_vec(), - application_id: fixture.target.application().as_bytes().to_vec(), - namespace_id: fixture.target.namespace().as_bytes().to_vec(), - partition: fixture.target.partition().to_vec(), - }; - let identity = mutation_identity(61); - let wire_identity = wire::MutationIdentity { - request_id: identity.request_id.as_bytes().to_vec(), - incarnation: handle.incarnation().as_bytes().to_vec(), - issued_at_ms: identity.issued_at_ms, - expires_at_ms: identity.expires_at_ms, - }; - let mut encoder = BoundedEncoder::new(64).unwrap(); - b"once".to_vec().encode(&mut encoder).unwrap(); - let input = encoder.finish(); - let mut encoder = BoundedEncoder::new(1).unwrap(); - ().encode(&mut encoder).unwrap(); - let query_input = encoder.finish(); - let description = CellDescription { - cell: handle.cell_id(), - incarnation: handle.incarnation(), - code: handle.code(), - schema: handle.schema(), - }; - let digest = command_operation_digest::(description, identity, &input).unwrap(); - let signer = PeerSigner::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - ed25519_dalek::SigningKey::from_bytes(&[13; 32]), - ); - let verifier = PeerVerifier::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - signer.verifying_key(), - ); - let dispatcher = PeerDispatcher::new( - Arc::clone(&fixture.registry), - Arc::new(LocalResolver { - target: fixture.target.clone(), - handle: handle.clone(), - }), - Arc::new(RepositoryAuthorizer), - ); - for field in ["current", "cell", "incarnation", "code", "schema"] { - let mut observed = expected.clone(); - match field { - "cell" => observed.cell_id[0] ^= 1, - "incarnation" => observed.incarnation[0] ^= 1, - "code" => observed.code[0] ^= 1, - "schema" => observed.schema += 1, - _ => {} - } - let operations = [ - crab_cell_runtime::peer::PeerOperation::Mutate(wire::MutationRequest { - target: Some(target.clone()), - identity: Some(wire_identity.clone()), - timeout_ms: 30_000, - expected: Some(observed.clone()), - operation: Some(wire::mutation_request::Operation::CellCommand( - wire::CellCommand { - command_id: CreateComment::ID, - codec_version: CreateComment::CODEC_VERSION, - input: input.clone(), - }, - )), - }), - crab_cell_runtime::peer::PeerOperation::Read(wire::ReadRequest { - target: Some(target.clone()), - timeout_ms: 30_000, - minimum: None, - expected: Some(observed.clone()), - operation: Some(wire::read_request::Operation::CellQuery(wire::CellQuery { - query_id: CountComments::ID, - codec_version: CountComments::CODEC_VERSION, - input: query_input.clone(), - })), - }), - crab_cell_runtime::peer::PeerOperation::Resolve(wire::ResolveRequest { - target: Some(target.clone()), - identity: Some(wire_identity.clone()), - operation_digest: digest.as_bytes().to_vec(), - expected: Some(observed), - }), - ]; - for operation in operations { - let signed = signer - .sign( - PeerPrincipal { - issuer: "https://identity.example".into(), - subject: "alice".into(), - actions: vec!["repository.issue.create".into()], - }, - identity.issued_at_ms, - identity.expires_at_ms, - 30_000, - operation, - ) - .unwrap(); - let verified = verifier.verify(&signed, identity.issued_at_ms).unwrap(); - let reply = dispatcher.dispatch(&verified, identity.issued_at_ms).await; - if field == "current" { - assert!(matches!( - (verified.operation_tag(), reply.outcome), - (10, Some(wire::peer_reply::Outcome::Mutation(_))) - | (11, Some(wire::peer_reply::Outcome::Read(_))) - | (12, Some(wire::peer_reply::Outcome::Resolve(_))) - )); - } else { - assert!( - matches!(reply.outcome, - Some(wire::peer_reply::Outcome::Error(error)) if error.code == wire::error::Code::Unavailable as i32 - && error.outcome == wire::error::Outcome::NotStarted as i32 - ), - "stale {field}" - ); - } - } - } - let client = CellClient::local(Arc::clone(&fixture.registry), handle.clone()); - assert_eq!( - client - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output, - 1 - ); - handle.drain().await.unwrap(); -} - -#[tokio::test] -async fn sqlite_full_preserves_local_cause_and_peer_not_started_outcome() { - let fixture = fixture_with_limits(Limits { - max_database_bytes: 512 * 1024, - ..Limits::default() - }) - .await; - let local = CellClient::local(fixture.registry.clone(), fixture.handle().clone()); - let signer = PeerSigner::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - ed25519_dalek::SigningKey::from_bytes(&[13; 32]), - ); - let transport = LoopbackRoundTrip { - verifier: Arc::new(PeerVerifier::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - signer.verifying_key(), - )), - dispatcher: Arc::new(PeerDispatcher::new( - fixture.registry.clone(), - Arc::new(LocalResolver { - target: fixture.target.clone(), - handle: fixture.handle().clone(), - }), - Arc::new(RepositoryAuthorizer), - )), - }; - let peer = CellClient::peer( - fixture.registry.clone(), - Arc::new(signer), - PeerPrincipal { - issuer: "https://identity.example".into(), - subject: "alice".into(), - actions: vec!["repository.issue.create".into()], - }, - Arc::new(transport), - ); - for (index, client) in [&local, &peer].into_iter().enumerate() { - let error = client - .command::( - &fixture.target, - mutation_identity(80 + index as u8), - 1024 * 1024, - ) - .await - .unwrap_err(); - if index == 0 { - assert!( - matches!(error, InvocationError::NotStarted(crab_cell_runtime::Error::Sqlite(ref cause)) - if cause.sqlite_error_code() == Some(crab_ltx::rusqlite::ErrorCode::DiskFull)), - "{error:?}" - ); - } else { - assert!( - matches!( - error, - InvocationError::NotStarted(crab_cell_runtime::Error::Capacity(_)) - ), - "{error:?}" - ); - } - assert_eq!( - client - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output, - 0 - ); - } - // Both refused transactions rolled back; the same owner can still publish. - let saved = peer - .command::( - &fixture.target, - mutation_identity(82), - b"after-full".to_vec(), - ) - .await - .unwrap(); - assert_eq!(saved.receipt.commit_sequence, 1); - assert_eq!( - local - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output, - 1 - ); - fixture.handle().drain().await.unwrap(); -} - -#[tokio::test] -async fn typed_client_publishes_replays_rejections_and_receipted_reads() { - let fixture = fixture().await; - let client = CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()); - let identity = mutation_identity(7); - - let committed = client - .command::(&fixture.target, identity, b"first".to_vec()) - .await - .unwrap(); - assert_eq!(committed.output, b"first"); - assert_eq!(committed.receipt.commit_sequence, 1); - - let replay = client - .command::(&fixture.target, identity, b"first".to_vec()) - .await - .unwrap(); - assert_eq!(replay, committed); - - let rejected = client - .command::(&fixture.target, mutation_identity(8), b"hidden".to_vec()) - .await - .unwrap_err(); - assert!(matches!( - rejected, - InvocationError::Rejected(ref outcome) - if outcome.output == b"moderated" && outcome.receipt.commit_sequence == 2 - )); - - let observed = client - .query::(&fixture.target, Some(committed.receipt), ()) - .await - .unwrap(); - assert_eq!(observed.output, 1); - assert_eq!(observed.receipt.commit_sequence, 2); - - fixture.handle().drain().await.unwrap(); -} -#[tokio::test] -async fn state_stream_serializes_local_queries_across_a_new_commit() { - let fixture = fixture().await; - let client = CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()); - let mut stream = client - .open_state_stream::( - &fixture.target, - std::time::Instant::now() + std::time::Duration::from_secs(5), - ) - .await - .unwrap(); - - let first = stream.emit(()).await.unwrap(); - assert_eq!(first.output, 0); - let committed = client - .command::(&fixture.target, mutation_identity(18), b"streamed".to_vec()) - .await - .unwrap(); - let second = stream.emit(()).await.unwrap(); - assert_eq!(second.output, 1); - assert!(second.receipt.commit_sequence >= committed.receipt.commit_sequence); - assert_eq!(stream.last_receipt(), Some(second.receipt)); - - stream.finish(); - fixture.handle().drain().await.unwrap(); -} -#[tokio::test] -async fn local_and_peer_command_share_digest_dedup_and_query_state() { - let fixture = fixture().await; - let client = CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()); - let identity = mutation_identity(14); - let committed = client - .command::(&fixture.target, identity, b"same".to_vec()) - .await - .unwrap(); - - let signer = PeerSigner::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - ed25519_dalek::SigningKey::from_bytes(&[13; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - signer.verifying_key(), - )); - let dispatcher = Arc::new(PeerDispatcher::new( - Arc::clone(&fixture.registry), - Arc::new(LocalResolver { - target: fixture.target.clone(), - handle: fixture.handle().clone(), - }), - Arc::new(RepositoryAuthorizer), - )); - let peer = CellClient::peer( - Arc::clone(&fixture.registry), - Arc::new(signer), - PeerPrincipal { - issuer: "https://identity.example".into(), - subject: "alice".into(), - actions: vec!["repository.issue.create".into()], - }, - Arc::new(LoopbackRoundTrip { - verifier, - dispatcher, - }), - ); - let observed = peer.clone().with_observed_description(CellDescription { - cell: fixture.handle().cell_id(), - incarnation: fixture.handle().incarnation(), - code: fixture.handle().code(), - schema: fixture.handle().schema(), - }); - for peer in [peer, observed] { - assert_eq!( - peer.command::(&fixture.target, identity, b"same".to_vec()) - .await - .unwrap(), - committed - ); - let result = peer - .query::(&fixture.target, Some(committed.receipt), ()) - .await - .unwrap(); - assert_eq!(result.output, 1); - assert_eq!(result.receipt.commit_sequence, 1); - } - - fixture.handle().drain().await.unwrap(); -} - -#[tokio::test] -async fn runtime_client_forwards_to_the_remote_owner() { - let fixture = fixture().await; - let caller = CellRuntime::new( - SqlWorkerPool::new(1, 4).unwrap(), - 4 * 1024 * 1024, - SessionId::from_bytes([6; 16]), - ) - .unwrap(); - let signer = PeerSigner::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - ed25519_dalek::SigningKey::from_bytes(&[13; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - SessionId::from_bytes([12; 16]), - fixture.registry.release_digest(), - signer.verifying_key(), - )); - let dispatcher = Arc::new(PeerDispatcher::new( - Arc::clone(&fixture.registry), - Arc::new(LocalResolver { - target: fixture.target.clone(), - handle: fixture.handle().clone(), - }), - Arc::new(RepositoryAuthorizer), - )); - let round_trip: Arc = Arc::new(LoopbackRoundTrip { - verifier, - dispatcher, - }); - let client = CellClient::runtime_with_peer( - Arc::clone(&fixture.registry), - caller.clone(), - fixture.layout.clone(), - Arc::new(signer), - PeerPrincipal { - issuer: "https://identity.example".into(), - subject: "alice".into(), - actions: vec!["repository.issue.create".into()], - }, - Arc::clone(&round_trip), - ); - let committed = client - .command::(&fixture.target, mutation_identity(70), b"remote".to_vec()) - .await - .unwrap(); - let observed = client - .query::(&fixture.target, Some(committed.receipt), ()) - .await - .unwrap(); - assert_eq!(observed.output, 1); - let local = CellClient::runtime_with_peer( - Arc::clone(&fixture.registry), - fixture.runtime.as_ref().unwrap().clone(), - fixture.layout.clone(), - Arc::new(PeerSigner::new( - SessionId::from_bytes([22; 16]), - fixture.registry.release_digest(), - ed25519_dalek::SigningKey::from_bytes(&[23; 32]), - )), - PeerPrincipal { - issuer: "https://identity.example".into(), - subject: "alice".into(), - actions: vec!["repository.issue.create".into()], - }, - round_trip, - ); - // The signer is not enrolled in the loopback verifier; this succeeds only - // when the current local owner is selected before the peer transport. - assert_eq!( - local - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output, - 1 - ); - caller.shutdown().await.unwrap(); - fixture.handle().drain().await.unwrap(); -} -#[tokio::test] -async fn typed_client_rejects_conflicting_identity_receipt_and_module_before_execution() { - let fixture = fixture().await; - let client = CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()); - let identity = mutation_identity(9); - let committed = client - .command::(&fixture.target, identity, b"first".to_vec()) - .await - .unwrap(); - - assert!(matches!( - client - .command::(&fixture.target, identity, b"different".to_vec()) - .await, - Err(InvocationError::NotStarted( - crab_cell_runtime::Error::RequestConflict - )) - )); - assert!(matches!( - client - .query::( - &fixture.target, - Some(Receipt { - incarnation: IncarnationId::from_bytes([99; 16]), - ..committed.receipt - }), - (), - ) - .await, - Err(InvocationError::NotStarted( - crab_cell_runtime::Error::Command("minimum receipt does not match Cell") - )) - )); - assert!(matches!( - client - .command::(&fixture.target, mutation_identity(10), ()) - .await, - Err(InvocationError::NotStarted( - crab_cell_runtime::Error::Registry("operation module does not own namespace") - )) - )); - let now_ms = i64::try_from( - std::time::SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap(); - assert!(matches!( - client - .command::( - &fixture.target, - MutationIdentity { - request_id: RequestId::from_bytes([11; 16]), - issued_at_ms: now_ms - 2, - expires_at_ms: now_ms - 1, - }, - b"expired".to_vec(), - ) - .await, - Err(InvocationError::NotStarted( - crab_cell_runtime::Error::Command("invalid mutation identity lifetime") - )) - )); - - fixture.handle().drain().await.unwrap(); -} -#[tokio::test] -async fn command_effects_require_declared_same_application_targets() { - let fixture = fixture().await; - let client = CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()); - - let result = client - .command::( - &fixture.target, - mutation_identity(50), - b"foreign".to_vec(), - ) - .await; - assert!(matches!( - result, - Err(InvocationError::NotStarted( - crab_cell_runtime::Error::Command("effect target is not declared") - )) - )); - assert_eq!( - client - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output, - 0 - ); - - fixture.handle().drain().await.unwrap(); -} -#[tokio::test] -async fn invalid_typed_result_preserves_the_published_receipt() { - let fixture = fixture().await; - let client = CellClient::local(Arc::clone(&fixture.registry), fixture.handle().clone()); - - let result = client - .command::(&fixture.target, mutation_identity(12), b"stored".to_vec()) - .await; - assert!(matches!( - result, - Err(InvocationError::InvalidPublishedResult { - receipt: Receipt { - commit_sequence: 1, - .. - }, - .. - }) - )); - assert_eq!( - client - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output, - 1 - ); - - fixture.handle().drain().await.unwrap(); -} -#[test] -fn command_digest_is_canonical_and_binds_incarnation_identity_and_input() { - let description = CellDescription { - cell: crab_cell_runtime::CellId::from_bytes([1; 32]), - incarnation: IncarnationId::from_bytes([2; 16]), - code: Digest::from_bytes([3; 32]), - schema: 1, - }; - let identity = MutationIdentity { - request_id: RequestId::from_bytes([4; 16]), - issued_at_ms: 5, - expires_at_ms: 6, - }; - let mut encoder = BoundedEncoder::new(64).unwrap(); - b"input".to_vec().encode(&mut encoder).unwrap(); - let input = encoder.finish(); - let digest = command_operation_digest::(description, identity, &input).unwrap(); - assert_eq!( - *digest.as_bytes(), - [ - 220, 198, 211, 0, 239, 12, 190, 71, 247, 29, 242, 5, 254, 27, 25, 124, 91, 249, 225, - 28, 47, 142, 161, 104, 78, 125, 152, 205, 181, 67, 244, 134, - ] - ); - assert_ne!( - digest, - command_operation_digest::( - CellDescription { - incarnation: IncarnationId::from_bytes([9; 16]), - ..description - }, - identity, - &input, - ) - .unwrap() - ); - assert_ne!( - digest, - command_operation_digest::(description, identity, b"other").unwrap() - ); -} diff --git a/crates/crab-cell-runtime/tests/qualification.rs b/crates/crab-cell-runtime/tests/qualification.rs deleted file mode 100644 index 2b18a9094..000000000 --- a/crates/crab-cell-runtime/tests/qualification.rs +++ /dev/null @@ -1,8 +0,0 @@ -//! Qualification receipt and preflight contract tests. - -mod qualification { - pub mod cluster; - pub mod preflight; - pub mod properties; - pub mod receipt; -} diff --git a/crates/crab-cell-runtime/tests/qualification/cluster.rs b/crates/crab-cell-runtime/tests/qualification/cluster.rs deleted file mode 100644 index c5d7f8d7c..000000000 --- a/crates/crab-cell-runtime/tests/qualification/cluster.rs +++ /dev/null @@ -1,627 +0,0 @@ -//! Cluster qualification receipt acceptance and rejection. -//! -//! `validate_cluster_receipt` gates the four-process failover receipt before a -//! protected provider workflow accepts it as release evidence, so the envelope -//! and every evidence clause need a case that proves the receipt is rejected -//! when that clause does not hold. - -use crab_cell_runtime::identity::Digest; -use crab_cell_runtime::qualification::cluster::validate_cluster_receipt; -use serde_json::{Value, json}; - -const SOURCE_REVISION: &str = "0123456789abcdef0123456789abcdef01234567"; -const IMAGE: Digest = Digest::from_bytes([0xaa; 32]); -const PROJECT: &str = "crab-http-cluster-qualification-receipt-contract"; -const MAX_RECEIPT_BYTES: usize = 8 << 20; - -fn repeated(byte: u8, bytes: usize) -> String { - format!("{byte:02x}").repeat(bytes) -} - -fn session(byte: u8) -> String { - repeated(byte, 16) -} - -fn node(byte: u8) -> String { - repeated(byte, 16) -} - -fn image_digest() -> String { - format!("sha256:{}", repeated(0xaa, 32)) -} - -fn root(byte: u8, checksum: u64, txid: u64, commit_sequence: u64) -> Value { - json!({ - "digest": repeated(byte, 32), - "checksum": checksum, - "txid": txid, - "commit_sequence": commit_sequence, - }) -} - -fn timing() -> Value { - json!({ - "owner_killed_ms": 10, - "advertisement_expired_ms": 20, - "recovery_sealed_ms": 30, - "first_served_ms": 40, - }) -} - -fn work_cycle() -> Value { - json!({ - "candidate_count": 1, - "affected_cells": 1, - "catalog_shards": 1, - "catalog_pages": 1, - "control_reads": 1, - "follower_pages": 1, - "follower_frames": 1, - "follower_bytes": 1, - "peer_requests": 1, - "bundle_bytes": 1, - "object_reads": 1, - "object_writes": 1, - "phases": { - "claim": {"count": 1, "duration_ms": 1}, - "witness": {"count": 1, "duration_ms": 1}, - "scope_validation": {"count": 1, "duration_ms": 1}, - "pin_attach": {"count": 1, "duration_ms": 1}, - "seal": {"count": 1, "duration_ms": 1}, - }, - }) -} - -fn empty_work_cycle() -> Value { - let mut cycle = work_cycle(); - for field in [ - "candidate_count", - "affected_cells", - "catalog_shards", - "catalog_pages", - "control_reads", - "follower_pages", - "follower_frames", - "follower_bytes", - "peer_requests", - "bundle_bytes", - "object_reads", - "object_writes", - ] { - cycle[field] = json!(0); - } - for phase in ["claim", "witness", "scope_validation", "pin_attach", "seal"] { - cycle["phases"][phase] = json!({"count": 0, "duration_ms": 0}); - } - cycle -} - -fn selection_cycle( - failed_session: &str, - successor_session: &str, - failed_node: &str, - successor_node: &str, - members: &[String], - original_follower: bool, -) -> Value { - json!({ - "failed_session": failed_session, - "successor_session": successor_session, - "failed_node": failed_node, - "successor_node": successor_node, - "failed_log_members": members, - "selected_original_follower": original_follower, - "terminal_result": "succeeded", - }) -} - -/// One internally consistent four-process failover receipt. -fn canonical_receipt() -> Value { - let failed_session = session(0x11); - let successor_session = session(0x22); - let second_successor_session = session(0x33); - let fallback_successor_session = session(0x44); - let failed_node = node(0xa1); - let successor_node = node(0xb2); - let second_successor_node = node(0xc3); - let fallback_successor_node = node(0xd4); - let first_root = root(0x11, 10, 1, 1); - let restored_root = root(0x22, 20, 2, 2); - let continued_root = root(0x33, 30, 3, 3); - let second_root = root(0x44, 40, 4, 4); - let fallback_root = root(0x55, 50, 5, 5); - let first_members = vec![ - failed_node.clone(), - successor_node.clone(), - second_successor_node.clone(), - ]; - let node_capacity = || { - json!({ - "version": 1, - "resources": {"memory_bytes": 1_000, "free_disk_bytes": 1_000}, - "admission": {"active_cells": 1}, - }) - }; - let placement_node = |session: &str, node: &str| { - json!({ - "live": true, - "session": session, - "advertisement": {"node": node, "placement": {}, "log": {}}, - }) - }; - json!({ - "version": 6, - "source_revision": SOURCE_REVISION, - "image": {"reference": "source-only", "digest": image_digest()}, - "project": PROJECT, - "owner_loss": { - "session_before": failed_session, - "session_after": successor_session, - "epoch_before": 1, - "epoch_after": 2, - "timing": timing(), - "root_before": first_root, - "root_after_restore": restored_root, - "root_continued": continued_root, - }, - "fleet_only_commit": { - "owner_uncovered_bytes": 1, - "follower_retained_bytes": 1, - "immutable_object_put_rejected": true, - "owner_disk_removed_before_policy_restore": true, - "control_before_owner_loss": { - "state": "serving", - "epoch": 1, - "owner": {"session": failed_session}, - "owner_lease": {"state": "live", "observed_at_ms": 5, "expires_at_ms": 10}, - "root": first_root, - }, - "node_log_before": {"member_nodes": first_members}, - "node_log_after": {"state": "open", "epoch": 2, "active": true, "member_nodes": first_members}, - "response": {"id": 1}, - "restored_labels": {"items": ["owner-loss"]}, - }, - "follower_replacement": { - "node_log_before": {"epoch": 1, "member_nodes": first_members}, - "node_log_after": {"epoch": 2, "member_nodes": [second_successor_node]}, - "object_covered_response": {"number": 3}, - "fleet_only_response": {"id": 2}, - }, - "second_owner_loss": { - "session_before": successor_session, - "session_after": second_successor_session, - "epoch_before": 2, - "epoch_after": 3, - "timing": timing(), - "root_before": continued_root, - "root_after": second_root, - "restored_labels": {"items": ["first", "second"]}, - }, - "fallback_owner_loss": { - "session_before": second_successor_session, - "session_after": fallback_successor_session, - "epoch_before": 3, - "epoch_after": 4, - "timing": timing(), - "root_before": second_root, - "root_after": fallback_root, - "node_log_before": {"active": true, "member_nodes": first_members}, - "response": {"id": 3}, - "restored_labels": {"items": ["first", "second", "third"]}, - "candidate_record": { - "live": true, - "session": fallback_successor_session, - "advertisement": {"node": fallback_successor_node}, - }, - }, - "capacity": { - "node_a": node_capacity(), - "node_b": node_capacity(), - "node_c": node_capacity(), - "node_d": node_capacity(), - }, - "measured_disk": { - "node_a_bytes": 1_000, - "node_b_bytes": 1_000, - "node_c_bytes": 1_000, - "node_d_bytes": 1_000, - "tolerance_bytes": 1, - }, - "placement": { - "node_a": placement_node(&failed_session, &failed_node), - "node_b": placement_node(&successor_session, &successor_node), - "node_c": placement_node(&second_successor_session, &second_successor_node), - "node_d": placement_node(&fallback_successor_session, &fallback_successor_node), - "node_a_session": failed_session, - "node_b_session": successor_session, - "node_c_session": second_successor_session, - "node_d_session": fallback_successor_session, - }, - "metrics": { - "node_a": "crab_cell_durability_proofs_total 1", - "node_b": "crab_cell_durability_proofs_total 1", - "node_c": "crab_cell_durability_proofs_total 1", - "node_d": "crab_cell_durability_proofs_total 1", - }, - "capacity_metric_parity": { - "local_disk": true, - "active_cells": true, - "measured_local_disk": true, - "signed_placement": true, - }, - "selection": { - "owner_loss": selection_cycle( - &failed_session, - &successor_session, - &failed_node, - &successor_node, - &first_members, - true, - ), - "second_owner_loss": selection_cycle( - &successor_session, - &second_successor_session, - &successor_node, - &second_successor_node, - std::slice::from_ref(&second_successor_node), - true, - ), - "fallback": selection_cycle( - &second_successor_session, - &fallback_successor_session, - &second_successor_node, - &fallback_successor_node, - &first_members, - false, - ), - }, - "work": { - "owner_loss": work_cycle(), - "second_owner_loss": work_cycle(), - "fallback": work_cycle(), - }, - }) -} - -fn expect_rejected(receipt: &Value, published_image: bool, expected: &str) { - let bytes = serde_json::to_vec(receipt).unwrap(); - let error = validate_cluster_receipt(&bytes, SOURCE_REVISION, IMAGE, published_image) - .err() - .unwrap_or_else(|| panic!("receipt was accepted but must be rejected: {expected}")); - assert!( - error.to_string().contains(expected), - "expected an error containing {expected:?}, got {error}" - ); -} - -#[test] -fn canonical_failover_receipt_is_accepted() { - let bytes = serde_json::to_vec(&canonical_receipt()).unwrap(); - assert!(validate_cluster_receipt(&bytes, SOURCE_REVISION, IMAGE, false).is_ok()); - - let mut published = canonical_receipt(); - published["image"]["reference"] = json!(format!( - "ghcr.io/crabbuild/crab-http-server@{}", - image_digest() - )); - let bytes = serde_json::to_vec(&published).unwrap(); - assert!(validate_cluster_receipt(&bytes, SOURCE_REVISION, IMAGE, true).is_ok()); -} - -#[test] -fn follower_selection_uses_the_log_that_protected_the_acknowledged_commit() { - let mut receipt = canonical_receipt(); - // A fully covered log can be replaced before the follower-only write. - // Its old members must not override the active log recorded after that write. - receipt["fleet_only_commit"]["node_log_before"] = - json!({"state": "open", "epoch": 1, "active": false, "member_nodes": [node(0xc3)]}); - let bytes = serde_json::to_vec(&receipt).unwrap(); - assert!(validate_cluster_receipt(&bytes, SOURCE_REVISION, IMAGE, false).is_ok()); -} - -#[test] -fn follower_selection_requires_an_active_acknowledging_log() { - for (field, value) in [ - ("state", json!("closed")), - ("epoch", json!(0)), - ("active", json!(false)), - ("member_nodes", json!([])), - ] { - let mut receipt = canonical_receipt(); - receipt["fleet_only_commit"]["node_log_after"][field] = value; - expect_rejected(&receipt, false, "fleet-only active log"); - } -} - -#[test] -fn receipt_envelope_rejects_malformed_and_mismatched_bytes() { - assert!(validate_cluster_receipt(&[], SOURCE_REVISION, IMAGE, false).is_err()); - assert!( - validate_cluster_receipt( - &vec![b'{'; MAX_RECEIPT_BYTES + 1], - SOURCE_REVISION, - IMAGE, - false - ) - .is_err() - ); - assert!(validate_cluster_receipt(b"{", SOURCE_REVISION, IMAGE, false).is_err()); - - let mut version = canonical_receipt(); - version["version"] = json!(5); - expect_rejected(&version, false, "receipt version"); - - let mut revision = canonical_receipt(); - revision["source_revision"] = json!(repeated(0x01, 20)); - expect_rejected(&revision, false, "source revision"); - - let mut digest = canonical_receipt(); - digest["image"]["digest"] = json!(format!("sha256:{}", repeated(0xbb, 32))); - expect_rejected(&digest, false, "image digest"); - - let mut published_under_source_mode = canonical_receipt(); - published_under_source_mode["image"]["reference"] = json!(format!( - "ghcr.io/crabbuild/crab-http-server@{}", - image_digest() - )); - expect_rejected(&published_under_source_mode, false, "source image"); - - let source_under_published_mode = canonical_receipt(); - expect_rejected(&source_under_published_mode, true, "image reference"); - - let mut project = canonical_receipt(); - project["project"] = json!("cluster-qualification"); - expect_rejected(&project, false, "project"); - - let mut unknown_field = canonical_receipt(); - unknown_field["unexpected_evidence"] = json!(true); - let bytes = serde_json::to_vec(&unknown_field).unwrap(); - assert!(validate_cluster_receipt(&bytes, SOURCE_REVISION, IMAGE, false).is_err()); -} - -#[test] -fn receipt_evidence_clauses_reject_tampering() { - type Mutation = fn(&mut Value); - let cases: &[(&str, &str, Mutation)] = &[ - ( - "owner loss without a session change", - "did not change session", - |receipt| { - receipt["owner_loss"]["session_after"] = json!(session(0x11)); - }, - ), - ( - "owner loss without an epoch advance", - "epoch did not advance", - |receipt| { - receipt["owner_loss"]["epoch_after"] = json!(1); - }, - ), - ( - "owner loss where restore did not advance the root", - "did not advance", - |receipt| { - receipt["owner_loss"]["root_after_restore"] = root(0x11, 10, 1, 1); - }, - ), - ( - "two owner losses from unrelated histories", - "owner chain", - |receipt| { - receipt["owner_loss"]["session_after"] = json!(session(0x99)); - }, - ), - ( - "fleet-only commit bound to another owner session", - "owner chain", - |receipt| { - receipt["fleet_only_commit"]["control_before_owner_loss"]["owner"]["session"] = - json!(session(0xee)); - }, - ), - ( - "fleet-only commit bound to another root", - "root chain", - |receipt| { - receipt["fleet_only_commit"]["control_before_owner_loss"]["root"]["digest"] = - json!(repeated(0xee, 32)); - }, - ), - ( - "second owner loss that regresses the root", - "root chain", - |receipt| { - receipt["second_owner_loss"]["root_before"] = root(0x22, 20, 2, 2); - }, - ), - ( - "owner-loss timing that expires before the kill", - "timing", - |receipt| { - receipt["owner_loss"]["timing"]["owner_killed_ms"] = json!(50); - }, - ), - ( - "fleet-only commit without uncovered bytes", - "fleet-only durability evidence", - |receipt| { - receipt["fleet_only_commit"]["owner_uncovered_bytes"] = json!(0); - }, - ), - ( - "fleet-only commit with an expired owner lease", - "fleet-only owner lease", - |receipt| { - receipt["fleet_only_commit"]["control_before_owner_loss"]["owner_lease"]["state"] = - json!("expired"); - }, - ), - ( - "fleet-only response that is not the first command", - "fleet-only response evidence", - |receipt| { - receipt["fleet_only_commit"]["response"]["id"] = json!(2); - }, - ), - ( - "follower replacement that does not advance the log", - "follower replacement log", - |receipt| { - receipt["follower_replacement"]["node_log_after"]["epoch"] = json!(1); - }, - ), - ( - "follower replacement with the wrong object response", - "follower replacement response", - |receipt| { - receipt["follower_replacement"]["object_covered_response"]["number"] = json!(2); - }, - ), - ( - "second owner loss with too few restored labels", - "second owner loss labels", - |receipt| { - receipt["second_owner_loss"]["restored_labels"]["items"] = json!(["first"]); - }, - ), - ( - "fallback response that is not the third command", - "fallback owner-loss evidence", - |receipt| { - receipt["fallback_owner_loss"]["response"]["id"] = json!(2); - }, - ), - ( - "capacity report from another schema", - "capacity schema", - |receipt| { - receipt["capacity"]["node_c"]["version"] = json!(2); - }, - ), - ( - "capacity report without usable memory", - "capacity resources", - |receipt| { - receipt["capacity"]["node_b"]["resources"]["memory_bytes"] = json!(0); - }, - ), - ( - "measured disk without one node", - "cluster receipt field", - |receipt| { - receipt["measured_disk"] - .as_object_mut() - .unwrap() - .remove("node_d_bytes"); - }, - ), - ( - "placement with a node that is not live", - "placement node is not live", - |receipt| { - receipt["placement"]["node_d"]["live"] = json!(false); - }, - ), - ( - "placement without a signed log", - "cluster receipt field", - |receipt| { - receipt["placement"]["node_a"]["advertisement"] - .as_object_mut() - .unwrap() - .remove("log"); - }, - ), - ( - "metrics that leak a node identifier", - "metrics payload contains an identifier", - |receipt| { - receipt["metrics"]["node_b"] = - json!("crab_cell_node_log_bytes_total{node=\"b\"} 1"); - }, - ), - ( - "metrics payload that is empty", - "metrics payload", - |receipt| { - receipt["metrics"]["node_c"] = json!(""); - }, - ), - ( - "capacity metric parity that is not proven", - "capacity metric parity", - |receipt| { - receipt["capacity_metric_parity"]["local_disk"] = json!(false); - }, - ), - ( - "successor that was not an original follower", - "successor was not an original follower", - |receipt| { - let members = vec![node(0xa1), node(0xc3)]; - receipt["fleet_only_commit"]["node_log_after"]["member_nodes"] = json!(members); - }, - ), - ( - "owner-loss selection that names another node", - "owner-loss selection identity", - |receipt| { - receipt["selection"]["owner_loss"]["failed_node"] = json!(node(0xe5)); - }, - ), - ( - "owner-loss selection that drops the successor", - "owner-loss successor is not a follower", - |receipt| { - receipt["selection"]["owner_loss"]["failed_log_members"] = json!([node(0xa1)]); - }, - ), - ( - "second owner-loss selection outside the replaced log", - "second owner-loss selection identity", - |receipt| { - receipt["selection"]["second_owner_loss"]["successor_node"] = json!(node(0xd4)); - }, - ), - ( - "fallback selection that claims an original follower", - "cluster selection evidence", - |receipt| { - receipt["selection"]["fallback"]["selected_original_follower"] = json!(true); - }, - ), - ( - "recovery work without candidates", - "cluster recovery work counter", - |receipt| { - receipt["work"]["owner_loss"]["candidate_count"] = json!(0); - }, - ), - ( - "recovery work without a seal phase", - "cluster recovery phase evidence", - |receipt| { - receipt["work"]["owner_loss"]["phases"]["seal"]["count"] = json!(0); - }, - ), - ( - "fallback recovery work that is empty", - "cluster recovery work counter", - |receipt| { - receipt["work"]["fallback"] = empty_work_cycle(); - }, - ), - ]; - for (name, expected, mutate) in cases { - let mut receipt = canonical_receipt(); - mutate(&mut receipt); - let bytes = serde_json::to_vec(&receipt).unwrap(); - let error = validate_cluster_receipt(&bytes, SOURCE_REVISION, IMAGE, false) - .err() - .unwrap_or_else(|| panic!("{name}: receipt was accepted")); - assert!( - error.to_string().contains(expected), - "{name}: expected an error containing {expected:?}, got {error}" - ); - } -} diff --git a/crates/crab-cell-runtime/tests/qualification/preflight.rs b/crates/crab-cell-runtime/tests/qualification/preflight.rs deleted file mode 100644 index 4c540a3da..000000000 --- a/crates/crab-cell-runtime/tests/qualification/preflight.rs +++ /dev/null @@ -1,49 +0,0 @@ -#[test] -fn production_hot_paths_use_singleton_or_indexed_work() { - let queue = include_str!("../../src/primitives/queue.rs"); - let queue_info = queue - .split_once("pub fn queue_info") - .and_then(|(_, tail)| tail.split_once("pub fn verify_queue_counts")) - .map(|(body, _)| body) - .expect("queue info function is present"); - assert!(!queue_info.contains("count(")); - assert!(queue_info.contains("ready_count")); - - let workflow = include_str!("../../src/primitives/workflow.rs"); - let next_sequence = workflow - .split_once("fn next_sequence") - .and_then(|(_, tail)| tail.split_once("pub fn verify_workflow_event_count")) - .map(|(body, _)| body) - .expect("workflow sequence allocator is present"); - assert!(!next_sequence.contains("count(")); - assert!(next_sequence.contains("event_count")); - - let effects = include_str!("../../src/primitives/effects.rs"); - let insertion = effects - .split_once("fn effect_insert") - .and_then(|(_, tail)| tail.split_once("fn effect_operation")) - .map(|(body, _)| body) - .expect("effect insertion function is present"); - assert!(!insertion.contains("SELECT count")); - assert!(effects.contains("operation_bytes")); -} - -#[test] -fn qualification_workload_and_profile_are_reproducible() { - use crab_cell_runtime::qualification::{QualificationProfile, QualificationWorkload}; - - let profile = QualificationProfile::pr_contract(); - let workload = QualificationWorkload::generate(&profile, 17).expect("workload generation"); - workload - .verify_for_profile(&profile) - .expect("workload identity"); - assert_eq!( - workload.profile_digest(), - profile.digest().expect("profile digest") - ); - assert_eq!( - QualificationWorkload::decode(&workload.encode().expect("workload encoding")) - .expect("workload decoding"), - workload - ); -} diff --git a/crates/crab-cell-runtime/tests/qualification/properties.rs b/crates/crab-cell-runtime/tests/qualification/properties.rs deleted file mode 100644 index bfce72356..000000000 --- a/crates/crab-cell-runtime/tests/qualification/properties.rs +++ /dev/null @@ -1,63 +0,0 @@ -//! Properties of the qualification decoders and the receipt gate. -//! -//! Profile, workload, and receipt bytes arrive from evidence files, CI -//! artifacts, and other machines, so the contracts that matter beyond the -//! fixtures are that a decode never panics and that the canonical form is -//! stable: re-encoding an accepted artifact and decoding it again must land on -//! the same bytes. - -use crab_cell_runtime::identity::Digest; -use crab_cell_runtime::qualification::cluster::validate_cluster_receipt; -use crab_cell_runtime::qualification::{QualificationProfile, QualificationWorkload}; -use proptest::prelude::*; - -const SOURCE_REVISION: &str = "0123456789abcdef0123456789abcdef01234567"; -const IMAGE: Digest = Digest::from_bytes([0xaa; 32]); - -proptest! { - #![proptest_config(ProptestConfig { cases: 256, ..ProptestConfig::default() })] - - /// An accepted profile has a canonical form that re-parses and re-encodes - /// unchanged, whatever bytes it arrived as. - #[test] - fn profile_decoding_has_a_stable_canonical_form( - bytes in prop::collection::vec(any::(), 0..1_024), - ) { - if let Ok(profile) = QualificationProfile::decode(&bytes) { - let encoded = profile.encode().expect("an accepted profile encodes"); - let reparsed = QualificationProfile::decode(&encoded) - .expect("the canonical form of an accepted profile decodes"); - prop_assert_eq!( - reparsed.encode().expect("the canonical form re-encodes"), - encoded - ); - prop_assert!(reparsed == profile); - } - } - - /// The same for the canonical workload artifact. - #[test] - fn workload_decoding_has_a_stable_canonical_form( - bytes in prop::collection::vec(any::(), 0..1_024), - ) { - if let Ok(workload) = QualificationWorkload::decode(&bytes) { - let encoded = workload.encode().expect("an accepted workload encodes"); - let reparsed = QualificationWorkload::decode(&encoded) - .expect("the canonical form of an accepted workload decodes"); - prop_assert_eq!( - reparsed.encode().expect("the canonical form re-encodes"), - encoded - ); - } - } - - /// The release-evidence gate parses untrusted bytes: it may accept or - /// reject, but it must return rather than panic. - #[test] - fn the_cluster_receipt_gate_is_total( - bytes in prop::collection::vec(any::(), 0..4_096), - published in any::(), - ) { - let _ = validate_cluster_receipt(&bytes, SOURCE_REVISION, IMAGE, published); - } -} diff --git a/crates/crab-cell-runtime/tests/qualification/receipt.rs b/crates/crab-cell-runtime/tests/qualification/receipt.rs deleted file mode 100644 index dc28631f1..000000000 --- a/crates/crab-cell-runtime/tests/qualification/receipt.rs +++ /dev/null @@ -1,438 +0,0 @@ -use std::{future::Future, pin::Pin, time::Duration}; - -use crab_cell_runtime::Result; -use crab_cell_runtime::identity::Digest; -use crab_cell_runtime::qualification::{ - QUALIFICATION_PROTECTED_EVIDENCE_MAX_AGE_MS, - QUALIFICATION_PROTECTED_EVIDENCE_MAX_CLOCK_SKEW_MS, QualificationOwnership, - QualificationProviderEvidence, QualificationReceipt, QualificationRunner, -}; -use crab_cell_runtime::qualification::{ - QualificationExecution, QualificationOperation, QualificationOperationExecutor, - QualificationProfile, QualificationWorkload, -}; -use ed25519_dalek::SigningKey; - -const SOURCE: &str = "qualification-source"; -const IMAGE: Digest = Digest::from_bytes([7; 32]); -const MATRIX_ROWS: [&str; 10] = [ - "protocol", - "storage", - "publication", - "warm-path", - "churn", - "fleet", - "failover", - "primitives", - "accounting", - "compatibility", -]; - -fn receipt(workload: &str, artifact: &[u8]) -> QualificationReceipt { - QualificationRunner::new(SigningKey::from_bytes(&[11; 32])) - .emit_with_profile( - &QualificationProfile::pr_contract(), - SOURCE.into(), - IMAGE, - "local".into(), - workload.into(), - "none".into(), - Vec::new(), - artifact, - true, - ( - "rustc".into(), - "test".into(), - "local".into(), - 0, - 0, - 0, - false, - ), - ) - .expect("fixture receipt") -} - -struct MeasuredExecutor; - -impl QualificationOperationExecutor for MeasuredExecutor { - type Future<'a> = Pin> + Send + 'a>>; - - fn execute<'a>(&'a mut self, _operation: QualificationOperation) -> Self::Future<'a> { - Box::pin(async { - tokio::time::sleep(Duration::from_millis(150)).await; - Ok(QualificationExecution::acknowledged(true)) - }) - } -} - -#[test] -fn workload_identity_is_reproducible_and_seed_bound() { - let profile = QualificationProfile::pr_contract(); - let first = QualificationWorkload::generate(&profile, 17).expect("workload generation"); - let same = QualificationWorkload::generate(&profile, 17).expect("same workload generation"); - let changed = - QualificationWorkload::generate(&profile, 18).expect("changed workload generation"); - - assert_eq!(first, same); - assert_eq!(first.outcome_digest(), same.outcome_digest()); - assert_ne!(first.outcome_digest(), changed.outcome_digest()); - first - .verify_for_profile(&profile) - .expect("canonical workload identity"); - assert_eq!( - QualificationWorkload::decode(&first.encode().expect("workload encoding")) - .expect("workload decoding"), - first - ); -} - -#[test] -fn public_matrix_verifier_rejects_missing_duplicate_and_forged_evidence() { - let artifacts = MATRIX_ROWS - .iter() - .map(|row| format!("artifact-{row}").into_bytes()) - .collect::>(); - let receipts = MATRIX_ROWS - .iter() - .enumerate() - .map(|(index, row)| receipt(row, &artifacts[index])) - .collect::>(); - let artifact_views = artifacts - .iter() - .map(|artifact| vec![artifact.as_slice()]) - .collect::>(); - let evidence = MATRIX_ROWS - .iter() - .enumerate() - .map(|(index, row)| (*row, &receipts[index], artifact_views[index].as_slice())) - .collect::>(); - - QualificationReceipt::verify_matrix(SOURCE, IMAGE, &evidence).expect("complete matrix"); - - let mut missing = evidence.clone(); - missing.pop(); - assert!(QualificationReceipt::verify_matrix(SOURCE, IMAGE, &missing).is_err()); - - let mut duplicate = evidence.clone(); - duplicate[1].0 = MATRIX_ROWS[0]; - assert!(QualificationReceipt::verify_matrix(SOURCE, IMAGE, &duplicate).is_err()); - - let mut forged_artifact = artifacts[0].clone(); - forged_artifact.push(b'!'); - let forged_view = [forged_artifact.as_slice()]; - let mut forged = evidence; - forged[0].2 = &forged_view; - assert!(QualificationReceipt::verify_matrix(SOURCE, IMAGE, &forged).is_err()); -} - -#[test] -fn public_receipt_verifier_binds_source_image_and_artifact() { - let artifact = b"raw-evidence"; - let receipt = receipt("protocol", artifact); - receipt - .verify_for(SOURCE, IMAGE, artifact) - .expect("exact release identity"); - assert!(receipt.verify_for("other-source", IMAGE, artifact).is_err()); - assert!( - receipt - .verify_for(SOURCE, Digest::from_bytes([8; 32]), artifact) - .is_err() - ); - assert!( - receipt - .verify_for(SOURCE, IMAGE, b"forged-evidence") - .is_err() - ); - assert!( - receipt - .verify_for_trusted_signer(SOURCE, IMAGE, artifact, [12; 32]) - .is_err() - ); -} - -#[test] -fn protected_evidence_freshness_rejects_stale_and_future_receipts() { - let current = receipt("protocol", b"raw-evidence") - .with_evidence( - 900, - 1_000, - b"none", - vec![Digest::from_bytes( - *blake3::hash(b"raw-evidence").as_bytes(), - )], - Vec::new(), - ) - .expect("evidence timestamps"); - - current.verify_fresh_at(1_000).expect("current evidence"); - current - .verify_fresh_at(1_000 + QUALIFICATION_PROTECTED_EVIDENCE_MAX_AGE_MS) - .expect("boundary evidence"); - assert!( - current - .verify_fresh_at(1_001 + QUALIFICATION_PROTECTED_EVIDENCE_MAX_AGE_MS) - .is_err() - ); - let future = receipt("protocol", b"raw-evidence") - .with_evidence( - QUALIFICATION_PROTECTED_EVIDENCE_MAX_CLOCK_SKEW_MS + 1_000, - QUALIFICATION_PROTECTED_EVIDENCE_MAX_CLOCK_SKEW_MS + 2_000, - b"none", - vec![Digest::from_bytes( - *blake3::hash(b"raw-evidence").as_bytes(), - )], - Vec::new(), - ) - .expect("future evidence timestamps"); - assert!(future.verify_fresh_at(1_000).is_err()); - assert!(current.verify_fresh_at(0).is_err()); -} - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn public_protected_matrix_binds_run_artifact_profile_and_signer() { - let mut profile_value = serde_json::to_value( - QualificationProfile::new("protected-contract".into(), 1, 8, 1, 5_000) - .expect("protected profile"), - ) - .expect("protected profile value"); - profile_value["provider"] = serde_json::Value::String("protected-provider".into()); - let profile: QualificationProfile = - serde_json::from_value(profile_value).expect("named protected profile"); - let workload = QualificationWorkload::generate_with_size(&profile, 91, 1, 8, 1) - .expect("protected workload"); - let mut executor = MeasuredExecutor; - let summary = workload - .run(&mut executor) - .await - .expect("measured workload"); - let run = summary.artifact(&workload).expect("run artifact"); - let workload_bytes = workload.encode().expect("workload encoding"); - let run_bytes = run.encode().expect("run encoding"); - let provider_evidence = - QualificationProviderEvidence::new(&profile, workload.seed(), true, true, true) - .expect("provider evidence") - .encode() - .expect("provider evidence encoding"); - run.verify_for_profile(&profile) - .expect("measured run thresholds"); - - let signing_key = SigningKey::from_bytes(&[12; 32]); - let trusted_signer = signing_key.verifying_key().to_bytes(); - let image = Digest::from_bytes([13; 32]); - let mut receipts = Vec::with_capacity(MATRIX_ROWS.len()); - let mut artifacts = Vec::with_capacity(MATRIX_ROWS.len()); - for workload_name in MATRIX_ROWS { - let row_artifacts = if workload_name == "primitives" { - vec![ - workload_bytes.clone(), - run_bytes.clone(), - provider_evidence.clone(), - ] - } else { - vec![format!("protected-{workload_name}").into_bytes()] - }; - let metrics = if workload_name == "primitives" { - summary.metrics().expect("run metrics") - } else { - vec![ - crab_cell_runtime::qualification::QualificationMetric::new( - "cells".into(), - 1, - "cells".into(), - ) - .expect("cell metric"), - crab_cell_runtime::qualification::QualificationMetric::new( - "operations".into(), - 8, - "operations".into(), - ) - .expect("operation metric"), - crab_cell_runtime::qualification::QualificationMetric::new( - "duration_secs".into(), - 1, - "seconds".into(), - ) - .expect("duration metric"), - crab_cell_runtime::qualification::QualificationMetric::new( - "p99_latency_ms".into(), - 1, - "ms".into(), - ) - .expect("latency metric"), - ] - }; - let primary = row_artifacts[0].clone(); - let raw_digests = row_artifacts - .iter() - .map(|artifact| Digest::from_bytes(*blake3::hash(artifact).as_bytes())) - .collect(); - let receipt = QualificationRunner::new(signing_key.clone()) - .emit_with_profile_and_evidence( - &profile, - "protected-source".into(), - image, - "protected-provider".into(), - workload_name.into(), - "none".into(), - metrics, - &primary, - true, - ( - "rustc".into(), - "release".into(), - "protected-topology".into(), - if workload_name == "primitives" { - workload.seed() - } else { - 0 - }, - 1, - 1, - false, - ), - 1, - 1_001, - b"none", - raw_digests, - vec![QualificationOwnership::new( - 1, - 1, - Digest::from_bytes([14; 32]), - )], - ) - .expect("protected receipt"); - receipts.push(receipt); - artifacts.push(row_artifacts); - } - - let artifact_views = artifacts - .iter() - .map(|row| row.iter().map(Vec::as_slice).collect::>()) - .collect::>(); - let evidence = MATRIX_ROWS - .iter() - .enumerate() - .map(|(index, workload_name)| { - ( - *workload_name, - &receipts[index], - artifact_views[index].as_slice(), - ) - }) - .collect::>(); - QualificationReceipt::verify_matrix_for_profile_with_signer( - "protected-source", - image, - &profile, - &evidence, - trusted_signer, - ) - .expect("complete protected matrix"); - QualificationReceipt::verify_matrix_for_profile_with_signer_fresh_at( - "protected-source", - image, - &profile, - &evidence, - trusted_signer, - 2, - ) - .expect("fresh protected matrix"); - assert!( - QualificationReceipt::verify_matrix_for_profile_with_signer_fresh_at( - "protected-source", - image, - &profile, - &evidence, - trusted_signer, - 1_001 + QUALIFICATION_PROTECTED_EVIDENCE_MAX_AGE_MS + 1, - ) - .is_err() - ); - let mismatched_metrics = summary - .metrics() - .expect("run metrics") - .into_iter() - .map(|metric| { - let value = if metric.name() == "p99_latency_ms" { - metric.value().saturating_add(1) - } else { - metric.value() - }; - crab_cell_runtime::qualification::QualificationMetric::new( - metric.name().into(), - value, - metric.unit().into(), - ) - .expect("mismatched metric") - }) - .collect(); - let primitive_artifacts = &artifacts[7]; - let mismatched_primitive = QualificationRunner::new(signing_key.clone()) - .emit_with_profile_and_evidence( - &profile, - "protected-source".into(), - image, - "protected-provider".into(), - "primitives".into(), - "none".into(), - mismatched_metrics, - &primitive_artifacts[0], - true, - ( - "rustc".into(), - "release".into(), - "protected-topology".into(), - workload.seed(), - 1, - 1, - false, - ), - 1, - 2, - b"none", - primitive_artifacts - .iter() - .map(|artifact| Digest::from_bytes(*blake3::hash(artifact).as_bytes())) - .collect(), - vec![QualificationOwnership::new( - 1, - 1, - Digest::from_bytes([14; 32]), - )], - ) - .expect("mismatched protected receipt"); - let mut mismatched_receipts = receipts.clone(); - mismatched_receipts[7] = mismatched_primitive; - let mismatched_evidence = MATRIX_ROWS - .iter() - .enumerate() - .map(|(index, workload_name)| { - ( - *workload_name, - &mismatched_receipts[index], - artifact_views[index].as_slice(), - ) - }) - .collect::>(); - assert!( - QualificationReceipt::verify_matrix_for_profile_with_signer( - "protected-source", - image, - &profile, - &mismatched_evidence, - trusted_signer, - ) - .is_err() - ); - assert!( - QualificationReceipt::verify_matrix_for_profile_with_signer( - "protected-source", - image, - &profile, - &evidence, - [15; 32], - ) - .is_err() - ); -} diff --git a/crates/crab-cell-runtime/tests/runtime.rs b/crates/crab-cell-runtime/tests/runtime.rs deleted file mode 100644 index 8c2e06009..000000000 --- a/crates/crab-cell-runtime/tests/runtime.rs +++ /dev/null @@ -1,20 +0,0 @@ -//! Cell lifecycle, durability, and publication integration tests. -//! -//! The suite is one test binary. Its modules live in `tests/runtime/`; shared -//! fixtures stay in `tests/support/` and are reached through the crate root, so -//! no target needs a `#[path]` attribute. - -mod support; - -mod runtime { - pub mod backup; - pub mod catalog; - pub mod fault_fs; - pub mod lifecycle; - pub mod migration; - pub mod publication; - pub mod release_progress; - pub mod scheduler; - pub mod scheduler_properties; - pub mod workers; -} diff --git a/crates/crab-cell-runtime/tests/runtime/backup.rs b/crates/crab-cell-runtime/tests/runtime/backup.rs deleted file mode 100644 index c5ba5f93e..000000000 --- a/crates/crab-cell-runtime/tests/runtime/backup.rs +++ /dev/null @@ -1,222 +0,0 @@ -use crab_cell_runtime::cell::application::ApplicationIdentity; -use crab_cell_runtime::cell::catalog::CellCatalog; -use crab_cell_runtime::control::Control; -use crab_cell_runtime::identity::RequestId; -use crab_cell_runtime::ltx::{Host as ReplicaHost, Limits as ReplicaLimits}; -use crab_cell_runtime::recovery::backup::{BackupPinStore, PinnedCatalogShard}; -use std::sync::Arc; - -use crab_ltx::{CellObjectKind, CellStorageLayout}; -use crab_storage::Store; -use object_store::{ObjectStoreExt as _, memory::InMemory, path::Path}; - -use crab_cell_runtime::cell::application::ApplicationIdentityStore; -use crab_cell_runtime::cell::catalog::CatalogEntry; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::control::{ControlState, Owner}; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::recovery::release::ReleaseStore; - -async fn pinned_catalog(catalog: &CellCatalog) -> Vec { - let mut pinned = Vec::with_capacity(256); - for shard in 0_u8..=u8::MAX { - let mut scan = catalog.scan_shard(shard).await.unwrap(); - let revision = scan.revision(); - let pages = scan.page_digests().to_vec(); - while scan.next_page().await.unwrap().is_some() {} - pinned.push(PinnedCatalogShard { - shard, - revision, - pages, - }); - } - pinned -} - -#[tokio::test] -async fn pin_verifies_roots_and_fails_closed_when_a_dependency_is_missing() { - let backend = Arc::new(InMemory::new()); - let store = Store::new(backend.clone()); - let identity = ApplicationIdentity::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - ); - let layout = CellStorageLayout::new( - store, - Path::from("runtime"), - *identity.application().as_bytes(), - ); - let target = CellTarget::new( - identity.tenant(), - identity.application(), - NamespaceId::from_bytes([3; 16]), - b"repository", - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([4; 16]); - let limits = ReplicaLimits::default(); - let replica = crab_ltx::CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - limits, - ) - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let mut database = crab_ltx::Db::open(&directory.path().join("cell.sqlite"), limits).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE values_(value BLOB NOT NULL);\ - INSERT INTO values_ VALUES(randomblob(100000))", - ) - }) - .unwrap(); - let root = replica - .prepare(None, &database.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - database.close().unwrap(); - - let mut control = Control::initial( - cell, - incarnation, - Owner { - session: SessionId::from_bytes([5; 16]), - endpoint: "https://node.internal:8789".to_owned(), - }, - Digest::from_bytes([6; 32]), - 1, - ) - .unwrap(); - control.revision = 2; - control.progress = 2; - control.state = ControlState::Serving; - control.root = - Some(crab_cell_runtime::control::RootRef::from_ltx(cell, incarnation, root).unwrap()); - control.encode().unwrap(); - - let catalog = CellCatalog::new(layout.clone(), identity.tenant()); - catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Repository, - Digest::from_bytes([6; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let pinned = pinned_catalog(&catalog).await; - let descriptor = br#"{"runtime":"test","version":1}"#; - let descriptor_digest = Digest::from_bytes(*blake3::hash(descriptor).as_bytes()); - let releases = ReleaseStore::new(layout.clone(), identity).unwrap(); - let operation = RequestId::from_bytes([9; 16]); - let prepared = releases - .prepare( - descriptor, - descriptor_digest, - 0, - &format!("sha256:{}", "a".repeat(64)), - operation, - ) - .await - .unwrap(); - let activating = releases - .start_activation(prepared.revision(), operation) - .await - .unwrap(); - releases - .complete_activation(activating.revision(), operation) - .await - .unwrap(); - - let pins = - BackupPinStore::new(layout.clone(), identity, limits, ReplicaHost::default()).unwrap(); - let id = RequestId::from_bytes([7; 16]); - let pin = pins - .create(id, 1_000, pinned.clone(), vec![control.clone()]) - .await - .unwrap(); - assert_eq!(pin.control_count(), 1); - assert_eq!(pins.load(id).await.unwrap(), Some(pin.clone())); - assert_eq!(pins.controls(&pin).await.unwrap(), vec![control.clone()]); - assert_eq!(pins.verify(&pin).await.unwrap(), vec![control.clone()]); - assert_eq!( - pins.create(id, 1_000, pinned.clone(), vec![control.clone()]) - .await - .unwrap(), - pin - ); - - let destination_root = Path::from("restored"); - let restored = pins.restore(&pin, destination_root.clone()).await.unwrap(); - assert_eq!(restored.application(), identity.application()); - assert_eq!(restored.pin(), id); - assert_eq!(restored.control_count(), 1); - assert!(restored.immutable_object_count() > 0); - assert_eq!(restored.nonempty_catalog_shards(), 1); - assert_eq!( - pins.restore(&pin, destination_root.clone()).await.unwrap(), - restored - ); - - let restored_store = Store::new(backend.clone()); - let restored_identities = - ApplicationIdentityStore::new(restored_store.clone(), destination_root.clone()); - assert_eq!(restored_identities.load().await.unwrap(), Some(identity)); - let restored_layout = restored_identities.layout(identity).await.unwrap(); - let restored_control = CellAuthority::new(restored_layout.clone()) - .load(cell) - .await - .unwrap() - .unwrap(); - assert_eq!(restored_control.value().state, ControlState::Idle); - assert_eq!(restored_control.value().owner, None); - assert_eq!(restored_control.value().root, control.root); - assert_eq!( - ReleaseStore::new(restored_layout.clone(), identity) - .unwrap() - .load() - .await - .unwrap() - .unwrap() - .record(), - releases.load().await.unwrap().unwrap().record() - ); - - let missing = replica - .reachable_objects(&root) - .await - .unwrap() - .into_iter() - .find(|object| object.kind == CellObjectKind::Ltx) - .unwrap(); - let missing_path = layout.incarnation_object_path( - cell.as_bytes(), - incarnation.as_bytes(), - &missing.digest, - missing.kind, - ); - backend.delete(&missing_path).await.unwrap(); - let restored_pins = - BackupPinStore::new(restored_layout, identity, limits, ReplicaHost::default()).unwrap(); - let restored_pin = restored_pins.load(id).await.unwrap().unwrap(); - assert_eq!(restored_pins.verify(&restored_pin).await.unwrap().len(), 1); - let next = RequestId::from_bytes([8; 16]); - assert!( - pins.create(next, 2_000, pinned, vec![control]) - .await - .is_err() - ); - assert!(pins.load(next).await.unwrap().is_none()); - assert!(pins.verify(&pin).await.is_err()); -} diff --git a/crates/crab-cell-runtime/tests/runtime/catalog.rs b/crates/crab-cell-runtime/tests/runtime/catalog.rs deleted file mode 100644 index a18f66ff9..000000000 --- a/crates/crab-cell-runtime/tests/runtime/catalog.rs +++ /dev/null @@ -1,549 +0,0 @@ -use std::{collections::HashMap, sync::Arc}; - -use bytes::Bytes; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::catalog::{CatalogEntry, CellCatalog}; -use crab_cell_runtime::control::Owner; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::fleet::telemetry::CatalogReadKind; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_ltx::CellStorageLayout; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -fn fixture() -> (CellStorageLayout, CellCatalog, CellTarget) { - let tenant = TenantId::from_bytes([1; 16]); - let application = ApplicationId::from_bytes([2; 16]); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - *application.as_bytes(), - ); - let catalog = CellCatalog::new(layout.clone(), tenant); - let target = CellTarget::new( - tenant, - application, - NamespaceId::from_bytes([3; 16]), - b"repository-1", - ) - .unwrap(); - (layout, catalog, target) -} - -fn entry(target: &CellTarget, role: CatalogRole, byte: u8) -> CatalogEntry { - CatalogEntry::new(target, role, Digest::from_bytes([byte; 32]), 1).unwrap() -} - -#[tokio::test] -async fn catalogs_isolate_tenants_with_colliding_shards() { - let (layout, first_catalog, first) = fixture(); - let tenant = TenantId::from_bytes([99; 16]); - let shard = first.cell_id().as_bytes()[0]; - let second = (0_u32..10_000) - .map(|partition| { - CellTarget::new( - tenant, - first.application(), - first.namespace(), - &partition.to_be_bytes(), - ) - .unwrap() - }) - .find(|target| target.cell_id().as_bytes()[0] == shard) - .unwrap(); - let second_catalog = CellCatalog::new(layout.clone(), tenant); - for (catalog, target) in [(&first_catalog, &first), (&second_catalog, &second)] { - catalog - .provision(entry(target, CatalogRole::Sql, 4)) - .await - .unwrap(); - } - // Reopen both catalogs: no process-local routing state may hide a mixed head. - for (target, other) in [(&first, &second), (&second, &first)] { - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - assert_eq!( - catalog - .lookup(target.cell_id()) - .await - .unwrap() - .unwrap() - .entry() - .cell(), - target.cell_id() - ); - assert!(catalog.lookup(other.cell_id()).await.unwrap().is_none()); - let mut scan = catalog.scan_shard(shard).await.unwrap(); - let page = scan.next_page().await.unwrap().unwrap(); - assert_eq!(page.entries().len(), 1); - assert_eq!(page.entries()[0].entry().cell(), target.cell_id()); - assert!(scan.next_page().await.unwrap().is_none()); - } -} - -#[tokio::test] -async fn catalog_provision_is_idempotent_and_precedes_control() { - let (layout, catalog, target) = fixture(); - let expected = entry(&target, CatalogRole::Repository, 4); - let first = catalog.provision(expected.clone()).await.unwrap(); - let second = catalog.provision(expected.clone()).await.unwrap(); - assert_eq!(first.revision(), 1); - assert_eq!(second.revision(), 1); - assert_eq!(first.entry(), &expected); - assert_eq!( - catalog - .lookup(target.cell_id()) - .await - .unwrap() - .unwrap() - .entry(), - &expected - ); - let authority = CellAuthority::new(layout); - assert!(authority.load(target.cell_id()).await.unwrap().is_none()); - let owner = Owner { - session: SessionId::from_bytes([10; 16]), - endpoint: "https://node.internal:8081".into(), - }; - let created = authority - .create_initial(&first, IncarnationId::from_bytes([11; 16]), owner.clone()) - .await - .unwrap(); - let adopted = authority - .create_initial(&first, IncarnationId::from_bytes([11; 16]), owner) - .await - .unwrap(); - assert_eq!(created.value(), adopted.value()); - assert!(matches!( - authority - .create_initial( - &first, - IncarnationId::from_bytes([11; 16]), - Owner { - session: SessionId::from_bytes([12; 16]), - endpoint: "https://other.internal:8081".into(), - }, - ) - .await, - Err(crab_cell_runtime::Error::CellAlreadyActive) - )); -} - -#[tokio::test] -async fn concurrent_catalog_writers_merge_entries_on_one_shard() { - let (_, catalog, first, second) = same_shard_targets(); - let first_entry = entry(&first, CatalogRole::Repository, 5); - let second_entry = entry(&second, CatalogRole::Repository, 6); - let barrier = Arc::new(tokio::sync::Barrier::new(2)); - let first_writer = { - let catalog = catalog.clone(); - let barrier = barrier.clone(); - let entry = first_entry.clone(); - tokio::spawn(async move { - barrier.wait().await; - catalog.provision(entry).await - }) - }; - let second_writer = { - let catalog = catalog.clone(); - let entry = second_entry.clone(); - tokio::spawn(async move { - barrier.wait().await; - catalog.provision(entry).await - }) - }; - first_writer.await.unwrap().unwrap(); - second_writer.await.unwrap().unwrap(); - - assert_eq!( - catalog - .lookup(first.cell_id()) - .await - .unwrap() - .unwrap() - .entry(), - &first_entry - ); - assert_eq!( - catalog - .lookup(second.cell_id()) - .await - .unwrap() - .unwrap() - .entry(), - &second_entry - ); -} - -#[tokio::test] -async fn catalog_rejects_conflicting_bootstrap_contract_for_one_cell() { - let (_, catalog, target) = fixture(); - catalog - .provision(entry(&target, CatalogRole::Repository, 7)) - .await - .unwrap(); - assert!(matches!( - catalog.provision(entry(&target, CatalogRole::Sql, 8)).await, - Err(crab_cell_runtime::Error::CatalogCollision) - )); -} - -#[tokio::test] -async fn catalog_page_digest_is_checked_before_entry_use() { - let (layout, catalog, target) = fixture(); - catalog - .provision(entry(&target, CatalogRole::Repository, 9)) - .await - .unwrap(); - let (head, _) = layout - .store() - .get_with_etag_bounded( - &layout.catalog_head_path(target.tenant().as_bytes(), target.cell_id().as_bytes()[0]), - 32 * 1024, - ) - .await - .unwrap(); - let head: serde_json::Value = serde_json::from_slice(&head).unwrap(); - let digest = decode_digest(head["pages"][0]["digest"].as_str().unwrap()); - layout - .store() - .put_overwrite( - &layout.catalog_object_path(&digest), - Bytes::from_static(b"{}"), - ) - .await - .unwrap(); - assert!(matches!( - catalog.lookup(target.cell_id()).await, - Err(crab_cell_runtime::Error::Catalog("page digest mismatch")) - )); -} - -#[tokio::test] -async fn catalog_lookup_reports_one_head_and_one_page_read() { - let tenant = TenantId::from_bytes([30; 16]); - let application = ApplicationId::from_bytes([31; 16]); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - *application.as_bytes(), - ); - let recorder = Arc::new(CatalogReadRecorder::default()); - let catalog = CellCatalog::with_telemetry( - layout.clone(), - tenant, - crab_cell_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(recorder.clone()), - ); - let target = CellTarget::new( - tenant, - application, - NamespaceId::from_bytes([32; 16]), - b"repository-metrics", - ) - .unwrap(); - let expected = entry(&target, CatalogRole::Repository, 12); - catalog.provision(expected.clone()).await.unwrap(); - // Provisioning reads catalog metadata, so the routing claim starts here. - recorder.reads.lock().unwrap().clear(); - assert_eq!( - catalog - .lookup(target.cell_id()) - .await - .unwrap() - .unwrap() - .entry(), - &expected - ); - assert_eq!( - recorder.reads.lock().unwrap().as_slice(), - [(CatalogReadKind::Head, true), (CatalogReadKind::Page, true)] - ); -} - -#[derive(Default)] -struct CatalogReadRecorder { - reads: std::sync::Mutex>, - control_reads: std::sync::atomic::AtomicUsize, -} - -impl crab_cell_runtime::fleet::telemetry::CellTelemetry for CatalogReadRecorder { - fn catalog_read(&self, kind: CatalogReadKind, _elapsed: std::time::Duration, succeeded: bool) { - self.reads.lock().unwrap().push((kind, succeeded)); - } - - fn control_read(&self, _elapsed: std::time::Duration, _succeeded: bool) { - self.control_reads - .fetch_add(1, std::sync::atomic::Ordering::AcqRel); - } -} - -#[tokio::test] -async fn due_scan_reads_one_control_record_per_cell() { - let (layout, tenant, _, entries) = provisioned_shard(40).await; - let shard = entries[0].cell().as_bytes()[0]; - let recorder = Arc::new(CatalogReadRecorder::default()); - let sink = - crab_cell_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(recorder.clone()); - let catalog = CellCatalog::with_telemetry(layout.clone(), tenant, sink.clone()); - let authority = CellAuthority::with_telemetry(layout.clone(), sink); - let mut scan = - crab_cell_runtime::fleet::scheduler::DueCellScan::new(&catalog, authority, shard) - .await - .unwrap(); - let mut due = 0; - while let Some(batch) = scan.next_batch_bounded(1, 8).await.unwrap() { - due += batch.len(); - } - assert_eq!(due, 0, "this shard arms no deadline"); - // The scan pays one control read for every Cell in the shard whether or - // not it is due, which is the discovery cost hint publication must remove. - assert_eq!( - recorder - .control_reads - .load(std::sync::atomic::Ordering::Acquire), - entries.len() - ); - // Pinning the head and reading its pages is the catalog half of that pass. - assert_eq!(recorder.reads.lock().unwrap().len(), 2); -} - -#[tokio::test] -async fn catalog_lookup_reads_only_the_page_that_can_hold_the_entry() { - let (layout, tenant, catalog, entries) = provisioned_shard(257).await; - let shard = entries[0].cell().as_bytes()[0]; - let head = head_json(&layout, tenant, shard).await; - let first_page = decode_digest(head["pages"][0]["digest"].as_str().unwrap()); - let last = entries.last().unwrap(); - assert!( - entries.first().unwrap().cell().as_bytes() < last.cell().as_bytes(), - "the shard must hold two pages with a distinct tail entry" - ); - // The earlier page cannot hold the tail entry, so a lookup that reads it - // would fail on the missing object. One page read is the whole contract. - layout - .store() - .delete(&layout.catalog_object_path(&first_page)) - .await - .unwrap(); - assert_eq!( - catalog.lookup(last.cell()).await.unwrap().unwrap().entry(), - last - ); - // The page that does hold the entry stays digest-verified. - let tail_page = decode_digest(head["pages"][1]["digest"].as_str().unwrap()); - layout - .store() - .delete(&layout.catalog_object_path(&tail_page)) - .await - .unwrap(); - assert!(catalog.lookup(last.cell()).await.is_err()); -} - -#[tokio::test] -async fn catalog_provision_replaces_only_the_affected_page() { - let (layout, tenant, _, entries) = provisioned_shard(257).await; - let shard = entries[0].cell().as_bytes()[0]; - let before = head_json(&layout, tenant, shard).await; - let second_first = before["pages"][1]["first"].as_str().unwrap(); - let second_digest = before["pages"][1]["digest"].clone(); - let application = ApplicationId::from_bytes([21; 16]); - let namespace = NamespaceId::from_bytes([22; 16]); - let target = (0_u32..) - .map(|partition| { - CellTarget::new(tenant, application, namespace, &partition.to_be_bytes()).unwrap() - }) - .find(|target| { - target.cell_id().as_bytes()[0] == shard - && hex(target.cell_id().as_bytes()).as_str() < second_first - && entries.iter().all(|entry| entry.cell() != target.cell_id()) - }) - .unwrap(); - let recorder = Arc::new(CatalogReadRecorder::default()); - let catalog = CellCatalog::with_telemetry( - layout.clone(), - tenant, - crab_cell_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(recorder.clone()), - ); - let expected = entry(&target, CatalogRole::Repository, 9); - catalog.provision(expected.clone()).await.unwrap(); - assert_eq!( - recorder.reads.lock().unwrap().as_slice(), - [(CatalogReadKind::Head, true), (CatalogReadKind::Page, true)] - ); - let after = head_json(&layout, tenant, shard).await; - assert_eq!(after["pages"][1]["digest"], second_digest); - assert_eq!( - catalog - .lookup(target.cell_id()) - .await - .unwrap() - .unwrap() - .entry(), - &expected - ); - let mut scan = catalog.scan_shard(shard).await.unwrap(); - let mut cells = Vec::new(); - while let Some(page) = scan.next_page().await.unwrap() { - cells.extend(page.entries().iter().map(|proof| proof.entry().cell())); - } - assert_eq!(cells.len(), 258); - assert!( - cells - .windows(2) - .all(|pair| pair[0].as_bytes() < pair[1].as_bytes()) - ); -} - -#[tokio::test] -async fn catalog_lookup_rejects_a_head_whose_locator_disagrees_with_its_page() { - let (layout, tenant, catalog, entries) = provisioned_shard(257).await; - let shard = entries[0].cell().as_bytes()[0]; - // The head body is canonical JSON, so tamper inside the encoded document: - // the first page now claims to open at the second entry, which the page - // does not have. A lookup that lands there must fail closed. - let (body, _) = layout - .store() - .get_with_etag_bounded( - &layout.catalog_head_path(tenant.as_bytes(), shard), - 64 * 1024, - ) - .await - .unwrap(); - let body = String::from_utf8(body.to_vec()).unwrap(); - let located = format!("\"first\":\"{}\"", hex(entries[0].cell().as_bytes())); - let moved = format!("\"first\":\"{}\"", hex(entries[1].cell().as_bytes())); - assert!(body.contains(&located), "head must locate its first page"); - let body = body.replacen(&located, &moved, 1); - layout - .store() - .put_overwrite( - &layout.catalog_head_path(tenant.as_bytes(), shard), - Bytes::from(body.into_bytes()), - ) - .await - .unwrap(); - match catalog.lookup(entries[2].cell()).await { - Err(crab_cell_runtime::Error::Catalog("catalog page locator disagrees with its page")) => {} - Ok(Some(proof)) => panic!( - "locator disagreement reported the entry {}", - hex(proof.entry().cell().as_bytes()) - ), - Ok(None) => panic!("locator disagreement reported absence"), - Err(error) => panic!("unexpected lookup error: {error:?}"), - } -} - -#[tokio::test] -async fn catalog_rejects_an_unordered_page_locator() { - let (layout, tenant, catalog, entries) = provisioned_shard(257).await; - let shard = entries[0].cell().as_bytes()[0]; - let mut head = head_json(&layout, tenant, shard).await; - head["pages"][1]["first"] = head["pages"][0]["first"].clone(); - layout - .store() - .put_overwrite( - &layout.catalog_head_path(tenant.as_bytes(), shard), - Bytes::from(serde_json::to_vec(&head).unwrap()), - ) - .await - .unwrap(); - assert!(matches!( - catalog.lookup(entries[0].cell()).await, - Err(crab_cell_runtime::Error::Catalog( - "catalog page locator is not ordered" - )) - )); -} - -async fn head_json(layout: &CellStorageLayout, tenant: TenantId, shard: u8) -> serde_json::Value { - let (head, _) = layout - .store() - .get_with_etag_bounded( - &layout.catalog_head_path(tenant.as_bytes(), shard), - 64 * 1024, - ) - .await - .unwrap(); - serde_json::from_slice(&head).unwrap() -} - -/// Provisions one shard past a page boundary and returns its sorted entries. -async fn provisioned_shard( - count: usize, -) -> (CellStorageLayout, TenantId, CellCatalog, Vec) { - let tenant = TenantId::from_bytes([20; 16]); - let application = ApplicationId::from_bytes([21; 16]); - let namespace = NamespaceId::from_bytes([22; 16]); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - *application.as_bytes(), - ); - let catalog = CellCatalog::new(layout.clone(), tenant); - let mut shards: HashMap> = HashMap::new(); - let mut targets = Vec::new(); - for partition in 0_u32.. { - let target = - CellTarget::new(tenant, application, namespace, &partition.to_be_bytes()).unwrap(); - let shard = target.cell_id().as_bytes()[0]; - let bucket = shards.entry(shard).or_default(); - bucket.push(target); - if bucket.len() == count { - targets = std::mem::take(bucket); - break; - } - } - let mut entries = targets - .iter() - .map(|target| entry(target, CatalogRole::Repository, 9)) - .collect::>(); - entries.sort_unstable_by_key(|entry| *entry.cell().as_bytes()); - for entry in &entries { - catalog.provision(entry.clone()).await.unwrap(); - } - (layout, tenant, catalog, entries) -} - -fn same_shard_targets() -> (CellStorageLayout, CellCatalog, CellTarget, CellTarget) { - let tenant = TenantId::from_bytes([10; 16]); - let application = ApplicationId::from_bytes([11; 16]); - let namespace = NamespaceId::from_bytes([12; 16]); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - *application.as_bytes(), - ); - let catalog = CellCatalog::new(layout.clone(), tenant); - let mut targets = HashMap::new(); - for partition in 0_u32..=256 { - let target = - CellTarget::new(tenant, application, namespace, &partition.to_be_bytes()).unwrap(); - let shard = target.cell_id().as_bytes()[0]; - if let Some(first) = targets.insert(shard, target.clone()) { - return (layout, catalog, first, target); - } - } - panic!("257 targets must contain a first-byte collision"); -} - -fn decode_digest(value: &str) -> [u8; 32] { - let mut digest = [0; 32]; - for (index, pair) in value.as_bytes().as_chunks::<2>().0.iter().enumerate() { - digest[index] = (nibble(pair[0]) << 4) | nibble(pair[1]); - } - digest -} - -fn hex(bytes: &[u8]) -> String { - bytes.iter().map(|byte| format!("{byte:02x}")).collect() -} - -fn nibble(byte: u8) -> u8 { - match byte { - b'0'..=b'9' => byte - b'0', - b'a'..=b'f' => byte - b'a' + 10, - _ => panic!("catalog digest must be lowercase hex"), - } -} diff --git a/crates/crab-cell-runtime/tests/runtime/fault_fs.rs b/crates/crab-cell-runtime/tests/runtime/fault_fs.rs deleted file mode 100644 index a1d252ec0..000000000 --- a/crates/crab-cell-runtime/tests/runtime/fault_fs.rs +++ /dev/null @@ -1,126 +0,0 @@ -//! Fault injection at local LTX capture and cleanup boundaries. - -use std::{ - io, - path::{Path, PathBuf}, - sync::{ - Mutex, - atomic::{AtomicBool, Ordering}, - }, -}; - -use crab_ltx::environment::{DirectFileSystem, FileIo, FileSystem}; - -pub struct FaultFileSystem { - fail_next_capture: AtomicBool, - fail_next_remove: AtomicBool, - cache_opens: Mutex>, -} - -impl FaultFileSystem { - pub fn new() -> Self { - Self { - fail_next_capture: AtomicBool::new(false), - fail_next_remove: AtomicBool::new(false), - cache_opens: Mutex::new(Vec::new()), - } - } - - pub fn fail_next_capture(&self) { - self.fail_next_capture.store(true, Ordering::SeqCst); - } - - pub fn capture_failure_consumed(&self) -> bool { - !self.fail_next_capture.load(Ordering::SeqCst) - } - - pub fn fail_next_prune(&self) { - self.fail_next_remove.store(true, Ordering::SeqCst); - } - - pub fn prune_failure_consumed(&self) -> bool { - !self.fail_next_remove.load(Ordering::SeqCst) - } - - pub fn cache_opens(&self) -> Vec<(PathBuf, std::thread::ThreadId)> { - self.cache_opens.lock().unwrap().clone() - } -} - -impl FileSystem for FaultFileSystem { - fn cleanup_private_temporaries(&self, root: &Path) -> io::Result<()> { - self.cache_opens - .lock() - .unwrap() - .push((root.to_owned(), std::thread::current().id())); - DirectFileSystem.cleanup_private_temporaries(root) - } - - fn open(&self, path: &Path) -> io::Result> { - DirectFileSystem.open(path) - } - - fn open_rw(&self, path: &Path) -> io::Result> { - DirectFileSystem.open_rw(path) - } - - fn create(&self, path: &Path) -> io::Result> { - if path.to_string_lossy().ends_with(".ltx.tmp") - && self.fail_next_capture.swap(false, Ordering::SeqCst) - { - return Err(io::Error::new( - io::ErrorKind::PermissionDenied, - "injected capture failure after SQLite commit", - )); - } - DirectFileSystem.create(path) - } - - fn file_len(&self, path: &Path) -> io::Result { - DirectFileSystem.file_len(path) - } - - fn create_dir_all(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.create_dir_all(path) - } - - fn rename(&self, from: &Path, to: &Path) -> io::Result<()> { - DirectFileSystem.rename(from, to) - } - - fn remove_file(&self, path: &Path) -> io::Result<()> { - if path.extension().is_some_and(|extension| extension == "ltx") - && self.fail_next_remove.swap(false, Ordering::SeqCst) - { - return Err(io::Error::new( - io::ErrorKind::PermissionDenied, - "injected prune failure after root publication", - )); - } - DirectFileSystem.remove_file(path) - } - - fn canonicalize(&self, path: &Path) -> io::Result { - DirectFileSystem.canonicalize(path) - } - - fn exists(&self, path: &Path) -> io::Result { - DirectFileSystem.exists(path) - } - - fn create_dir(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.create_dir(path) - } - - fn sync_parent(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.sync_parent(path) - } - - fn persist_new(&self, path: &Path, bytes: &[u8]) -> io::Result<()> { - DirectFileSystem.persist_new(path, bytes) - } - - fn persist_file_new(&self, source: &Path, destination: &Path) -> io::Result<()> { - DirectFileSystem.persist_file_new(source, destination) - } -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle.rs deleted file mode 100644 index e1f46fa6e..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle.rs +++ /dev/null @@ -1,752 +0,0 @@ -use std::{ - fmt, - sync::{ - Arc, Mutex, - atomic::{AtomicBool, AtomicUsize, Ordering}, - mpsc, - }, -}; - -use bytes::Bytes; -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogEntry; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::executor::Resolution; -use crab_cell_runtime::cell::executor::{HandlerOutcome, StoredOutcome}; -use crab_cell_runtime::cell::worker::ACTIVE_CELL_FILE_DESCRIPTORS; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::control::{ControlState, Owner, Transition}; -use crab_cell_runtime::fleet::pressure::{PressureSample, PressureState}; -use crab_cell_runtime::follower::FollowerReceipt; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::identity::{IncarnationId, NodeId}; -use crab_cell_runtime::ltx::{DiskBudget, Host as ReplicaHost}; -use crab_cell_runtime::node::durability::{NodeDurability, NodeLogAuthority}; -use crab_cell_runtime::node::lease::NodeLeaseGuard; -use crab_cell_runtime::node::log::{DurabilityGate, NodeLogRotationBarrier}; -use crab_cell_runtime::node::log_shipper::NodeLogShipper; -use crab_cell_runtime::node::log_transport::{ - AppendRequest, NodeLogTransport, RetireRequest, SealRequest, TailRequest, -}; -use crab_cell_runtime::primitives::effects::InboxDelivery; -use crab_cell_runtime::primitives::queue::install_queue_schema; -use crab_cell_runtime::primitives::workflow::install_workflow_schema; -use crab_ltx::{CellObjectKind, CellStorageLayout}; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::{ObjectStoreCredentials, RetryPolicy, Store, build_explicit_store}; -use futures_util::stream::BoxStream; -use object_store::{ - CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, - PutMode, PutMultipartOptions, PutOptions, PutPayload, PutResult, memory::InMemory, path::Path, -}; -use tokio::sync::Notify; - -use crate::support::fencing::fence_session; -use crate::support::fixtures::{mutation_identity_window, now_ms}; - -// Capability modules keep the suite navigable; shared fixtures and -// helpers used by more than one capability stay here. -pub mod durability; -pub mod execution; -pub mod idle; -pub mod ownership; -pub mod read_replica; -pub mod residency; - -#[derive(Debug)] -struct PausingStore { - inner: Arc, - armed: AtomicBool, - update_armed: AtomicBool, - failing: AtomicBool, - transient_put_failures: AtomicUsize, - lost_update_response: AtomicBool, - failed: AtomicBool, - blocked: AtomicBool, - released: AtomicBool, - get_armed: AtomicBool, - fail_next_get: AtomicBool, - transient_get_failures: AtomicUsize, - get_blocked: AtomicBool, - get_released: AtomicBool, - entered: Notify, - release: Notify, - get_entered: Notify, - get_release: Notify, - parallel_catalog_heads: AtomicBool, - catalog_head_barrier: tokio::sync::Barrier, -} - -impl PausingStore { - fn new(inner: Arc) -> Self { - Self { - inner, - armed: AtomicBool::new(false), - update_armed: AtomicBool::new(false), - failing: AtomicBool::new(false), - transient_put_failures: AtomicUsize::new(0), - lost_update_response: AtomicBool::new(false), - failed: AtomicBool::new(false), - blocked: AtomicBool::new(false), - released: AtomicBool::new(false), - get_armed: AtomicBool::new(false), - fail_next_get: AtomicBool::new(false), - transient_get_failures: AtomicUsize::new(0), - get_blocked: AtomicBool::new(false), - get_released: AtomicBool::new(false), - entered: Notify::new(), - release: Notify::new(), - get_entered: Notify::new(), - get_release: Notify::new(), - parallel_catalog_heads: AtomicBool::new(false), - catalog_head_barrier: tokio::sync::Barrier::new(2), - } - } - - fn require_parallel_catalog_heads(&self) { - self.parallel_catalog_heads.store(true, Ordering::Release); - } - - fn arm(&self) { - self.armed.store(true, Ordering::Release); - } - - fn arm_next_update(&self) { - self.update_armed.store(true, Ordering::Release); - } - - fn fail_puts(&self) { - self.failing.store(true, Ordering::Release); - } - - fn fail_next_put_transiently(&self) { - self.transient_put_failures.store(1, Ordering::Release); - } - - fn lose_next_update_response(&self) { - self.lost_update_response.store(true, Ordering::Release); - } - - fn lost_update_response_consumed(&self) -> bool { - !self.lost_update_response.load(Ordering::Acquire) - } - - fn allow_puts(&self) { - self.failing.store(false, Ordering::Release); - } - - async fn wait_until_failed(&self) { - while !self.failed.load(Ordering::Acquire) { - self.entered.notified().await; - } - } - - async fn wait_until_blocked(&self) { - while !self.blocked.load(Ordering::Acquire) { - self.entered.notified().await; - } - } - - fn release(&self) { - self.released.store(true, Ordering::Release); - self.release.notify_waiters(); - } - - fn arm_gets(&self) { - self.get_armed.store(true, Ordering::Release); - } - - fn fail_next_get(&self) { - self.fail_next_get.store(true, Ordering::Release); - } - - async fn wait_until_get_blocked(&self) { - while !self.get_blocked.load(Ordering::Acquire) { - self.get_entered.notified().await; - } - } - - fn release_gets(&self) { - self.get_released.store(true, Ordering::Release); - self.get_release.notify_waiters(); - } -} - -impl fmt::Display for PausingStore { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter.write_str("pausing-store") - } -} - -#[async_trait::async_trait] -impl ObjectStore for PausingStore { - async fn put_opts( - &self, - location: &Path, - payload: PutPayload, - options: PutOptions, - ) -> object_store::Result { - if self - .transient_put_failures - .fetch_update(Ordering::AcqRel, Ordering::Acquire, |remaining| { - remaining.checked_sub(1) - }) - .is_ok() - { - self.failed.store(true, Ordering::Release); - self.entered.notify_waiters(); - return Err(object_store::Error::Generic { - store: "pausing-store", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::ConnectionReset, - "injected transient put failure", - )), - }); - } - if self.failing.load(Ordering::Acquire) { - self.failed.store(true, Ordering::Release); - self.entered.notify_waiters(); - return Err(object_store::Error::PermissionDenied { - path: location.to_string(), - source: Box::new(std::io::Error::new( - std::io::ErrorKind::PermissionDenied, - "injected put failure", - )), - }); - } - let update = matches!(&options.mode, PutMode::Update(_)); - if (self.armed.load(Ordering::Acquire) - || (update && self.update_armed.load(Ordering::Acquire))) - && !self.blocked.swap(true, Ordering::AcqRel) - { - self.entered.notify_waiters(); - while !self.released.load(Ordering::Acquire) { - self.release.notified().await; - } - } - let result = self.inner.put_opts(location, payload, options).await?; - if update && self.lost_update_response.swap(false, Ordering::AcqRel) { - return Err(object_store::Error::Generic { - store: "pausing-store", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::ConnectionReset, - "control CAS response lost after commit", - )), - }); - } - Ok(result) - } - - async fn put_multipart_opts( - &self, - location: &Path, - options: PutMultipartOptions, - ) -> object_store::Result> { - self.inner.put_multipart_opts(location, options).await - } - - async fn get_opts( - &self, - location: &Path, - options: GetOptions, - ) -> object_store::Result { - if self.get_armed.load(Ordering::Acquire) && !self.get_blocked.swap(true, Ordering::AcqRel) - { - self.get_entered.notify_waiters(); - while !self.get_released.load(Ordering::Acquire) { - self.get_release.notified().await; - } - } - if self.parallel_catalog_heads.load(Ordering::Acquire) - && location.as_ref().ends_with("/head.json") - { - self.catalog_head_barrier.wait().await; - } - if options.range.is_some() - && self - .transient_get_failures - .fetch_update(Ordering::AcqRel, Ordering::Acquire, |remaining| { - remaining.checked_sub(1) - }) - .is_ok() - { - return Err(object_store::Error::Generic { - store: "pausing-store", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::ConnectionReset, - "injected transient get failure", - )), - }); - } - if self.fail_next_get.swap(false, Ordering::AcqRel) { - return Err(object_store::Error::PermissionDenied { - path: location.to_string(), - source: Box::new(std::io::Error::new( - std::io::ErrorKind::PermissionDenied, - "injected origin read failure during hydration", - )), - }); - } - self.inner.get_opts(location, options).await - } - - fn delete_stream( - &self, - locations: BoxStream<'static, object_store::Result>, - ) -> BoxStream<'static, object_store::Result> { - self.inner.delete_stream(locations) - } - - fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { - self.inner.list(prefix) - } - - async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { - self.inner.list_with_delimiter(prefix).await - } - - async fn copy_opts( - &self, - from: &Path, - to: &Path, - options: CopyOptions, - ) -> object_store::Result<()> { - self.inner.copy_opts(from, to, options).await - } -} - -#[derive(Default)] -struct TestNodeAuthority { - activations: Mutex>, - coverage: Mutex>, - closes: Mutex>, -} - -impl NodeLogAuthority for TestNodeAuthority { - fn activate<'a>( - &'a self, - log_epoch: u64, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result<()>> { - Box::pin(async move { - self.activations.lock().unwrap().push(log_epoch); - Ok(()) - }) - } - - fn advance_coverage<'a>( - &'a self, - log_epoch: u64, - tiered_through: u64, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result<()>> { - Box::pin(async move { - self.coverage - .lock() - .unwrap() - .push((log_epoch, tiered_through)); - Ok(()) - }) - } - - fn close<'a>( - &'a self, - barrier: &'a NodeLogRotationBarrier, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result<()>> { - Box::pin(async move { - self.closes.lock().unwrap().push(barrier.log_epoch()); - Ok(()) - }) - } -} - -#[derive(Default)] -struct TestNodeTransport(Option); - -impl NodeLogTransport for TestNodeTransport { - fn append<'a>( - &'a self, - _member: NodeId, - request: AppendRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result> { - let delay = self.0; - Box::pin(async move { - if let Some(delay) = delay { - tokio::time::sleep(delay).await; - } - let last = request.frames.last().ok_or(crab_cell_runtime::Error::Node( - "test node-log append is empty", - ))?; - let frame = crab_ltx::inspect_node_frame(last.clone(), Limits::default())?; - Ok(FollowerReceipt { - base_sequence: 1, - durable_through: frame.scope().node_sequence, - }) - }) - } - - fn seal<'a>( - &'a self, - _member: NodeId, - _request: SealRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result> { - Box::pin(async { Err(crab_cell_runtime::Error::Node("unused test seal")) }) - } - - fn tail<'a>( - &'a self, - _member: NodeId, - _request: TailRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result>> { - Box::pin(async { Err(crab_cell_runtime::Error::Node("unused test tail")) }) - } - - fn retire<'a>( - &'a self, - _member: NodeId, - request: RetireRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result> { - Box::pin(async move { - Ok(FollowerReceipt { - base_sequence: request.covered_through.saturating_add(1), - durable_through: request.covered_through, - }) - }) - } -} - -struct LostAckFollowerTransport { - inner: crab_cell_runtime::node::log_transport::LocalFollowerTransport, - acknowledged_once: AtomicBool, - lost_ticket: Mutex< - Option<( - NodeId, - crab_ltx::NodeFrameScope, - crab_ltx::NodeFrameScope, - FollowerReceipt, - )>, - >, -} - -impl NodeLogTransport for LostAckFollowerTransport { - fn append<'a>( - &'a self, - member: NodeId, - request: AppendRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result> { - Box::pin(async move { - let first = crab_ltx::inspect_node_frame( - request.frames.first().unwrap().clone(), - Limits::default(), - )?; - let last = crab_ltx::inspect_node_frame( - request.frames.last().unwrap().clone(), - Limits::default(), - )?; - let receipt = self.inner.append(member, request).await?; - if !self.acknowledged_once.swap(true, Ordering::AcqRel) { - return Ok(receipt); - } - *self.lost_ticket.lock().unwrap() = - Some((member, first.scope(), last.scope(), receipt)); - Err(crab_cell_runtime::Error::Node( - "injected follower acknowledgement loss", - )) - }) - } - - fn seal<'a>( - &'a self, - member: NodeId, - request: SealRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result> { - self.inner.seal(member, request) - } - - fn tail<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result>> { - self.inner.tail(member, request) - } - - fn retire<'a>( - &'a self, - member: NodeId, - request: RetireRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result> { - self.inner.retire(member, request) - } -} - -async fn fence_log_session( - layout: &CellStorageLayout, - session: SessionId, - claimant: SessionId, - member: SessionId, - tiered_through: u64, -) -> crab_cell_runtime::node::FencedNodeSession { - let fleet = Digest::from_bytes([90; 32]); - let image = Digest::from_bytes([91; 32]); - let release = Digest::from_bytes([92; 32]); - let directory = - crab_cell_runtime::node::NodeDirectory::new(layout.clone(), fleet, image, release); - let key = ed25519_dalek::SigningKey::from_bytes(&[93; 32]); - let signed = - |session: crab_cell_runtime::SessionId, endpoint: &str, issued_at_ms, expires_at_ms| { - crab_cell_runtime::node::NodeAdvertisement::sign( - crab_cell_runtime::identity::NodeId::from_bytes(*session.as_bytes()), - session, - endpoint.into(), - fleet, - Digest::from_bytes([94; 32]), - image, - release, - &key, - 1, - issued_at_ms, - expires_at_ms, - vec![Digest::from_bytes([95; 32])], - vec![1], - crab_cell_runtime::node::NodeFailureDomain::default(), - crab_cell_runtime::node::NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 1, - follower_free_bytes: 1, - follower_retained_bytes: 0, - job_credits: 1, - log_protocol: crab_cell_runtime::node::NODE_LOG_PROTOCOL_VERSION, - }, - ) - .unwrap() - }; - let leader = directory - .create( - signed(session, "https://expired.internal:8081", 1, 10_001), - 1, - ) - .await - .unwrap(); - directory - .create( - signed(member, "https://follower.internal:8081", 10_000, 20_000), - 2, - ) - .await - .unwrap(); - let enrolled = directory.recruit_log(&leader, 1, 1, 2, 2).await.unwrap(); - let enrolled = directory.activate_log(&enrolled, 3).await.unwrap(); - if tiered_through != 0 { - directory - .advance_log_coverage(&enrolled, tiered_through, 4) - .await - .unwrap(); - } - if claimant != member { - directory - .create( - signed(claimant, "https://claimant.internal:8081", 10_000, 20_000), - 3, - ) - .await - .unwrap(); - } - directory - .claim_expired(session, claimant, 10_001) - .await - .unwrap() -} - -struct Fixture { - _directory: tempfile::TempDir, - database: std::path::PathBuf, - target: CellTarget, - layout: CellStorageLayout, - replica: CellReplica, -} - -fn fixture() -> Fixture { - fixture_for(b"repository-1") -} - -fn fixture_for(partition: &[u8]) -> Fixture { - fixture_with_limits(partition, Limits::default()) -} - -fn fixture_with_limits(partition: &[u8], limits: Limits) -> Fixture { - fixture_with_limits_and_store(partition, limits, Store::new(Arc::new(InMemory::new()))) -} - -fn fixture_with_limits_and_store(partition: &[u8], limits: Limits, store: Store) -> Fixture { - fixture_with_limits_and_store_at_prefix(partition, limits, store, Path::from("runtime")) -} - -fn fixture_with_limits_and_store_at_prefix( - partition: &[u8], - limits: Limits, - store: Store, - prefix: Path, -) -> Fixture { - let target = CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([3; 16]), - NamespaceId::from_bytes([6; 16]), - partition, - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([2; 16]); - let layout = CellStorageLayout::new(store, prefix, [3; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - limits, - ) - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let database = directory.path().join("cell.sqlite"); - Fixture { - _directory: directory, - database, - target, - layout, - replica, - } -} - -#[cfg(feature = "test-support")] -fn filesystem_fixture(partition: &[u8], root: &std::path::Path) -> Fixture { - let store = crab_cell_runtime::test_support::FilesystemCasStore::new(root).unwrap(); - fixture_with_limits_and_store(partition, Limits::default(), Store::new(Arc::new(store))) -} - -async fn activate( - fixture: &Fixture, - node_bytes: usize, -) -> crab_cell_runtime::cell::actor::CellHandle { - activate_runtime(fixture, node_bytes).await.1 -} - -async fn activate_runtime( - fixture: &Fixture, - node_bytes: usize, -) -> ( - CellRuntime, - crab_cell_runtime::cell::actor::CellHandle, - SqlWorkerPool, -) { - let session = SessionId::from_bytes([4; 16]); - let pool = SqlWorkerPool::new(2, 10).unwrap(); - let runtime = CellRuntime::new(pool.clone(), node_bytes, session).unwrap(); - let handle = bootstrap_on(&runtime, fixture, session).await; - (runtime, handle, pool) -} - -async fn wait_for_persisted_work( - handle: &crab_cell_runtime::cell::actor::CellHandle, - role: CatalogRole, - blocker: &'static str, -) { - tokio::time::timeout(std::time::Duration::from_secs(3), async { - loop { - if handle - .persisted_work_inventory(role) - .await - .unwrap() - .first_blocker() - == Some(blocker) - { - return; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); -} - -async fn bootstrap_on( - runtime: &CellRuntime, - fixture: &Fixture, - session: SessionId, -) -> crab_cell_runtime::cell::actor::CellHandle { - bootstrap_role_on( - runtime, - fixture, - session, - CatalogRole::Repository, - |transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (0)", - )?; - Ok(()) - }, - ) - .await -} - -async fn bootstrap_role_on( - runtime: &CellRuntime, - fixture: &Fixture, - session: SessionId, - role: CatalogRole, - initialize: F, -) -> crab_cell_runtime::cell::actor::CellHandle -where - F: for<'connection> FnOnce( - &crab_ltx::rusqlite::Transaction<'connection>, - ) -> crab_cell_runtime::Result<()> - + Send - + 'static, -{ - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .provision( - CatalogEntry::new(&fixture.target, role, Digest::from_bytes([5; 32]), 1).unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .create_initial( - &proof, - IncarnationId::from_bytes([2; 16]), - Owner { - session, - endpoint: "https://node.internal:8081".into(), - }, - ) - .await - .unwrap(); - runtime - .bootstrap( - proof, - fixture.replica.clone(), - authority, - observed, - fixture.database.clone(), - initialize, - ) - .await - .unwrap() -} - -async fn delete_control_root(fixture: &Fixture) { - let cell = fixture.target.cell_id(); - let authority = CellAuthority::new(fixture.layout.clone()); - let control = authority.load(cell).await.unwrap().unwrap(); - let root = control.value().root.as_ref().unwrap(); - let root_path = fixture.layout.incarnation_object_path( - cell.as_bytes(), - control.value().incarnation.as_bytes(), - root.digest.as_bytes(), - CellObjectKind::Root, - ); - fixture.layout.store().delete(&root_path).await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/durability.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/durability.rs deleted file mode 100644 index 2632d09f4..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/durability.rs +++ /dev/null @@ -1,26 +0,0 @@ -//! Node-log durability, fleet proofs, and byte admission. - -use super::*; -use crab_cell_runtime::fleet::telemetry::{CellTelemetry, CommandResponseSource}; - -mod admission; -mod proofs; -mod recovery; - -#[derive(Default)] -pub(super) struct RecordingResponses(pub(super) Mutex>); - -impl CellTelemetry for RecordingResponses { - fn command_response( - &self, - source: CommandResponseSource, - elapsed: std::time::Duration, - confirmation: std::time::Duration, - ) { - assert!(confirmation <= elapsed); - if source == CommandResponseSource::Recorded { - assert!(confirmation.is_zero()); - } - self.0.lock().unwrap().push(source); - } -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/durability/admission.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/durability/admission.rs deleted file mode 100644 index 2c843bb20..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/durability/admission.rs +++ /dev/null @@ -1,69 +0,0 @@ -//! Node durability byte admission and reservation bounds. - -use super::*; - -#[tokio::test] -async fn node_byte_reservation_rejects_overcommit_and_releases_capacity() { - let session = SessionId::from_bytes([40; 16]); - let local_disk = DiskBudget::new(4_096); - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 1).unwrap(), - 1_024, - session, - ReplicaHost::default().with_local_disk_budget(local_disk.clone()), - ) - .unwrap(); - let disk = local_disk.try_reserve(512).unwrap(); - let held = runtime.try_reserve_node_bytes(1_024).unwrap(); - - let full = runtime.stats(); - assert_eq!(full.active_cells(), 0); - assert_eq!(full.active_cell_capacity(), 1); - assert_eq!(full.file_descriptors(), 0); - assert_eq!( - full.file_descriptor_capacity(), - ACTIVE_CELL_FILE_DESCRIPTORS - ); - assert_eq!(full.retained_bytes(), 1_024); - assert_eq!(full.retained_capacity_bytes(), 1_024); - assert_eq!(full.local_disk_reserved_bytes(), 512); - assert_eq!(full.local_disk_capacity_bytes(), 4_096); - - assert!(matches!( - runtime.try_reserve_node_bytes(1), - Err(crab_cell_runtime::Error::Capacity("node retained bytes")) - )); - drop(held); - drop(disk); - let empty = runtime.stats(); - assert_eq!(empty.file_descriptors(), 0); - assert_eq!( - empty.file_descriptor_capacity(), - ACTIVE_CELL_FILE_DESCRIPTORS - ); - assert_eq!(empty.retained_bytes(), 0); - assert_eq!(empty.local_disk_reserved_bytes(), 0); - let released = runtime.try_reserve_node_bytes(1_024).unwrap(); - drop(released); - - runtime.shutdown().await.unwrap(); -} -#[tokio::test] -async fn node_byte_admission_rejects_before_sql_execution() { - let fixture = fixture(); - let handle = activate(&fixture, 1024 * 1024).await; - assert!(matches!( - handle - .execute( - mutation_identity_window(12, 10, 10_000), - Digest::from_bytes([13; 32]), - 20, - 1_025, - 1024 * 1024, - |_| Ok(HandlerOutcome::Success(Vec::new())), - ) - .await, - Err(crab_cell_runtime::Error::Capacity(_)) - )); - handle.drain().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/durability/proofs.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/durability/proofs.rs deleted file mode 100644 index eb5f0451f..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/durability/proofs.rs +++ /dev/null @@ -1,1067 +0,0 @@ -//! Follower and fleet proofs, object publication, and binding epochs. - -use super::*; -use crate::runtime::fault_fs::FaultFileSystem; - -#[tokio::test(flavor = "multi_thread")] -async fn accepted_control_cas_with_lost_response_releases_once() { - let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let object_store: Arc = store.clone(); - let fixture = fixture_with_limits_and_store( - b"lost-control-cas-response", - Limits::default(), - Store::new(object_store), - ); - let (runtime, handle, _) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - let responses = Arc::new(RecordingResponses::default()); - runtime.install_telemetry(responses.clone()).unwrap(); - let request = mutation_identity_window(136, 10, 10_000); - let digest = Digest::from_bytes([137; 32]); - let calls = Arc::new(AtomicUsize::new(0)); - let observed = calls.clone(); - store.lose_next_update_response(); - - let first = handle - .execute(request, digest, 20, 1_024, 1_024, move |transaction| { - observed.fetch_add(1, Ordering::SeqCst); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"published".to_vec())) - }) - .await - .unwrap(); - assert!(matches!( - first, - StoredOutcome::Success { ref result, commit_sequence: 1 } if result == b"published" - )); - assert!(store.lost_update_response_consumed()); - let authority = CellAuthority::new(fixture.layout.clone()); - let control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let root = control.value().ltx_root().unwrap(); - eprintln!( - "fault_seed=136 schedule=accepted_control_cas_response_lost request={request:?} operation={digest:?} committed_sequence={} selected_follower_tickets=[] observed_control={:?}", - first.commit_sequence(), - control.value(), - ); - assert_eq!(root.commit_sequence, 1); - assert_eq!(root.position.txid, 2); - - let duplicate_calls = calls.clone(); - let replay = handle - .execute(request, digest, 21, 1_024, 1_024, move |_| { - duplicate_calls.fetch_add(1, Ordering::SeqCst); - Ok(HandlerOutcome::Success(b"duplicate".to_vec())) - }) - .await - .unwrap(); - assert_eq!(replay, first); - assert_eq!(calls.load(Ordering::SeqCst), 1); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - assert_eq!( - responses.0.lock().unwrap().as_slice(), - &[ - CommandResponseSource::Object, - CommandResponseSource::Recorded - ] - ); - - let restored = fixture._directory.path().join("lost-cas-restored.sqlite"); - let verified = fixture.replica.open_root(&root).await.unwrap(); - assert_eq!(verified.restore(&restored).await.unwrap(), root.position); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) - .unwrap(), - 1 - ); - assert_eq!( - connection - .query_row( - "SELECT result FROM sys_requests WHERE request_id = ?1", - [request.request_id.as_bytes().as_slice()], - |row| row.get::<_, Vec>(0), - ) - .unwrap(), - b"published" - ); -} - -#[tokio::test(flavor = "multi_thread")] -async fn capture_failure_after_sql_commit_fences_until_authoritative_recovery() { - let fixture = fixture_for(b"capture-after-sql-commit"); - let filesystem = Arc::new(FaultFileSystem::new()); - let session = SessionId::from_bytes([133; 16]); - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - session, - ReplicaHost::default().with_filesystem(filesystem.clone()), - ) - .unwrap(); - let responses = Arc::new(RecordingResponses::default()); - runtime.install_telemetry(responses.clone()).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - let authority = CellAuthority::new(fixture.layout.clone()); - let before = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let root = before.value().ltx_root().unwrap(); - let request = mutation_identity_window(134, 10, 10_000); - let digest = Digest::from_bytes([135; 32]); - let calls = Arc::new(AtomicUsize::new(0)); - let observed = calls.clone(); - filesystem.fail_next_capture(); - let first = handle - .execute(request, digest, 20, 1_024, 1_024, move |transaction| { - observed.fetch_add(1, Ordering::SeqCst); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"unpublished".to_vec())) - }) - .await; - let observed_control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(&fixture.database).unwrap(); - let committed_sequence: i64 = connection - .query_row( - "SELECT commit_sequence FROM sys_requests WHERE request_id = ?1", - [request.request_id.as_bytes().as_slice()], - |row| row.get(0), - ) - .unwrap(); - eprintln!( - "fault_seed=133 schedule=after_sql_commit_before_capture request={request:?} operation={digest:?} committed_sequence={committed_sequence} selected_follower_tickets=[] observed_control={:?} result={first:?}", - observed_control.value(), - ); - assert_eq!(committed_sequence, 1); - drop(connection); - assert!( - matches!( - first, - Err(crab_cell_runtime::Error::OutcomeUnknown { - request_id, - operation_digest, - .. - }) if request_id == request.request_id && operation_digest == digest - ), - "{first:?}" - ); - assert!(filesystem.capture_failure_consumed()); - assert_eq!(calls.load(Ordering::SeqCst), 1); - assert_eq!(observed_control.value().ltx_root(), Some(root)); - assert_eq!( - handle.resolve(request, digest, 21, 1_024).await.unwrap(), - Resolution::Unknown - ); - runtime.shutdown().await.unwrap(); - assert_eq!(responses.0.lock().unwrap().as_slice(), &[]); - - let restored = fixture._directory.path().join("capture-restored.sqlite"); - let verified = fixture.replica.open_root(&root).await.unwrap(); - assert_eq!(verified.restore(&restored).await.unwrap(), root.position); - let mut recovered = crab_cell_runtime::cell::executor::CellExecutor::new( - crab_ltx::Db::open(&restored, Limits::default()).unwrap(), - fixture.target.cell_id(), - IncarnationId::from_bytes([2; 16]), - 1, - ); - assert_eq!( - recovered.resolve(request, digest, 22, 1_024).unwrap(), - Resolution::Absent - ); - recovered.close().unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn object_root_publication_fences_actor_when_local_pruning_fails() { - let fixture = fixture_for(b"prune-after-root-cas"); - let filesystem = Arc::new(FaultFileSystem::new()); - let host = ReplicaHost::default().with_filesystem(filesystem.clone()); - let session = SessionId::from_bytes([130; 16]); - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - session, - host, - ) - .unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - let request = mutation_identity_window(131, 10, 10_000); - let digest = Digest::from_bytes([132; 32]); - let calls = Arc::new(AtomicUsize::new(0)); - let observed = calls.clone(); - filesystem.fail_next_prune(); - let first = handle - .execute(request, digest, 20, 1_024, 1_024, move |transaction| { - observed.fetch_add(1, Ordering::SeqCst); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"committed".to_vec())) - }) - .await; - let authority = CellAuthority::new(fixture.layout.clone()); - let control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - eprintln!( - "fault_seed=130 schedule=after_root_cas_before_prune request={request:?} operation={digest:?} committed_sequence=1 selected_follower_tickets=[] observed_control={:?} result={first:?}", - control.value(), - ); - assert!( - matches!( - first, - Err(crab_cell_runtime::Error::OutcomeUnknown { - request_id, - operation_digest, - .. - }) if request_id == request.request_id && operation_digest == digest - ), - "{first:?}" - ); - assert!(filesystem.prune_failure_consumed()); - assert_eq!(calls.load(Ordering::SeqCst), 1); - let root = control.value().ltx_root().unwrap(); - assert_eq!(root.commit_sequence, 1); - let duplicate_calls = calls.clone(); - assert!(matches!( - handle - .execute(request, digest, 21, 1_024, 1_024, move |_| { - duplicate_calls.fetch_add(1, Ordering::SeqCst); - Ok(HandlerOutcome::Success(b"duplicate".to_vec())) - }) - .await, - Err(crab_cell_runtime::Error::Fenced) - )); - runtime.shutdown().await.unwrap(); - - let restored = fixture._directory.path().join("prune-restored.sqlite"); - let verified = fixture.replica.open_root(&root).await.unwrap(); - assert_eq!(verified.restore(&restored).await.unwrap(), root.position); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) - .unwrap(), - 1 - ); - assert_eq!( - connection - .query_row( - "SELECT result FROM sys_requests WHERE request_id = ?1", - [request.request_id.as_bytes().as_slice()], - |row| row.get::<_, Vec>(0), - ) - .unwrap(), - b"committed" - ); - assert_eq!(calls.load(Ordering::SeqCst), 1); -} - -#[tokio::test(flavor = "multi_thread")] -async fn runtime_replaces_node_durability_binding_for_a_new_epoch() { - let fixture = fixture_for(b"node-durability-rotation"); - let session = SessionId::from_bytes([60; 16]); - let leader = NodeId::from_bytes([61; 16]); - let follower = NodeId::from_bytes([62; 16]); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - session, - ReplicaHost::default(), - ) - .unwrap(); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - runtime.install_node_lease(lease.clone()).unwrap(); - let transport: Arc = Arc::new(TestNodeTransport(None)); - let make_durability = |log_epoch| { - let gate = DurabilityGate::new(session, leader, log_epoch, [follower]).unwrap(); - let shipper = - NodeLogShipper::new(gate.clone(), Arc::clone(&transport), Limits::default()).unwrap(); - let authority: Arc = Arc::new(TestNodeAuthority::default()); - Arc::new(NodeDurability::new( - gate, - shipper, - authority, - Arc::clone(&transport), - lease.clone(), - )) - }; - let first = make_durability(1); - runtime - .install_node_durability(fixture.target.application(), Arc::clone(&first)) - .unwrap(); - let second = make_durability(2); - - let previous = runtime - .replace_node_durability(fixture.target.application(), Arc::clone(&second)) - .unwrap(); - - assert!(Arc::ptr_eq(&previous, &first)); - let (application, current) = runtime.node_durability().unwrap(); - assert_eq!(application, fixture.target.application()); - assert!(Arc::ptr_eq(¤t, &second)); - runtime.shutdown().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn command_flows_through_follower_proof_object_coverage_and_clean_close() { - let fixture = fixture_for(b"node-durability-command"); - let session = SessionId::from_bytes([46; 16]); - let leader = NodeId::from_bytes([47; 16]); - let follower = NodeId::from_bytes([48; 16]); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - session, - ReplicaHost::default(), - ) - .unwrap(); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - runtime.install_node_lease(lease.clone()).unwrap(); - let gate = DurabilityGate::new(session, leader, 1, [follower]).unwrap(); - let transport: Arc = Arc::new(TestNodeTransport::default()); - let shipper = - NodeLogShipper::new(gate.clone(), Arc::clone(&transport), Limits::default()).unwrap(); - let authority = Arc::new(TestNodeAuthority::default()); - let node_authority: Arc = authority.clone(); - let durability = Arc::new(NodeDurability::new( - gate, - shipper, - node_authority, - transport, - lease, - )); - runtime - .install_node_durability(fixture.target.application(), durability) - .unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - - let outcome = handle - .execute( - mutation_identity_window(49, 10, 10_000), - Digest::from_bytes([50; 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"durable".to_vec())) - }, - ) - .await - .unwrap(); - - assert!(matches!( - outcome, - StoredOutcome::Success { - ref result, - commit_sequence: 1 - } if result == b"durable" - )); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - assert_eq!(*authority.coverage.lock().unwrap(), vec![(1, 1)]); - assert_eq!(*authority.closes.lock().unwrap(), vec![1]); -} -#[tokio::test(flavor = "multi_thread")] -async fn follower_proofs_advance_logical_head_and_bound_the_object_backlog() { - let pausing = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let object_store: Arc = pausing.clone(); - let fixture = fixture_with_limits_and_store( - b"dual-head-command", - Limits::default(), - Store::new(object_store), - ); - let session = SessionId::from_bytes([51; 16]); - let leader = NodeId::from_bytes([52; 16]); - let follower = NodeId::from_bytes([53; 16]); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - session, - ReplicaHost::default(), - ) - .unwrap(); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - runtime.install_node_lease(lease.clone()).unwrap(); - let gate = DurabilityGate::new(session, leader, 1, [follower]).unwrap(); - let transport: Arc = Arc::new(TestNodeTransport::default()); - let shipper = - NodeLogShipper::new(gate.clone(), Arc::clone(&transport), Limits::default()).unwrap(); - let authority = Arc::new(TestNodeAuthority::default()); - let node_authority: Arc = authority.clone(); - runtime - .install_node_durability( - fixture.target.application(), - Arc::new(NodeDurability::new( - gate, - shipper, - node_authority, - transport, - lease, - )), - ) - .unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - pausing.arm(); - - let first = tokio::time::timeout( - std::time::Duration::from_secs(2), - handle.execute( - mutation_identity_window(54, 10, 10_000), - Digest::from_bytes([55; 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"one".to_vec())) - }, - ), - ) - .await - .unwrap() - .unwrap(); - assert_eq!(first.commit_sequence(), 1); - tokio::time::timeout( - std::time::Duration::from_secs(2), - pausing.wait_until_blocked(), - ) - .await - .unwrap(); - - let second = tokio::time::timeout( - std::time::Duration::from_secs(2), - handle.execute( - mutation_identity_window(56, 10, 10_000), - Digest::from_bytes([57; 32]), - 21, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"two".to_vec())) - }, - ), - ) - .await - .unwrap() - .unwrap(); - assert_eq!(second.commit_sequence(), 2); - let value = handle - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(); - assert_eq!(i64::from_be_bytes(value.try_into().unwrap()), 2); - - for sequence in 3_u8..=64 { - let outcome = tokio::time::timeout( - std::time::Duration::from_secs(2), - handle.execute( - mutation_identity_window(sequence.saturating_add(54), 10, 10_000), - Digest::from_bytes([sequence.saturating_add(55); 32]), - 20 + i64::from(sequence), - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ), - ) - .await - .unwrap() - .unwrap(); - assert_eq!(outcome.commit_sequence(), u64::from(sequence)); - } - let sixty_fifth = handle.execute( - mutation_identity_window(119, 10, 10_000), - Digest::from_bytes([120; 32]), - 85, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ); - tokio::pin!(sixty_fifth); - assert!( - tokio::time::timeout(std::time::Duration::from_millis(100), &mut sixty_fifth) - .await - .is_err() - ); - let blocked = CellAuthority::new(fixture.layout.clone()) - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(blocked.value().root.as_ref().unwrap().commit_sequence, 0); - assert!(runtime.stats().unpublished_node_log_bytes() > 0); - - pausing.release(); - let outcome = tokio::time::timeout(std::time::Duration::from_secs(5), &mut sixty_fifth) - .await - .unwrap() - .unwrap(); - assert_eq!(outcome.commit_sequence(), 65); - tokio::time::timeout(std::time::Duration::from_secs(5), handle.drain()) - .await - .unwrap() - .unwrap(); - assert_eq!(runtime.stats().unpublished_node_log_bytes(), 0); - runtime.shutdown().await.unwrap(); - let released = CellAuthority::new(fixture.layout.clone()) - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(released.value().root.as_ref().unwrap().commit_sequence, 65); - assert_eq!(*authority.activations.lock().unwrap(), vec![1]); - assert!( - authority - .coverage - .lock() - .unwrap() - .last() - .is_some_and(|(epoch, covered)| *epoch == 1 && *covered >= 65) - ); -} -struct FaultFollowerTransport { - inner: crab_cell_runtime::node::log_transport::LocalFollowerTransport, - after_object_failure: Option>, - release: tokio::sync::Semaphore, - tickets: Mutex>, - receipts: Mutex>, -} - -impl NodeLogTransport for FaultFollowerTransport { - fn append<'a>( - &'a self, - member: NodeId, - request: AppendRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result> { - Box::pin(async move { - let first = crab_ltx::inspect_node_frame( - request.frames.first().unwrap().clone(), - Limits::default(), - )?; - let last = crab_ltx::inspect_node_frame( - request.frames.last().unwrap().clone(), - Limits::default(), - )?; - self.tickets - .lock() - .unwrap() - .push((member, first.scope(), last.scope())); - if let Some(store) = &self.after_object_failure { - store.wait_until_failed().await; - self.release - .acquire() - .await - .map_err(|_| crab_cell_runtime::Error::RuntimeClosed)? - .forget(); - } - let receipt = self.inner.append(member, request).await?; - self.receipts.lock().unwrap().push(receipt); - Ok(receipt) - }) - } - - fn seal<'a>( - &'a self, - member: NodeId, - request: SealRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result> { - self.inner.seal(member, request) - } - - fn tail<'a>( - &'a self, - member: NodeId, - request: TailRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result>> { - self.inner.tail(member, request) - } - - fn retire<'a>( - &'a self, - member: NodeId, - request: RetireRequest, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result> { - self.inner.retire(member, request) - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn follower_fsync_can_acknowledge_before_object_root_cas() { - exercise_fleet_ack_drain(false).await; -} - -#[tokio::test(flavor = "multi_thread")] -async fn fleet_ack_drain_reconciles_a_lost_root_cas_response() { - exercise_fleet_ack_drain(true).await; -} - -async fn exercise_fleet_ack_drain(lose_response: bool) { - let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let object_store: Arc = store.clone(); - let fixture = fixture_with_limits_and_store( - b"follower-before-object-cas", - Limits::default(), - Store::new(object_store), - ); - let session = SessionId::from_bytes([141; 16]); - let follower = NodeId::from_bytes([142; 16]); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - session, - ReplicaHost::default(), - ) - .unwrap(); - let responses = Arc::new(RecordingResponses::default()); - runtime.install_telemetry(responses.clone()).unwrap(); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - runtime.install_node_lease(lease.clone()).unwrap(); - let gate = DurabilityGate::new( - session, - NodeId::from_bytes(*session.as_bytes()), - 1, - [follower], - ) - .unwrap(); - let follower_directory = tempfile::TempDir::new().unwrap(); - let follower_store = crab_cell_runtime::FollowerStore::open( - follower_directory.path().to_owned(), - Limits::default(), - DiskBudget::new(1 << 30), - ) - .unwrap(); - let transport = Arc::new(FaultFollowerTransport { - inner: crab_cell_runtime::node::log_transport::LocalFollowerTransport::new( - follower, - follower_store.clone(), - ), - after_object_failure: None, - release: tokio::sync::Semaphore::new(0), - tickets: Mutex::new(Vec::new()), - receipts: Mutex::new(Vec::new()), - }); - let node_transport: Arc = transport.clone(); - let shipper = - NodeLogShipper::new(gate.clone(), Arc::clone(&node_transport), Limits::default()).unwrap(); - runtime - .install_node_durability( - fixture.target.application(), - Arc::new(NodeDurability::new( - gate, - shipper, - Arc::new(TestNodeAuthority::default()), - node_transport, - lease, - )), - ) - .unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - let authority = CellAuthority::new(fixture.layout.clone()); - let initial_root = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - assert_eq!(initial_root.commit_sequence, 0); - // Finish activation's actor turn before subscribing to command publication. - runtime.active_catalog_entries().await.unwrap(); - let mut publications = runtime.subscribe_publications(); - let request = mutation_identity_window(143, 10, 10_000); - let digest = Digest::from_bytes([144; 32]); - let calls = Arc::new(AtomicUsize::new(0)); - let observed = calls.clone(); - store.arm_next_update(); - let command_handle = handle.clone(); - let command = tokio::spawn(async move { - command_handle - .execute(request, digest, 20, 1_024, 1_024, move |transaction| { - observed.fetch_add(1, Ordering::SeqCst); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"fleet-before-cas".to_vec())) - }) - .await - }); - tokio::time::timeout( - std::time::Duration::from_secs(2), - store.wait_until_blocked(), - ) - .await - .expect("seed=141: object control CAS did not pause"); - let pending = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(pending.value().ltx_root(), Some(initial_root)); - let outcome = tokio::time::timeout(std::time::Duration::from_secs(2), command) - .await - .expect("seed=141: follower fsync should prove the command while CAS is paused") - .unwrap() - .unwrap(); - assert_eq!(outcome.commit_sequence(), 1); - assert!( - matches!( - publications.try_recv(), - Err(tokio::sync::broadcast::error::TryRecvError::Empty) - ), - "fleet-only acknowledgement emitted an object-publication hint" - ); - let (selected_member, first_frame, last_frame) = { - let tickets = transport.tickets.lock().unwrap(); - assert_eq!(tickets.len(), 1); - tickets[0] - }; - assert_eq!(selected_member, follower); - assert_eq!(first_frame.leader_session, *session.as_bytes()); - assert_eq!(first_frame.log_epoch, 1); - assert_eq!(first_frame.node_sequence, 1); - assert_eq!(last_frame.node_sequence, 1); - assert_eq!(first_frame.commit_sequence, 1); - assert_eq!( - transport.receipts.lock().unwrap().as_slice(), - &[FollowerReceipt { - base_sequence: 1, - durable_through: 1, - }] - ); - assert!(follower_store.retained_bytes() > 0); - eprintln!( - "fault_seed=141 schedule=follower_fsync_before_object_cas request={request:?} committed_sequence={} selected_follower_ticket=({selected_member:?},epoch={},{}..={}) follower_receipts={:?} observed_control={:?}", - outcome.commit_sequence(), - first_frame.log_epoch, - first_frame.node_sequence, - last_frame.node_sequence, - transport.receipts.lock().unwrap().as_slice(), - pending.value(), - ); - assert_eq!( - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root(), - Some(initial_root) - ); - let duplicate_calls = calls.clone(); - assert_eq!( - handle - .execute(request, digest, 21, 1_024, 1_024, move |_| { - duplicate_calls.fetch_add(1, Ordering::SeqCst); - Ok(HandlerOutcome::Success(b"duplicate".to_vec())) - }) - .await - .unwrap(), - outcome - ); - assert_eq!(calls.load(Ordering::SeqCst), 1); - let drain = handle.drain(); - tokio::pin!(drain); - let early = tokio::time::timeout(std::time::Duration::from_millis(50), &mut drain).await; - if lose_response { - store.lose_next_update_response(); - } - // Release the provider before asserting so a failed gate does not strand SQL work. - store.release(); - assert!( - early.is_err(), - "fleet-only acknowledgement must not complete drain" - ); - tokio::time::timeout(std::time::Duration::from_secs(5), &mut drain) - .await - .unwrap() - .unwrap(); - if lose_response { - assert!(store.lost_update_response_consumed()); - } - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - let control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - if control - .value() - .ltx_root() - .is_some_and(|root| root.commit_sequence == 1) - { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - let published_root = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - assert_eq!(published_root.position.txid, 2); - assert_ne!(published_root.digest, initial_root.digest); - let notified = tokio::time::timeout(std::time::Duration::from_secs(2), publications.recv()) - .await - .unwrap() - .unwrap(); - assert_eq!(notified.cell(), fixture.target.cell_id()); - runtime.shutdown().await.unwrap(); - assert_eq!( - responses.0.lock().unwrap().as_slice(), - &[ - CommandResponseSource::Fleet, - CommandResponseSource::Recorded - ] - ); - - let verified = fixture.replica.open_root(&published_root).await.unwrap(); - let recovered = fixture - ._directory - .path() - .join("follower-before-cas-recovered.sqlite"); - assert_eq!(verified.root(), published_root); - assert_eq!( - verified.restore(&recovered).await.unwrap(), - published_root.position - ); - let connection = crab_ltx::rusqlite::Connection::open(recovered).unwrap(); - assert_eq!( - connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) - .unwrap(), - 1 - ); - assert_eq!( - connection - .query_row( - "SELECT result FROM sys_requests WHERE request_id = ?1", - [request.request_id.as_bytes().as_slice()], - |row| row.get::<_, Vec>(0), - ) - .unwrap(), - b"fleet-before-cas" - ); -} - -#[tokio::test(flavor = "multi_thread")] -async fn fleet_proof_retains_owner_when_object_publication_fails_first() { - let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let object_store: Arc = store.clone(); - let fixture = fixture_with_limits_and_store( - b"fleet-object-failure", - Limits::default(), - Store::new(object_store), - ); - let session = SessionId::from_bytes([121; 16]); - let leader = NodeId::from_bytes([122; 16]); - let follower = NodeId::from_bytes([123; 16]); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - session, - ReplicaHost::default(), - ) - .unwrap(); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - runtime.install_node_lease(lease.clone()).unwrap(); - let gate = DurabilityGate::new(session, leader, 1, [follower]).unwrap(); - let follower_directory = tempfile::TempDir::new().unwrap(); - let follower_store = crab_cell_runtime::FollowerStore::open( - follower_directory.path().to_owned(), - Limits::default(), - DiskBudget::new(1 << 30), - ) - .unwrap(); - let transport = Arc::new(FaultFollowerTransport { - inner: crab_cell_runtime::node::log_transport::LocalFollowerTransport::new( - follower, - follower_store, - ), - after_object_failure: Some(store.clone()), - release: tokio::sync::Semaphore::new(0), - tickets: Mutex::new(Vec::new()), - receipts: Mutex::new(Vec::new()), - }); - let node_transport: Arc = transport.clone(); - let shipper = - NodeLogShipper::new(gate.clone(), Arc::clone(&node_transport), Limits::default()).unwrap(); - runtime - .install_node_durability( - fixture.target.application(), - Arc::new(NodeDurability::new( - gate, - shipper, - Arc::new(TestNodeAuthority::default()), - node_transport, - lease, - )), - ) - .unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - store.fail_puts(); - - let command = handle.execute( - mutation_identity_window(124, 10, 10_000), - Digest::from_bytes([125; 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"fleet-proof".to_vec())) - }, - ); - tokio::pin!(command); - tokio::select! { - () = store.wait_until_failed() => {} - result = &mut command => panic!("command answered before object failure: {result:?}"), - } - assert!( - tokio::time::timeout(std::time::Duration::from_millis(100), &mut command) - .await - .is_err() - ); - transport.release.add_permits(1); - let outcome = tokio::time::timeout(std::time::Duration::from_secs(2), &mut command) - .await - .unwrap() - .unwrap(); - assert_eq!(outcome.commit_sequence(), 1); - let (selected_member, first_frame, last_frame) = { - let tickets = transport.tickets.lock().unwrap(); - assert_eq!(tickets.len(), 1); - tickets[0] - }; - assert_eq!(selected_member, follower); - assert_eq!(first_frame.leader_session, *session.as_bytes()); - assert_eq!(first_frame.log_epoch, 1); - assert_eq!(first_frame.node_sequence, 1); - assert_eq!(last_frame.node_sequence, 1); - assert_eq!(first_frame.commit_sequence, 1); - assert_eq!( - transport.receipts.lock().unwrap().as_slice(), - &[FollowerReceipt { - base_sequence: 1, - durable_through: 1, - }] - ); - tokio::time::timeout(std::time::Duration::from_secs(2), store.wait_until_failed()) - .await - .unwrap(); - - let control = CellAuthority::new(fixture.layout.clone()) - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - eprintln!( - "fault_seed=121 schedule=object_failure_before_follower_fsync request={:?} committed_sequence={} selected_follower_ticket=({selected_member:?},epoch={},{}..={}) follower_receipts={:?} observed_control={:?}", - mutation_identity_window(124, 10, 10_000), - outcome.commit_sequence(), - first_frame.log_epoch, - first_frame.node_sequence, - last_frame.node_sequence, - transport.receipts.lock().unwrap().as_slice(), - control.value(), - ); - assert_eq!(control.value().root.as_ref().unwrap().commit_sequence, 0); - let value = handle - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(); - assert_eq!(i64::from_be_bytes(value.try_into().unwrap()), 1); - assert_eq!(runtime.stats().active_cells(), 1); - let pending = CellAuthority::new(fixture.layout.clone()) - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(pending.value().state, ControlState::Serving); - assert_eq!(pending.value().owner.as_ref().unwrap().session, session); - assert_eq!(pending.value().root.as_ref().unwrap().commit_sequence, 0); - assert!(runtime.stats().unpublished_node_log_bytes() > 0); - store.allow_puts(); - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - let control = CellAuthority::new(fixture.layout.clone()) - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - if control.value().root.as_ref().unwrap().commit_sequence == 1 - && runtime.stats().unpublished_node_log_bytes() == 0 - { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - runtime.shutdown().await.unwrap(); -} -#[tokio::test] -async fn post_commit_publication_failure_returns_resolvable_unknown_outcome() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - delete_control_root(&fixture).await; - let request = mutation_identity_window(14, 10, 10_000); - let digest = Digest::from_bytes([15; 32]); - assert!(matches!( - handle - .execute(request, digest, 20, 1_024, 1_024, |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"not-yet-published".to_vec())) - }) - .await, - Err(crab_cell_runtime::Error::OutcomeUnknown { - request_id, - operation_digest, - .. - }) if request_id == request.request_id && operation_digest == digest - )); - assert!(matches!( - handle - .execute(request, digest, 21, 1_024, 1_024, |_| { - Ok(HandlerOutcome::Success(Vec::new())) - }) - .await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert_eq!( - handle.resolve(request, digest, 21, 1_024).await.unwrap(), - Resolution::Unknown - ); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/durability/recovery.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/durability/recovery.rs deleted file mode 100644 index d021c03d1..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/durability/recovery.rs +++ /dev/null @@ -1,282 +0,0 @@ -//! Recovery inventory reads and lost-acknowledgement recovery. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn recovery_inventory_reads_catalog_heads_concurrently() { - let pausing = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let object_store: Arc = pausing.clone(); - let fixture = fixture_with_limits_and_store( - b"parallel-recovery-inventory", - Limits::default(), - Store::new(object_store), - ); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let authority = CellAuthority::new(fixture.layout.clone()); - pausing.require_parallel_catalog_heads(); - - let inventory = tokio::time::timeout( - std::time::Duration::from_secs(1), - crab_cell_runtime::node::log_recovery::recoverable_cells( - &catalog, - &authority, - SessionId::from_bytes([99; 16]), - 10, - ), - ) - .await - .expect("catalog head reads remained serial") - .unwrap(); - - assert!(inventory.is_empty()); -} -#[tokio::test(flavor = "multi_thread")] -async fn lost_ack_suffix_recovers_an_ambiguous_command_without_reexecution() { - let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let object_store: Arc = store.clone(); - let fixture = fixture_with_limits_and_store( - b"lost-ack-recovery", - Limits::default(), - Store::new(object_store), - ); - let leader = SessionId::from_bytes([126; 16]); - let follower = SessionId::from_bytes([127; 16]); - let member = NodeId::from_bytes(*follower.as_bytes()); - let successor = SessionId::from_bytes([128; 16]); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - leader, - ReplicaHost::default(), - ) - .unwrap(); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - runtime.install_node_lease(lease.clone()).unwrap(); - let follower_directory = tempfile::TempDir::new().unwrap(); - let follower_store = crab_cell_runtime::FollowerStore::open( - follower_directory.path().to_owned(), - Limits::default(), - DiskBudget::new(1 << 30), - ) - .unwrap(); - let lost_ack_transport = Arc::new(LostAckFollowerTransport { - inner: crab_cell_runtime::node::log_transport::LocalFollowerTransport::new( - member, - follower_store.clone(), - ), - acknowledged_once: AtomicBool::new(false), - lost_ticket: Mutex::new(None), - }); - let transport: Arc = lost_ack_transport.clone(); - let gate = - DurabilityGate::new(leader, NodeId::from_bytes(*leader.as_bytes()), 1, [member]).unwrap(); - let shipper = - NodeLogShipper::new(gate.clone(), Arc::clone(&transport), Limits::default()).unwrap(); - runtime - .install_node_durability( - fixture.target.application(), - Arc::new(NodeDurability::new( - gate, - shipper, - Arc::new(TestNodeAuthority::default()), - Arc::clone(&transport), - lease, - )), - ) - .unwrap(); - let handle = bootstrap_on(&runtime, &fixture, leader).await; - let first = handle - .execute( - mutation_identity_window(126, 10, 10_000), - Digest::from_bytes([126; 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"published".to_vec())) - }, - ) - .await - .unwrap(); - assert_eq!(first.commit_sequence(), 1); - let authority = CellAuthority::new(fixture.layout.clone()); - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - if current.value().root.as_ref().unwrap().commit_sequence == 1 - && runtime.stats().unpublished_node_log_bytes() == 0 - { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - - let request_identity = mutation_identity_window(127, 10, 10_000); - let operation_digest = Digest::from_bytes([127; 32]); - store.fail_puts(); - let result = tokio::time::timeout( - std::time::Duration::from_secs(5), - handle.execute( - request_identity, - operation_digest, - 21, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"ambiguous".to_vec())) - }, - ), - ) - .await - .unwrap(); - assert!(matches!( - result, - Err(crab_cell_runtime::Error::OutcomeUnknown { - request_id, - operation_digest: digest, - .. - }) if request_id == request_identity.request_id && digest == operation_digest - )); - tokio::time::timeout(std::time::Duration::from_secs(2), store.wait_until_failed()) - .await - .unwrap(); - tokio::time::timeout(std::time::Duration::from_secs(5), async { - while runtime.stats().active_cells() != 0 { - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - assert!(follower_store.retained_bytes() > 0); - store.allow_puts(); - assert!(runtime.shutdown().await.is_err()); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let stale = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(stale.value().root.as_ref().unwrap().commit_sequence, 1); - let (selected_member, first_frame, last_frame, fsynced_receipt) = - lost_ack_transport.lost_ticket.lock().unwrap().unwrap(); - assert_eq!(selected_member, member); - assert_eq!(first_frame.leader_session, *leader.as_bytes()); - assert_eq!(first_frame.log_epoch, 1); - assert_eq!(first_frame.node_sequence, 2); - assert_eq!(last_frame.node_sequence, 2); - assert_eq!(first_frame.commit_sequence, 2); - assert_eq!(fsynced_receipt.durable_through, 2); - eprintln!( - "fault_seed=126 schedule=lost_follower_ack_and_object_failure request={request_identity:?} operation={operation_digest:?} committed_sequence=2 selected_follower_ticket=({selected_member:?},epoch={},{}..={}) fsynced_receipt={fsynced_receipt:?} observed_control={:?}", - first_frame.log_epoch, - first_frame.node_sequence, - last_frame.node_sequence, - stale.value(), - ); - let fenced = fence_log_session(&fixture.layout, leader, successor, follower, 1).await; - let recovery = crab_cell_runtime::node::log_recovery::NodeLogRecovery::from_fenced( - Arc::clone(&transport), - &fenced, - Limits::default(), - ) - .unwrap(); - let manifests = crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - fixture.layout.clone(), - Limits::default(), - ); - let coordinator = crab_cell_runtime::node::log_recovery::RecoveryCoordinator::new( - recovery, - manifests.clone(), - ); - let inventory = - crab_cell_runtime::node::log_recovery::recoverable_cells(&catalog, &authority, leader, 10) - .await - .unwrap(); - let directory = crab_cell_runtime::node::NodeDirectory::new( - fixture.layout.clone(), - Digest::from_bytes([90; 32]), - Digest::from_bytes([91; 32]), - Digest::from_bytes([92; 32]), - ); - let completed = coordinator - .recover_and_seal(&directory, fenced, inventory, 10_002) - .await - .unwrap(); - let attached = completed.controls.into_iter().next().unwrap(); - let successor_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - successor, - ) - .unwrap(); - let restored = successor_runtime - .takeover_restored( - proof, - fixture.replica.clone(), - authority, - attached, - completed.takeover, - manifests, - fixture._directory.path().join("lost-ack-successor.sqlite"), - Owner { - session: successor, - endpoint: "https://lost-ack-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - - let replayed = restored - .execute( - request_identity, - operation_digest, - 30, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"executed-twice".to_vec())) - }, - ) - .await - .unwrap(); - assert!(matches!( - replayed, - StoredOutcome::Success { - ref result, - commit_sequence: 2 - } if result == b"ambiguous" - )); - let value = restored - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(); - assert_eq!(i64::from_be_bytes(value.try_into().unwrap()), 2); - restored.drain().await.unwrap(); - successor_runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/execution.rs deleted file mode 100644 index 30a7e9df9..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution.rs +++ /dev/null @@ -1,10 +0,0 @@ -//! Dispatch, compaction, and command/query/resolve semantics. - -use super::*; - -mod admission; -mod capacity; -mod dispatcher; -mod handlers; -mod resolve; -mod shutdown; diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/admission.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/admission.rs deleted file mode 100644 index 9f8c41ec8..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/admission.rs +++ /dev/null @@ -1,63 +0,0 @@ -//! Per-Cell request admission caps. - -use super::*; - -#[tokio::test] -async fn per_cell_request_admission_caps_inflight_and_queued_commands() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - let (started_tx, started_rx) = mpsc::channel(); - let (release_tx, release_rx) = mpsc::channel(); - let first = { - let handle = handle.clone(); - tokio::spawn(async move { - handle - .execute( - mutation_identity_window(29, 10, 10_000), - Digest::from_bytes([29; 32]), - 20, - 0, - 1, - move |_| { - started_tx.send(()).unwrap(); - release_rx.recv().unwrap(); - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - }) - }; - tokio::task::spawn_blocking(move || started_rx.recv().unwrap()) - .await - .unwrap(); - - let (result_tx, mut result_rx) = tokio::sync::mpsc::unbounded_channel(); - for byte in 30..94 { - let handle = handle.clone(); - let result_tx = result_tx.clone(); - tokio::spawn(async move { - let result = handle - .execute( - mutation_identity_window(byte, 10, 10_000), - Digest::from_bytes([byte; 32]), - 21, - 0, - 1, - |_| Ok(HandlerOutcome::Success(Vec::new())), - ) - .await; - let _ = result_tx.send(result); - }); - } - drop(result_tx); - assert!(matches!( - result_rx.recv().await.unwrap(), - Err(crab_cell_runtime::Error::Capacity(_)) - )); - release_tx.send(()).unwrap(); - first.await.unwrap().unwrap(); - for _ in 0..63 { - assert!(result_rx.recv().await.unwrap().is_ok()); - } - handle.drain().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/capacity.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/capacity.rs deleted file mode 100644 index 2d794e53c..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/capacity.rs +++ /dev/null @@ -1,393 +0,0 @@ -//! Database admission includes runtime receipts and persists across owner changes. - -use super::*; -use crab_cell_runtime::cell::actor::CellHandle; -use crab_cell_runtime::primitives::capacity::SCHEMA; - -const RESULT_BYTES: usize = 100 * 1024; - -async fn hold_capacity(handle: &CellHandle) { - handle.execute( - mutation_identity_window(80, 10, 10_000), Digest::from_bytes([80; 32]), - 20, 1024, 1024, |transaction| { - transaction.execute_batch(SCHEMA)?; - transaction.execute_batch( - "INSERT INTO capacity_reservations VALUES(X'01', 96); UPDATE capacity_total SET pages = 96;", - )?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ).await.unwrap(); -} - -fn release_sql(transaction: &crab_ltx::rusqlite::Transaction<'_>) -> crab_cell_runtime::Result<()> { - transaction - .execute_batch("DELETE FROM capacity_reservations; UPDATE capacity_total SET pages = 0;")?; - Ok(()) -} - -async fn assert_held(handle: &CellHandle) { - let observed = handle.query(64, 64, |connection| { - let values = connection.query_row( - "SELECT value, (SELECT pages FROM capacity_total), (SELECT SUM(pages) FROM capacity_reservations) FROM counter", [], - |row| Ok([row.get::<_, i64>(0)?, row.get::<_, i64>(1)?, row.get::<_, i64>(2)?]), - )?; - Ok(values.into_iter().flat_map(i64::to_be_bytes).collect()) - }).await.unwrap(); - assert_eq!( - observed, - [0_i64, 96, 96] - .into_iter() - .flat_map(i64::to_be_bytes) - .collect::>() - ); -} - -#[tokio::test(flavor = "multi_thread")] -async fn reservation_survives_receipt_refusal_failed_release_and_owner_restore() { - let fixture = fixture_with_limits( - b"reserved-command", - Limits { - max_database_bytes: 512 * 1024, - ..Limits::default() - }, - ); - let (runtime, handle, _pool) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - hold_capacity(&handle).await; - let failed_identity = mutation_identity_window(81, 10, 10_000); - let failed_digest = Digest::from_bytes([81; 32]); - let failed = handle - .execute( - failed_identity, - failed_digest, - 21, - 1024, - RESULT_BYTES, - |transaction| { - transaction.execute("UPDATE counter SET value = 1", [])?; - Ok(HandlerOutcome::Success(vec![1; RESULT_BYTES])) - }, - ) - .await; - assert!(matches!(failed, Err(crab_cell_runtime::Error::Capacity(_)))); - assert_held(&handle).await; - assert_eq!( - handle - .resolve(failed_identity, failed_digest, 22, RESULT_BYTES) - .await - .unwrap(), - Resolution::Absent - ); - - for (request, reject) in [(82, false), (83, true)] { - let result = handle - .execute( - mutation_identity_window(request, 10, 10_000), - Digest::from_bytes([request; 32]), - 23, - 1024, - 1024, - move |transaction| { - release_sql(transaction)?; - transaction.execute("UPDATE counter SET value = 99", [])?; - if reject { - return Ok(HandlerOutcome::Rejected(Vec::new())); - } - Err(crab_cell_runtime::Error::Command("injected after release")) - }, - ) - .await; - if reject { - assert!(matches!(result, Ok(StoredOutcome::Rejected { .. }))); - } else { - assert!(matches!( - result, - Err(crab_cell_runtime::Error::Command("injected after release")) - )); - } - assert_held(&handle).await; - } - runtime.shutdown().await.unwrap(); - let session = SessionId::from_bytes([5; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(2, 10).unwrap(), - 16 * 1024 * 1024, - session, - ) - .unwrap(); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - observed, - fixture._directory.path().join("capacity-restored.sqlite"), - Owner { - session, - endpoint: "https://capacity-restored.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_held(&restored).await; - assert!(matches!( - restored - .execute( - failed_identity, - failed_digest, - 24, - 1024, - RESULT_BYTES, - |transaction| { - transaction.execute("UPDATE counter SET value = 1", [])?; - Ok(HandlerOutcome::Success(vec![1; RESULT_BYTES])) - }, - ) - .await, - Err(crab_cell_runtime::Error::Capacity(_)) - )); - let committed = restored - .execute( - mutation_identity_window(84, 10, 10_000), - Digest::from_bytes([84; 32]), - 25, - 1024, - RESULT_BYTES, - |transaction| { - release_sql(transaction)?; - transaction.execute("UPDATE counter SET value = 1", [])?; - Ok(HandlerOutcome::Success(vec![1; RESULT_BYTES])) - }, - ) - .await - .unwrap(); - assert!(matches!( - committed, - StoredOutcome::Success { - commit_sequence: 3, - .. - } - )); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn reservation_protects_effect_inbox_and_allows_retry_after_release() { - let fixture = fixture_with_limits( - b"reserved-effect", - Limits { - max_database_bytes: 512 * 1024, - ..Limits::default() - }, - ); - let (runtime, handle, _pool) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - hold_capacity(&handle).await; - let delivery = InboxDelivery { - effect_id: [85; 32], - operation_digest: Digest::from_bytes([85; 32]), - expires_at_ms: 10_000, - }; - let apply = |transaction: &crab_ltx::rusqlite::Transaction<'_>| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(vec![1; RESULT_BYTES])) - }; - assert!(matches!( - handle - .deliver_effect(delivery, 21, 1024, RESULT_BYTES, apply) - .await, - Err(crab_cell_runtime::Error::Capacity(_)) - )); - assert_held(&handle).await; - assert_eq!( - handle - .resolve_effect(delivery, 22, RESULT_BYTES) - .await - .unwrap(), - Resolution::Absent - ); - handle - .execute( - mutation_identity_window(86, 10, 10_000), - Digest::from_bytes([86; 32]), - 23, - 1024, - 1024, - |transaction| { - release_sql(transaction)?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - assert!(matches!( - handle - .deliver_effect(delivery, 24, 1024, RESULT_BYTES, apply) - .await - .unwrap(), - StoredOutcome::Success { - commit_sequence: 3, - .. - } - )); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn bootstrap_refuses_overcommitted_database_reservations_before_publication() { - let fixture = fixture_with_limits( - b"reserved-bootstrap", - Limits { - max_database_bytes: 512 * 1024, - ..Limits::default() - }, - ); - let session = SessionId::from_bytes([4; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 8).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .provision( - CatalogEntry::new( - &fixture.target, - CatalogRole::Repository, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .create_initial( - &proof, - IncarnationId::from_bytes([2; 16]), - Owner { - session, - endpoint: "https://capacity-bootstrap.internal:8081".into(), - }, - ) - .await - .unwrap(); - let result = runtime.bootstrap(proof, fixture.replica.clone(), authority.clone(), observed, - fixture.database.clone(), |transaction| { - transaction.execute_batch(SCHEMA)?; - transaction.execute_batch("INSERT INTO capacity_reservations VALUES(X'01', 128); UPDATE capacity_total SET pages = 128;")?; - Ok(()) - }).await; - assert!(matches!(result, Err(crab_cell_runtime::Error::Capacity(_)))); - assert!( - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .root - .is_none() - ); - runtime.shutdown().await.unwrap(); -} - -struct CapacityMigration; - -impl crab_cell_runtime::registry::CellModule for CapacityMigration { - const NAME: &'static str = "capacity-migration"; - - fn descriptor(&self) -> &'static crab_cell_runtime::registry::ModuleDescriptor { - use crab_cell_runtime::registry::{ - MigrationDescriptor, ModuleDescriptor, NamespaceDescriptor, RetainedCodeDescriptor, - }; - static DESCRIPTOR: std::sync::OnceLock = std::sync::OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: Self::NAME, - source_digest: Digest::from_bytes([91; 32]), - retained_codes: Box::leak(Box::new([RetainedCodeDescriptor { code: Digest::from_bytes([5; 32]), schema_min: 1, schema_max: 1 }])), - schema_min: 1, schema_max: 2, - migrations: Box::leak(Box::new([SCHEMA, - "CREATE TABLE migration_payload(value BLOB); INSERT INTO migration_payload VALUES(zeroblob(102400)); UPDATE counter SET value = 1;" - ].into_iter().enumerate().map(|(index, sql)| MigrationDescriptor { - version: index as u32 + 1, sql, digest: Digest::from_bytes(*blake3::hash(sql.as_bytes()).as_bytes()), - }).collect::>())), - commands: &[], queries: &[], workflow_definitions: &[], activity_types: &[], - namespaces: Box::leak(Box::new([NamespaceDescriptor { - id: NamespaceId::from_bytes([6; 16]), name: "capacity-migration", role: CatalogRole::Repository, - shards: 1, effect_targets: &[], dead_letter: None, - }])), - }) - } - - fn register( - self, - _: &mut crab_cell_runtime::registry::RegistryBuilder, - ) -> crab_cell_runtime::Result<()> { - Ok(()) - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn migration_cannot_spend_prepared_work_reservations() { - use crab_cell_runtime::registry::{BuildDescriptor, RegistryBuilder}; - let fixture = fixture_with_limits( - b"reserved-migration", - Limits { - max_database_bytes: 512 * 1024, - ..Limits::default() - }, - ); - let (runtime, handle, _pool) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - hold_capacity(&handle).await; - let authority = CellAuthority::new(fixture.layout.clone()); - let before = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let mut registry = RegistryBuilder::new(BuildDescriptor { - source_revision: "capacity-test".into(), - cargo_lock_digest: Digest::from_bytes([92; 32]), - }); - registry.register(CapacityMigration).unwrap(); - let registry = registry.finish().unwrap(); - let plan = registry - .next_migration(fixture.target.namespace(), Digest::from_bytes([5; 32]), 1) - .unwrap() - .unwrap(); - assert!(matches!( - handle.migrate(plan, 30).await, - Err(crab_cell_runtime::Error::Capacity(_)) - )); - let after = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(after.value().root, before.value().root); - // Migrations drain the old capability even on refusal. Inspect the original - // SQL file after shutdown to prove both schema and handler writes rolled back. - runtime.shutdown().await.unwrap(); - let connection = crab_ltx::rusqlite::Connection::open_with_flags( - &fixture.database, - crab_ltx::rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY, - ) - .unwrap(); - let state = connection.query_row("SELECT value, (SELECT pages FROM capacity_total), (SELECT schema_version FROM sys_meta), (SELECT COUNT(*) FROM sqlite_schema WHERE name = 'migration_payload') FROM counter", [], |row| Ok((row.get::<_,i64>(0)?, row.get::<_,i64>(1)?, row.get::<_,i64>(2)?, row.get::<_,i64>(3)?))).unwrap(); - assert_eq!(state, (0, 96, 1, 0)); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/dispatcher.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/dispatcher.rs deleted file mode 100644 index b482e122b..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/dispatcher.rs +++ /dev/null @@ -1,336 +0,0 @@ -//! Command dispatch, compaction, and publication ordering. - -use super::*; - -#[tokio::test] -async fn dispatcher_serializes_and_publishes_commands_before_drain() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - let first = { - let handle = handle.clone(); - tokio::spawn(async move { - handle - .execute( - mutation_identity_window(6, 10, 10_000), - Digest::from_bytes([7; 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"one".to_vec())) - }, - ) - .await - }) - }; - let second = { - let handle = handle.clone(); - tokio::spawn(async move { - handle - .execute( - mutation_identity_window(8, 10, 10_000), - Digest::from_bytes([9; 32]), - 21, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"two".to_vec())) - }, - ) - .await - }) - }; - assert!(matches!( - first.await.unwrap().unwrap(), - StoredOutcome::Success { ref result, commit_sequence: 1 } if result == b"one" - )); - assert!(matches!( - second.await.unwrap().unwrap(), - StoredOutcome::Success { ref result, commit_sequence: 2 } if result == b"two" - )); - handle.drain().await.unwrap(); - - let released = CellAuthority::new(fixture.layout.clone()) - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(released.value().state, ControlState::Idle); - assert!(released.value().owner.is_none()); - assert_eq!( - released.value().next_due_ms, - Some(10_000 + 24 * 60 * 60 * 1000) - ); - - let connection = crab_ltx::rusqlite::Connection::open(&fixture.database).unwrap(); - assert_eq!( - connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) - .unwrap(), - 2 - ); -} -#[tokio::test] -async fn dispatcher_compacts_before_segment_admission_is_exhausted() { - let fixture = fixture_with_limits( - b"compacting-repository", - Limits { - max_segments: 4, - ..Limits::default() - }, - ); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - for sequence in 1_u8..=10 { - assert!(matches!( - handle - .execute( - mutation_identity_window(sequence, 10, 10_000), - Digest::from_bytes([sequence.saturating_add(20); 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(), - StoredOutcome::Success { - commit_sequence, - .. - } if commit_sequence == u64::from(sequence) - )); - } - - let authority = CellAuthority::new(fixture.layout.clone()); - let control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let root = control.value().ltx_root().unwrap(); - assert!( - fixture - .replica - .open_root(&root) - .await - .unwrap() - .segment_count() - < 4 - ); - handle.drain().await.unwrap(); - let restored_directory = tempfile::TempDir::new().unwrap(); - let restored = restored_directory.path().join("restored.sqlite"); - fixture - .replica - .open_root(&root) - .await - .unwrap() - .restore(&restored) - .await - .unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) - .unwrap(), - 10 - ); -} -#[tokio::test] -async fn dispatcher_promotes_after_burst_becomes_quiet() { - let pausing = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let store: Arc = pausing.clone(); - let fixture = fixture_with_limits_and_store( - b"quiet-compaction-runtime", - Limits::default(), - Store::with_retry( - store, - RetryPolicy { - max_attempts: 1, - base: std::time::Duration::from_millis(1), - cap: std::time::Duration::from_millis(1), - }, - ), - ); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - for sequence in 1_u8..=8 { - handle - .execute( - mutation_identity_window(sequence, 10, 10_000), - Digest::from_bytes([sequence.saturating_add(30); 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - } - - let authority = CellAuthority::new(fixture.layout.clone()); - let before = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - let initial_segments = fixture - .replica - .open_root(&before) - .await - .unwrap() - .segment_count(); - assert!(initial_segments >= 8); - pausing.fail_next_put_transiently(); - tokio::time::timeout( - std::time::Duration::from_secs(3), - pausing.wait_until_failed(), - ) - .await - .unwrap(); - tokio::time::sleep(std::time::Duration::from_millis(300)).await; - assert_eq!( - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root(), - Some(before) - ); - - let promoted = tokio::time::timeout(std::time::Duration::from_secs(3), async { - loop { - let root = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - if fixture - .replica - .open_root(&root) - .await - .unwrap() - .segment_count() - < initial_segments - { - break root; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); - assert_eq!(promoted.position, before.position); - assert_eq!(promoted.commit_sequence, before.commit_sequence); - handle.drain().await.unwrap(); - let restored_directory = tempfile::TempDir::new().unwrap(); - let restored = restored_directory.path().join("restored.sqlite"); - fixture - .replica - .open_root(&promoted) - .await - .unwrap() - .restore(&restored) - .await - .unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) - .unwrap(), - 8 - ); -} -#[tokio::test(flavor = "multi_thread")] -async fn command_arriving_during_quiet_compaction_waits_for_publisher() { - let pausing = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let store: Arc = pausing.clone(); - let fixture = fixture_with_limits_and_store( - b"quiet-compaction-queued-command", - Limits::default(), - Store::new(store), - ); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - for sequence in 1_u8..=8 { - handle - .execute( - mutation_identity_window(sequence, 10, 10_000), - Digest::from_bytes([sequence.saturating_add(70); 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - } - pausing.arm(); - tokio::time::timeout( - std::time::Duration::from_secs(3), - pausing.wait_until_blocked(), - ) - .await - .unwrap(); - let ninth = handle.execute( - mutation_identity_window(9, 10, 10_000), - Digest::from_bytes([79; 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ); - tokio::pin!(ninth); - assert!( - tokio::time::timeout(std::time::Duration::from_millis(100), &mut ninth) - .await - .is_err() - ); - pausing.release(); - assert_eq!(ninth.await.unwrap().commit_sequence(), 9); - handle.drain().await.unwrap(); - let root = CellAuthority::new(fixture.layout.clone()) - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - let restored_directory = tempfile::TempDir::new().unwrap(); - let restored = restored_directory.path().join("restored.sqlite"); - fixture - .replica - .open_root(&root) - .await - .unwrap() - .restore(&restored) - .await - .unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) - .unwrap(), - 9 - ); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/handlers.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/handlers.rs deleted file mode 100644 index 2a3206dcd..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/handlers.rs +++ /dev/null @@ -1,472 +0,0 @@ -//! Handler deadlines, panics, rollbacks, and query interruption. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn queued_query_deadline_does_not_fence_untouched_cell() { - queued_deadline(1, "query").await; - queued_deadline(2, "query").await; -} - -#[tokio::test(flavor = "multi_thread")] -async fn queued_mutation_deadline_prevents_late_writes_without_fencing() { - for operation in ["command", "effect"] { - queued_deadline(2, operation).await; - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn queued_resolution_deadline_preserves_owner_for_retry() { - for operation in ["resolve", "resolve_effect"] { - queued_deadline(2, operation).await; - } -} - -async fn queued_deadline(workers: usize, operation: &str) { - let first = fixture_for(b"deadline-blocker"); - let lane = |cell: crab_cell_runtime::identity::CellId| { - let prefix: [u8; 8] = cell.as_bytes()[..8].try_into().unwrap(); - u64::from_be_bytes(prefix) as usize % workers - }; - // With two permits the second operation reaches the worker's actual - // queue; one permit exercises expiry while waiting for job admission. - let second = (0..100) - .map(|index| fixture_for(format!("deadline-queued-{index}").as_bytes())) - .find(|fixture| lane(fixture.target.cell_id()) == lane(first.target.cell_id())) - .unwrap(); - let session = SessionId::from_bytes([4; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(workers, 2).unwrap(), - 16 * 1024 * 1024, - session, - ) - .unwrap(); - let blocker = bootstrap_on(&runtime, &first, session).await; - let queued = bootstrap_on(&runtime, &second, session).await; - let (started_tx, started_rx) = mpsc::channel(); - let (release_tx, release_rx) = mpsc::channel(); - let blocking = tokio::spawn(async move { - blocker - .query(64, 64, move |_| { - started_tx.send(()).unwrap(); - release_rx - .recv_timeout(std::time::Duration::from_secs(10)) - .unwrap(); - Ok(Vec::new()) - }) - .await - }); - tokio::task::spawn_blocking(move || started_rx.recv().unwrap()) - .await - .unwrap(); - let entered = Arc::new(AtomicBool::new(false)); - let executed = entered.clone(); - let identity = mutation_identity_window(42, 10, 10_000); - let digest = Digest::from_bytes([43; 32]); - let delivery = InboxDelivery { - effect_id: [42; 32], - operation_digest: digest, - expires_at_ms: 10_000, - }; - let result = tokio::time::timeout(std::time::Duration::from_secs(7), async { - let write = move |transaction: &crab_ltx::rusqlite::Transaction<'_>| { - executed.store(true, Ordering::SeqCst); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }; - match operation { - "query" => { - let entered = entered.clone(); - queued - .query(64, 64, move |_| { - entered.store(true, Ordering::SeqCst); - Ok(Vec::new()) - }) - .await - .map(|_| ()) - } - "command" => queued - .execute(identity, digest, 20, 64, 64, write) - .await - .map(|_| ()), - "effect" => queued - .deliver_effect(delivery, 20, 64, 64, write) - .await - .map(|_| ()), - "resolve" => queued - .resolve(identity, digest, 20, 64) - .await - .map(|result| { - assert_eq!(result, Resolution::Unknown); - }), - "resolve_effect" => queued.resolve_effect(delivery, 20, 64).await.map(|result| { - assert_eq!(result, Resolution::Unknown); - }), - _ => unreachable!(), - } - }) - .await; - // Release before assertions so failures cannot leave teardown blocked. - release_tx.send(()).unwrap(); - let _ = blocking.await.unwrap(); - let result = result.unwrap(); - if operation.starts_with("resolve") { - result.unwrap(); - } else { - assert!( - matches!(result, Err(crab_cell_runtime::Error::Deadline)), - "{operation}: {result:?}" - ); - } - let readable = queued - .query(64, 64, |connection| { - let value: i64 = - connection.query_row("SELECT value FROM counter", [], |row| row.get(0))?; - Ok(value.to_le_bytes().to_vec()) - }) - .await; - runtime.shutdown().await.unwrap(); - assert_eq!(readable.unwrap(), 0_i64.to_le_bytes(), "{operation}"); - assert!( - !entered.load(Ordering::SeqCst), - "{operation} executed after queue expiry" - ); -} - -#[tokio::test(flavor = "multi_thread")] -async fn native_handler_deadline_discards_late_commit_and_reopens_authoritative_root() { - let fixture = fixture(); - let (runtime, handle, _pool) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - let authority = CellAuthority::new(fixture.layout.clone()); - let before = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .root - .clone(); - let (started_tx, started_rx) = mpsc::channel(); - let (release_tx, release_rx) = mpsc::channel(); - let mutation = { - let handle = handle.clone(); - tokio::spawn(async move { - handle - .execute( - mutation_identity_window(42, 10, 10_000), - Digest::from_bytes([43; 32]), - 20, - 1_024, - 1_024, - move |transaction| { - started_tx.send(()).unwrap(); - release_rx.recv().unwrap(); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - }) - }; - tokio::task::spawn_blocking(move || started_rx.recv().unwrap()) - .await - .unwrap(); - let outcome = tokio::time::timeout(std::time::Duration::from_secs(7), mutation) - .await - .unwrap() - .unwrap(); - match outcome { - Err(crab_cell_runtime::Error::OutcomeUnknown { source, .. }) => { - assert!(matches!(*source, crab_cell_runtime::Error::Deadline)); - } - other => panic!("expected deadline outcome, got {other:?}"), - } - assert!(matches!( - handle.query(1, 1, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::Fenced) - )); - - release_tx.send(()).unwrap(); - let after = tokio::time::timeout(std::time::Duration::from_secs(2), async { - loop { - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - if current.value().state == ControlState::Idle { - break current; - } - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - }) - .await - .unwrap(); - assert_eq!(after.value().root, before); - assert!(after.value().owner.is_none()); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let session = SessionId::from_bytes([4; 16]); - let recovered = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - after, - fixture._directory.path().join("deadline-recovered.sqlite"), - Owner { - session, - endpoint: "https://deadline-recovered.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!( - recovered - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 0_i64.to_be_bytes() - ); - recovered.drain().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn native_handler_panic_discards_transaction_and_reopens_authoritative_root() { - let fixture = fixture_for(b"panicking-command"); - let (runtime, handle, _pool) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - let authority = CellAuthority::new(fixture.layout.clone()); - let before = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .root - .clone(); - - let outcome = handle - .execute( - mutation_identity_window(44, 10, 10_000), - Digest::from_bytes([45; 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - panic!("command handler panic") - }, - ) - .await; - match outcome { - Err(crab_cell_runtime::Error::OutcomeUnknown { source, .. }) => { - assert!(matches!(*source, crab_cell_runtime::Error::NativePanic)); - } - other => panic!("expected native panic outcome, got {other:?}"), - } - assert!(matches!( - handle.query(1, 1, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::Fenced) - )); - - let idle = tokio::time::timeout(std::time::Duration::from_secs(2), async { - loop { - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - if current.value().state == ControlState::Idle { - break current; - } - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - }) - .await - .unwrap(); - assert_eq!(idle.value().root, before); - assert!(idle.value().owner.is_none()); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let recovered = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - idle, - fixture._directory.path().join("panic-recovered.sqlite"), - Owner { - session: SessionId::from_bytes([4; 16]), - endpoint: "https://panic-recovered.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!( - recovered - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 0_i64.to_be_bytes() - ); - recovered.drain().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn sqlite_query_is_interrupted_at_wall_deadline() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - let result = tokio::time::timeout( - std::time::Duration::from_secs(7), - handle.query(64, 64, |connection| { - let value = connection.query_row( - "WITH RECURSIVE counter(value) AS (VALUES(0) UNION ALL SELECT value + 1 FROM counter WHERE value < 1000000000) SELECT sum(value) FROM counter", - [], - |row| row.get::<_, i64>(0), - )?; - Ok(value.to_be_bytes().to_vec()) - }), - ) - .await - .unwrap(); - assert!(matches!(result, Err(crab_cell_runtime::Error::Deadline))); - assert!(matches!( - handle.query(1, 1, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::Fenced) - )); -} -#[tokio::test] -async fn proven_handler_rollback_keeps_the_cell_servable() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - assert!(matches!( - handle - .execute( - mutation_identity_window(16, 10, 10_000), - Digest::from_bytes([17; 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 10", [])?; - Err(crab_cell_runtime::Error::Command("application failure")) - }, - ) - .await, - Err(crab_cell_runtime::Error::Command("application failure")) - )); - assert!(matches!( - handle - .execute( - mutation_identity_window(24, 10, 10_000), - Digest::from_bytes([25; 32]), - 20, - 1_024, - 1, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 10", [])?; - Ok(HandlerOutcome::Success(b"too large".to_vec())) - }, - ) - .await, - Err(crab_cell_runtime::Error::Command( - "handler result exceeds command limit" - )) - )); - assert!(matches!( - handle - .execute( - mutation_identity_window(18, 10, 10_000), - Digest::from_bytes([19; 32]), - 21, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"recovered".to_vec())) - }, - ) - .await - .unwrap(), - StoredOutcome::Success { ref result, commit_sequence: 1 } if result == b"recovered" - )); - handle.drain().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn automatic_sqlite_rollback_keeps_the_cell_servable() { - let fixture = fixture_with_limits( - b"automatic-rollback", - Limits { - max_database_bytes: 512 * 1024, - ..Limits::default() - }, - ); - let (runtime, handle, _pool) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - let failed = handle - .execute( - mutation_identity_window(71, 10, 10_000), - Digest::from_bytes([72; 32]), - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = 99", [])?; - transaction.execute("CREATE TABLE oversized(value BLOB)", [])?; - transaction.execute("INSERT INTO oversized VALUES(zeroblob(1048576))", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await; - assert!( - matches!(failed, Err(crab_cell_runtime::Error::Sqlite(ref error)) - if error.sqlite_error_code() == Some(crab_ltx::rusqlite::ErrorCode::DiskFull)), - "expected a rolled-back capacity refusal, got {failed:?}" - ); - let outcome = handle - .execute( - mutation_identity_window(73, 10, 10_000), - Digest::from_bytes([74; 32]), - 21, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - let value: i64 = - transaction.query_row("SELECT value FROM counter", [], |row| row.get(0))?; - Ok(HandlerOutcome::Success(value.to_be_bytes().to_vec())) - }, - ) - .await - .unwrap(); - assert_eq!( - outcome, - StoredOutcome::Success { - result: 1_i64.to_be_bytes().to_vec(), - commit_sequence: 1, - } - ); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/resolve.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/resolve.rs deleted file mode 100644 index b16b74090..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/resolve.rs +++ /dev/null @@ -1,211 +0,0 @@ -//! Query and resolve ordering, identity, and fencing. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn query_waits_for_preceding_publication_and_cannot_write() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - let (started_tx, started_rx) = mpsc::channel(); - let (release_tx, release_rx) = mpsc::channel(); - let mutation = { - let handle = handle.clone(); - tokio::spawn(async move { - handle - .execute( - mutation_identity_window(52, 10, 10_000), - Digest::from_bytes([53; 32]), - 20, - 1_024, - 1_024, - move |transaction| { - started_tx.send(()).unwrap(); - release_rx.recv().unwrap(); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - }) - }; - tokio::task::spawn_blocking(move || started_rx.recv().unwrap()) - .await - .unwrap(); - let query = { - let handle = handle.clone(); - tokio::spawn(async move { - handle - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - }) - }; - release_tx.send(()).unwrap(); - mutation.await.unwrap().unwrap(); - assert_eq!(query.await.unwrap().unwrap(), 1_i64.to_be_bytes()); - - assert!(matches!( - handle.query(1, 1, |_| Ok(vec![0; 2])).await, - Err(crab_cell_runtime::Error::Command( - "query result exceeds command limit" - )) - )); - assert!(matches!( - handle - .query(64, 64, |connection| { - connection.execute("UPDATE counter SET value = 99", [])?; - Ok(Vec::new()) - }) - .await, - Err(crab_cell_runtime::Error::Sqlite(_)) - )); - assert_eq!( - handle - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 1_i64.to_be_bytes() - ); - handle.drain().await.unwrap(); -} -#[tokio::test] -async fn cancelled_command_waiter_is_resolved_by_original_identity() { - let fixture = fixture(); - use super::super::durability::RecordingResponses; - use crab_cell_runtime::fleet::telemetry::CommandResponseSource; - - let (runtime, handle, _) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - let responses = Arc::new(RecordingResponses::default()); - runtime.install_telemetry(responses.clone()).unwrap(); - let request = mutation_identity_window(10, 10, 10_000); - let digest = Digest::from_bytes([11; 32]); - let (started_tx, started_rx) = mpsc::channel(); - let (release_tx, release_rx) = mpsc::channel(); - let waiting = { - let handle = handle.clone(); - tokio::spawn(async move { - handle - .execute(request, digest, 20, 1_024, 1_024, move |transaction| { - started_tx.send(()).unwrap(); - release_rx.recv().unwrap(); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"survived".to_vec())) - }) - .await - }) - }; - tokio::task::spawn_blocking(move || started_rx.recv().unwrap()) - .await - .unwrap(); - waiting.abort(); - assert!(matches!(waiting.await, Err(error) if error.is_cancelled())); - release_tx.send(()).unwrap(); - - assert!(matches!( - handle - .execute(request, digest, 21, 1_024, 1_024, |_| { - Ok(HandlerOutcome::Success(b"wrong".to_vec())) - }) - .await - .unwrap(), - StoredOutcome::Success { ref result, commit_sequence: 1 } if result == b"survived" - )); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - assert_eq!( - responses.0.lock().unwrap().as_slice(), - &[CommandResponseSource::Recorded] - ); -} -#[tokio::test] -async fn resolve_distinguishes_committed_absent_conflict_and_expired() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - let request = mutation_identity_window(54, 10, 10_000); - let digest = Digest::from_bytes([55; 32]); - let outcome = handle - .execute(request, digest, 20, 1_024, 1_024, |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Rejected(b"recorded".to_vec())) - }) - .await - .unwrap(); - assert_eq!( - handle.resolve(request, digest, 21, 1_024).await.unwrap(), - Resolution::Committed(outcome) - ); - assert!(matches!( - handle - .resolve(request, Digest::from_bytes([56; 32]), 21, 1_024) - .await, - Err(crab_cell_runtime::Error::RequestConflict) - )); - assert_eq!( - handle - .resolve( - mutation_identity_window(57, 10, 10_000), - Digest::from_bytes([58; 32]), - 21, - 1_024 - ) - .await - .unwrap(), - Resolution::Absent - ); - assert_eq!( - handle - .resolve( - mutation_identity_window(59, 10, 10_000), - Digest::from_bytes([60; 32]), - 10_000, - 1_024 - ) - .await - .unwrap(), - Resolution::Expired - ); - handle.drain().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn resolve_waits_for_inflight_publication_and_returns_unknown_after_fence() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - delete_control_root(&fixture).await; - let request = mutation_identity_window(61, 10, 10_000); - let digest = Digest::from_bytes([62; 32]); - let (started_tx, started_rx) = mpsc::channel(); - let (release_tx, release_rx) = mpsc::channel(); - let mutation = { - let handle = handle.clone(); - tokio::spawn(async move { - handle - .execute(request, digest, 20, 1_024, 1_024, move |transaction| { - started_tx.send(()).unwrap(); - release_rx.recv().unwrap(); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }) - .await - }) - }; - tokio::task::spawn_blocking(move || started_rx.recv().unwrap()) - .await - .unwrap(); - let resolution = { - let handle = handle.clone(); - tokio::spawn(async move { handle.resolve(request, digest, 21, 1_024).await }) - }; - release_tx.send(()).unwrap(); - assert!(matches!( - mutation.await.unwrap(), - Err(crab_cell_runtime::Error::OutcomeUnknown { .. }) - )); - assert_eq!(resolution.await.unwrap().unwrap(), Resolution::Unknown); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/shutdown.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/shutdown.rs deleted file mode 100644 index 7c168b5ea..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/execution/shutdown.rs +++ /dev/null @@ -1,264 +0,0 @@ -//! Cancellation and shutdown drain behaviour. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn effect_delivery_survives_cancellation_and_exact_root_restore() { - let fixture = fixture(); - use super::super::durability::RecordingResponses; - use crab_cell_runtime::fleet::telemetry::CommandResponseSource; - - let (runtime, handle, _) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - let responses = Arc::new(RecordingResponses::default()); - runtime.install_telemetry(responses.clone()).unwrap(); - let delivery = InboxDelivery { - effect_id: [70; 32], - operation_digest: Digest::from_bytes([71; 32]), - expires_at_ms: 10_000, - }; - let (started_tx, started_rx) = mpsc::channel(); - let (release_tx, release_rx) = mpsc::channel(); - let first = tokio::spawn({ - let handle = handle.clone(); - async move { - handle - .deliver_effect(delivery, 20, 64, 1_024, move |transaction| { - started_tx.send(()).unwrap(); - release_rx.recv().unwrap(); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"delivered".to_vec())) - }) - .await - } - }); - tokio::task::spawn_blocking(move || started_rx.recv().unwrap()) - .await - .unwrap(); - first.abort(); - assert!(first.await.unwrap_err().is_cancelled()); - release_tx.send(()).unwrap(); - - let replayed = handle - .deliver_effect(delivery, 21, 64, 1_024, |_| { - panic!("published inbox delivery must not execute twice") - }) - .await - .unwrap(); - assert_eq!( - replayed, - StoredOutcome::Success { - result: b"delivered".to_vec(), - commit_sequence: 1, - } - ); - assert_eq!( - handle.resolve_effect(delivery, 22, 1_024).await.unwrap(), - Resolution::Committed(replayed.clone()) - ); - assert!(matches!( - handle - .resolve_effect( - InboxDelivery { - operation_digest: Digest::from_bytes([72; 32]), - ..delivery - }, - 22, - 1_024, - ) - .await, - Err(crab_cell_runtime::Error::RequestConflict) - )); - assert!(matches!( - handle - .resolve_effect( - InboxDelivery { - expires_at_ms: delivery.expires_at_ms + 1, - ..delivery - }, - 22, - 1_024, - ) - .await, - Err(crab_cell_runtime::Error::RequestConflict) - )); - assert_eq!( - handle - .resolve_effect( - InboxDelivery { - effect_id: [73; 32], - ..delivery - }, - 22, - 1_024, - ) - .await - .unwrap(), - Resolution::Absent - ); - let authority = CellAuthority::new(fixture.layout.clone()); - let published = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(published.value().root.as_ref().unwrap().commit_sequence, 1); - assert_eq!( - handle - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 1_i64.to_be_bytes() - ); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - assert_eq!( - responses.0.lock().unwrap().as_slice(), - &[CommandResponseSource::Recorded] - ); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let session = SessionId::from_bytes([45; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - session, - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - idle, - fixture._directory.path().join("effect-restored.sqlite"), - Owner { - session, - endpoint: "https://effect-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!( - restored.resolve_effect(delivery, 30, 1_024).await.unwrap(), - Resolution::Committed(replayed) - ); - assert_eq!( - restored - .resolve_effect(delivery, delivery.expires_at_ms, 1_024) - .await - .unwrap(), - Resolution::Expired - ); - assert_eq!( - restored - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 1_i64.to_be_bytes() - ); - restored.drain().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn runtime_shutdown_drains_accepted_work_and_releases_all_owners() { - let fixture = fixture(); - let (runtime, handle, pool) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - let second_fixture = fixture_for(b"repository-2"); - let second = bootstrap_on(&runtime, &second_fixture, SessionId::from_bytes([4; 16])).await; - let (started_tx, started_rx) = mpsc::channel(); - let (release_tx, release_rx) = mpsc::channel(); - let mutation = { - let handle = handle.clone(); - tokio::spawn(async move { - handle - .execute( - mutation_identity_window(111, 10, 10_000), - Digest::from_bytes([112; 32]), - 20, - 1_024, - 1_024, - move |transaction| { - started_tx.send(()).unwrap(); - release_rx.recv().unwrap(); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"published".to_vec())) - }, - ) - .await - }) - }; - tokio::task::spawn_blocking(move || started_rx.recv().unwrap()) - .await - .unwrap(); - - let shutdown = tokio::spawn({ - let runtime = runtime.clone(); - async move { runtime.shutdown().await } - }); - while !runtime.is_shutting_down() { - tokio::task::yield_now().await; - } - assert!(matches!( - handle.query(1, 1, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::RuntimeClosed) - )); - assert!(!shutdown.is_finished()); - - release_tx.send(()).unwrap(); - assert!(matches!( - mutation.await.unwrap().unwrap(), - StoredOutcome::Success { - ref result, - commit_sequence: 1 - } if result == b"published" - )); - tokio::time::timeout(std::time::Duration::from_secs(5), shutdown) - .await - .unwrap() - .unwrap() - .unwrap(); - assert!(matches!( - runtime.shutdown().await, - Err(crab_cell_runtime::Error::RuntimeClosed) - )); - assert!(matches!( - pool.shutdown().await, - Err(crab_cell_runtime::Error::RuntimeClosed) - )); - - let released = CellAuthority::new(fixture.layout.clone()) - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(released.value().state, ControlState::Idle); - assert!(released.value().owner.is_none()); - assert_eq!(released.value().root.as_ref().unwrap().commit_sequence, 1); - let second_released = CellAuthority::new(second_fixture.layout.clone()) - .load(second.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(second_released.value().state, ControlState::Idle); - assert!(second_released.value().owner.is_none()); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/idle.rs deleted file mode 100644 index f1640dc9b..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle.rs +++ /dev/null @@ -1,72 +0,0 @@ -//! Idle release, eviction, churn, and capacity release. - -use super::*; - -mod acquire; -mod bootstrap; -mod churn; -mod release; - -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn active_targets_preserve_tenants_after_cancelled_caller() { - let first = Arc::new(fixture()); - let mut second = fixture_with_limits_and_store( - first.target.partition(), - Limits::default(), - first.layout.store().clone(), - ); - second.target = CellTarget::new( - TenantId::from_bytes([9; 16]), - first.target.application(), - first.target.namespace(), - first.target.partition(), - ) - .unwrap(); - second.replica = CellReplica::new( - second.layout.clone(), - *second.target.cell_id().as_bytes(), - [2; 16], - Limits::default(), - ) - .unwrap(); - let session = SessionId::from_bytes([4; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 2).unwrap(), 4 << 20, session).unwrap(); - let started = Arc::new(Notify::new()); - let activation = { - let runtime = runtime.clone(); - let first = first.clone(); - let started = started.clone(); - tokio::spawn(async move { - let handle = bootstrap_on(&runtime, &first, session).await; - // The runtime has returned ownership, but caller-side bookkeeping - // has not run. Cancellation here must not hide the resident owner. - started.notify_one(); - std::future::pending::<()>().await; - handle - }) - }; - tokio::time::timeout(std::time::Duration::from_secs(10), started.notified()) - .await - .unwrap(); - activation.abort(); - assert!(matches!(activation.await, Err(error) if error.is_cancelled())); - tokio::time::timeout(std::time::Duration::from_secs(10), async { - while runtime.active_cell_targets().await.unwrap().is_empty() { - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - }) - .await - .unwrap(); - let second_handle = bootstrap_on(&runtime, &second, session).await; - let mut targets = runtime.active_cell_targets().await.unwrap(); - targets.sort_by_key(|target| *target.cell_id().as_bytes()); - let mut expected = vec![first.target.clone(), second.target.clone()]; - expected.sort_by_key(|target| *target.cell_id().as_bytes()); - assert_eq!(targets, expected); - second_handle.drain().await.unwrap(); - assert_eq!( - runtime.active_cell_targets().await.unwrap(), - vec![first.target.clone()] - ); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/acquire.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/acquire.rs deleted file mode 100644 index f690bb7e1..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/acquire.rs +++ /dev/null @@ -1,137 +0,0 @@ -//! Idle receiver failure and stop-acquiring behaviour. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn failed_idle_receiver_does_not_leave_authority_owned() { - let fixture = fixture_for(b"receiver-activation-failure"); - let session = SessionId::from_bytes([112; 16]); - let first_runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&first_runtime, &fixture, session).await; - handle.drain().await.unwrap(); - first_runtime.shutdown().await.unwrap(); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let expected_root = idle.value().root.clone(); - let successor = SessionId::from_bytes([113; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 16 * 1024 * 1024, - successor, - ) - .unwrap(); - let missing_parent = fixture - ._directory - .path() - .join("receiver-parent-does-not-exist") - .join("receiver.sqlite"); - assert!( - runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority.clone(), - idle, - missing_parent, - Owner { - session: successor, - endpoint: "https://receiver-failure.internal:8081".into(), - }, - ) - .await - .is_err() - ); - - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(current.value().state, ControlState::Idle); - assert!(current.value().owner.is_none()); - assert_eq!(current.value().root, expected_root); - runtime.shutdown().await.unwrap(); -} -#[tokio::test] -async fn stop_acquiring_keeps_existing_cell_serving() { - let first = fixture_for(b"scale-down-serving"); - let second = fixture_for(b"scale-down-new"); - let session = SessionId::from_bytes([96; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 2).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &first, session).await; - runtime.stop_acquiring().unwrap(); - assert!(!runtime.is_acquiring()); - assert_eq!( - handle - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 0_i64.to_be_bytes() - ); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - second.layout.clone(), - second.target.tenant(), - ); - let proof = catalog - .provision( - CatalogEntry::new( - &second.target, - CatalogRole::Repository, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(second.layout.clone()); - let observed = authority - .create_initial( - &proof, - IncarnationId::from_bytes([2; 16]), - Owner { - session, - endpoint: "https://node.internal:8081".into(), - }, - ) - .await - .unwrap(); - let result = runtime - .bootstrap( - proof, - second.replica.clone(), - authority, - observed, - second._directory.path().join("blocked.sqlite"), - |_| Ok(()), - ) - .await; - assert!(matches!( - result, - Err(crab_cell_runtime::Error::CellDraining) - )); - assert_eq!(runtime.unreleased_cell_count().await.unwrap(), 1); - handle.drain().await.unwrap(); - assert_eq!(runtime.unreleased_cell_count().await.unwrap(), 0); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/bootstrap.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/bootstrap.rs deleted file mode 100644 index bf10f919c..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/bootstrap.rs +++ /dev/null @@ -1,154 +0,0 @@ -//! Failed and panicking bootstrap capacity release. - -use super::*; - -#[tokio::test] -async fn failed_bootstrap_keeps_control_unpublished_and_releases_cell_capacity() { - let fixture = fixture(); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .provision( - CatalogEntry::new( - &fixture.target, - CatalogRole::Repository, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let session = SessionId::from_bytes([4; 16]); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .create_initial( - &proof, - IncarnationId::from_bytes([2; 16]), - Owner { - session, - endpoint: "https://node.internal:8081".into(), - }, - ) - .await - .unwrap(); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 2 * 1024 * 1024, session).unwrap(); - assert!(matches!( - runtime - .bootstrap( - proof.clone(), - fixture.replica.clone(), - authority.clone(), - observed, - fixture.database.clone(), - |transaction| { - transaction.execute("CREATE TABLE should_rollback(value INTEGER)", [])?; - Err(crab_cell_runtime::Error::Command("migration rejected")) - }, - ) - .await, - Err(crab_cell_runtime::Error::Command("migration rejected")) - )); - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(current.value().state, ControlState::Recovering); - assert!(current.value().root.is_none()); - - let handle = runtime - .bootstrap( - proof, - fixture.replica, - authority, - current, - fixture.database.with_file_name("replacement.sqlite"), - |transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (0)", - )?; - Ok(()) - }, - ) - .await - .unwrap(); - handle.drain().await.unwrap(); -} -#[tokio::test] -async fn panicking_bootstrap_keeps_worker_alive_and_releases_cell_capacity() { - let fixture = fixture_for(b"panicking-bootstrap"); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .provision( - CatalogEntry::new( - &fixture.target, - CatalogRole::Repository, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let session = SessionId::from_bytes([4; 16]); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .create_initial( - &proof, - IncarnationId::from_bytes([2; 16]), - Owner { - session, - endpoint: "https://node.internal:8081".into(), - }, - ) - .await - .unwrap(); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 2 * 1024 * 1024, session).unwrap(); - - assert!(matches!( - runtime - .bootstrap( - proof.clone(), - fixture.replica.clone(), - authority.clone(), - observed, - fixture.database.clone(), - |_| panic!("bootstrap initializer panic"), - ) - .await, - Err(crab_cell_runtime::Error::NativePanic) - )); - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(current.value().state, ControlState::Recovering); - assert!(current.value().root.is_none()); - - let handle = runtime - .bootstrap( - proof, - fixture.replica, - authority, - current, - fixture.database.with_file_name("panic-replacement.sqlite"), - |transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (0)", - )?; - Ok(()) - }, - ) - .await - .unwrap(); - handle.drain().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/churn.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/churn.rs deleted file mode 100644 index b3a30fd83..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/churn.rs +++ /dev/null @@ -1,405 +0,0 @@ -//! Pressure observation, churn eviction, and primitive inventory gates. - -use super::*; - -#[tokio::test] -async fn actor_pressure_observation_uses_hysteresis_and_shared_eviction_path() { - let session = SessionId::from_bytes([41; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 1_024, session).unwrap(); - // The actor samples its own ledger on the wall clock, so keep both - // observations ahead of any tick that could run while this test is - // scheduled. - let base_ms = i64::try_from( - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap() - + 5_000; - let high = PressureSample { - at_ms: base_ms, - memory_used_permille: 900, - disk_used_permille: 100, - jobs_used_permille: 100, - stale: false, - }; - assert_eq!( - runtime.observe_pressure(high).await.unwrap(), - PressureState::Normal - ); - assert_eq!( - runtime - .observe_pressure(PressureSample { - at_ms: base_ms + 1_000, - ..high - }) - .await - .unwrap(), - PressureState::Shedding - ); - assert_eq!(runtime.evict_idle(1).await.unwrap(), 0); - runtime.shutdown().await.unwrap(); -} - -/// Reserves retained bytes until the memory ledger reaches `target_permille`. -fn press_memory_ledger( - runtime: &CellRuntime, - target_permille: u64, -) -> crab_cell_runtime::cell::actor::NodeByteReservation { - let stats = runtime.stats(); - let limit = - u64::try_from(stats.resident_capacity_bytes() + stats.retained_capacity_bytes()).unwrap(); - let used = u64::try_from(stats.resident_bytes() + stats.retained_bytes()).unwrap(); - let target = limit * target_permille / 1_000; - assert!(target > used, "the reservation already exceeds the target"); - runtime - .try_reserve_node_bytes(usize::try_from(target - used).unwrap()) - .unwrap() -} - -#[tokio::test(flavor = "multi_thread")] -async fn node_ledger_pressure_sheds_a_settled_cell_without_an_external_sample() { - let fixture = fixture_for(b"ledger-pressure-shed"); - let session = SessionId::from_bytes([121; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - // Hold the reservation across the samples the classifier needs to see. - let reservation = press_memory_ledger(&runtime, 850); - - let authority = CellAuthority::new(fixture.layout.clone()); - tokio::time::timeout(std::time::Duration::from_secs(15), async { - loop { - if runtime.unreleased_cell_count().await.unwrap() == 0 { - break; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .expect("sustained ledger pressure must shed the settled Cell"); - - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(idle.value().state, ControlState::Idle); - assert!(idle.value().owner.is_none()); - assert_eq!(runtime.stats().active_cells(), 0); - - drop(reservation); - drop(handle); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn node_ledger_below_the_soft_reserve_keeps_the_settled_cell() { - let fixture = fixture_for(b"ledger-pressure-hold"); - let session = SessionId::from_bytes([122; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - let reservation = press_memory_ledger(&runtime, 500); - - // Longer than the classifier's dwell and the actor's sample interval. - tokio::time::sleep(std::time::Duration::from_secs(2)).await; - assert_eq!(runtime.unreleased_cell_count().await.unwrap(), 1); - assert_eq!(runtime.stats().active_cells(), 1); - - drop(reservation); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn churn_evicts_idle_cells_and_restores_exact_roots() { - let first = fixture_for(b"churn-first"); - let second = fixture_for(b"churn-second"); - let third = fixture_for(b"churn-third"); - let session = SessionId::from_bytes([74; 16]); - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 2).unwrap(), - 16 * 1024 * 1024, - session, - ReplicaHost::default() - .with_local_disk_budget(DiskBudget::new(4 * Limits::default().max_capture_bytes)), - ) - .unwrap(); - - let first_handle = bootstrap_on(&runtime, &first, session).await; - let second_handle = bootstrap_on(&runtime, &second, session).await; - assert!(runtime.local_disk_budget().used() > 0); - - let authority_first = CellAuthority::new(first.layout.clone()); - let authority_second = CellAuthority::new(second.layout.clone()); - let root_first = authority_first - .load(first.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - let root_second = authority_second - .load(second.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - - tokio::time::timeout(std::time::Duration::from_secs(3), async { - loop { - if runtime.evict_idle(1).await.unwrap() == 1 { - break; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); - - let (evicted_fixture, evicted_authority, evicted_root, evicted_idle) = - tokio::time::timeout(std::time::Duration::from_secs(3), async { - loop { - let first_control = authority_first - .load(first.target.cell_id()) - .await - .unwrap() - .unwrap(); - if first_control.value().state == ControlState::Idle { - break (&first, &authority_first, root_first, first_control); - } - let second_control = authority_second - .load(second.target.cell_id()) - .await - .unwrap() - .unwrap(); - if second_control.value().state == ControlState::Idle { - break (&second, &authority_second, root_second, second_control); - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); - assert_eq!(evicted_idle.value().ltx_root(), Some(evicted_root)); - assert_eq!(runtime.stats().active_cells(), 1); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - evicted_fixture.layout.clone(), - evicted_fixture.target.tenant(), - ); - let proof = catalog - .lookup(evicted_fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - evicted_fixture.replica.clone(), - evicted_authority.clone(), - evicted_idle, - evicted_fixture - ._directory - .path() - .join("churn-restored.sqlite"), - Owner { - session, - endpoint: "https://churn-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!( - restored - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 0_i64.to_be_bytes() - ); - restored.drain().await.unwrap(); - - let third_handle = bootstrap_on(&runtime, &third, session).await; - assert_eq!(runtime.stats().active_cells(), 2); - third_handle.drain().await.unwrap(); - if evicted_fixture.target.cell_id() == first.target.cell_id() { - second_handle.drain().await.unwrap(); - } else { - first_handle.drain().await.unwrap(); - } - assert_eq!(runtime.stats().active_cells(), 0); - assert_eq!(runtime.local_disk_budget().used(), 0); - runtime.shutdown().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn mixed_primitive_inventory_blocks_churn_until_drain_and_restores_root() { - mixed_primitive_inventory_churn(Store::new(Arc::new(InMemory::new())), Path::from("runtime")) - .await; -} -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires an isolated pre-created RustFS bucket, prefix and explicit test credentials"] -async fn rustfs_mixed_primitive_inventory_churn_preserves_exact_roots() { - let required = |name| std::env::var(name).unwrap_or_else(|_| panic!("missing {name}")); - let store = build_explicit_store( - &required("CRAB_CELL_TEST_BUCKET"), - ObjectStoreCredentials::Aws { - access_key_id: required("AWS_ACCESS_KEY_ID"), - secret_access_key: required("AWS_SECRET_ACCESS_KEY"), - session_token: None, - region: "us-east-1".into(), - }, - Some(&required("CRAB_CELL_TEST_ENDPOINT")), - true, - ) - .unwrap(); - let prefix = Path::from(format!( - "{}/mixed-primitive-churn", - required("CRAB_CELL_TEST_PREFIX") - )); - mixed_primitive_inventory_churn(store, prefix).await; -} -async fn mixed_primitive_inventory_churn(store: Store, prefix: Path) { - let queue = fixture_with_limits_and_store_at_prefix( - b"mixed-queue", - Limits::default(), - store.clone(), - prefix.clone(), - ); - let workflow = fixture_with_limits_and_store_at_prefix( - b"mixed-workflow", - Limits::default(), - store.clone(), - prefix.clone(), - ); - let third = - fixture_with_limits_and_store_at_prefix(b"mixed-third", Limits::default(), store, prefix); - let session = SessionId::from_bytes([80; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 2).unwrap(), 16 * 1024 * 1024, session).unwrap(); - - let queue_handle = bootstrap_role_on( - &runtime, - &queue, - session, - CatalogRole::Queue, - |transaction| { - install_queue_schema(transaction)?; - transaction.execute( - "INSERT INTO queue_messages VALUES (zeroblob(16), X'01', 0, 0, 1, 1000, NULL, NULL, NULL, NULL)", - [], - )?; - Ok(()) - }, - ) - .await; - let workflow_handle = bootstrap_role_on( - &runtime, - &workflow, - session, - CatalogRole::Workflow, - |transaction| { - install_workflow_schema(transaction)?; - transaction.execute( - "INSERT INTO workflow_runs VALUES (X'02', zeroblob(16), zeroblob(32), 0, X'', 0, NULL, NULL)", - [], - )?; - Ok(()) - }, - ) - .await; - - wait_for_persisted_work( - &queue_handle, - CatalogRole::Queue, - "maintenance release is blocked by retained Queue messages", - ) - .await; - wait_for_persisted_work( - &workflow_handle, - CatalogRole::Workflow, - "maintenance release is blocked by retained Workflow runs", - ) - .await; - - assert_eq!(runtime.evict_idle(2).await.unwrap(), 0); - assert_eq!(runtime.stats().active_cells(), 2); - - let queue_authority = CellAuthority::new(queue.layout.clone()); - let queue_root = queue_authority - .load(queue.target.cell_id()) - .await - .unwrap() - .unwrap(); - let queue_root = queue_root.value().ltx_root().unwrap(); - - queue_handle.drain().await.unwrap(); - workflow_handle.drain().await.unwrap(); - assert_eq!(runtime.stats().active_cells(), 0); - let queue_idle = queue_authority - .load(queue.target.cell_id()) - .await - .unwrap() - .unwrap(); - - let queue_proof = crab_cell_runtime::cell::catalog::CellCatalog::new( - queue.layout.clone(), - queue.target.tenant(), - ) - .lookup(queue.target.cell_id()) - .await - .unwrap() - .unwrap(); - let restored = runtime - .acquire_idle_restored( - queue_proof, - queue.replica.clone(), - queue_authority, - queue_idle, - queue._directory.path().join("mixed-queue-restored.sqlite"), - Owner { - session, - endpoint: "https://mixed-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!( - restored - .query(64, 64, |connection| { - let count = - connection.query_row("SELECT count(*) FROM queue_messages", [], |row| { - row.get::<_, i64>(0) - })?; - Ok(count.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 1_i64.to_be_bytes() - ); - assert_eq!( - crab_cell_runtime::control::authority::CellAuthority::new(queue.layout.clone()) - .load(queue.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root(), - Some(queue_root) - ); - restored.drain().await.unwrap(); - - let third_handle = bootstrap_on(&runtime, &third, session).await; - assert_eq!(runtime.stats().active_cells(), 1); - third_handle.drain().await.unwrap(); - assert_eq!(runtime.stats().active_cells(), 0); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/release.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/release.rs deleted file mode 100644 index 6cab341ac..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/idle/release.rs +++ /dev/null @@ -1,219 +0,0 @@ -//! Exact idle release generation, inventory, and drain rules. - -use super::*; - -#[tokio::test] -async fn exact_idle_release_checks_generation_and_confirms_authority_release() { - let fixture = fixture_for(b"exact-idle-release"); - let session = SessionId::from_bytes([91; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let _handle = bootstrap_on(&runtime, &fixture, session).await; - let candidates = runtime.idle_transfer_candidates().await.unwrap(); - assert_eq!(candidates.len(), 1); - let (cell, generation, _, role) = candidates[0]; - assert_eq!(role, CatalogRole::Repository); - assert!( - runtime - .release_idle_cell(cell, SessionId::from_bytes([92; 16]), generation) - .await - .is_err() - ); - assert!( - runtime - .release_idle_cell(cell, session, generation + 1) - .await - .is_err() - ); - runtime - .release_idle_cell(cell, session, generation) - .await - .unwrap(); - assert_eq!(runtime.stats().active_cells(), 0); - let idle = CellAuthority::new(fixture.layout.clone()) - .load(cell) - .await - .unwrap() - .unwrap(); - assert_eq!(idle.value().state, ControlState::Idle); - assert!(idle.value().owner.is_none()); - runtime.shutdown().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn exact_idle_release_blocks_new_work_while_inventory_is_pending() { - let target_fixture = fixture_for(b"exact-idle-preflight"); - let blocker_fixture = fixture_for(b"exact-idle-preflight-blocker"); - let session = SessionId::from_bytes([96; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 2).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let target = bootstrap_on(&runtime, &target_fixture, session).await; - let blocker = bootstrap_on(&runtime, &blocker_fixture, session).await; - let candidates = runtime.idle_transfer_candidates().await.unwrap(); - let (cell, generation, _, _) = candidates - .iter() - .copied() - .find(|(cell, _, _, _)| *cell == target_fixture.target.cell_id()) - .unwrap(); - - let (started, started_signal) = tokio::sync::oneshot::channel(); - let blocker_task = tokio::spawn(async move { - blocker - .query(1, 1, move |_| { - let _ = started.send(()); - std::thread::sleep(std::time::Duration::from_millis(250)); - Ok(Vec::new()) - }) - .await - }); - started_signal.await.unwrap(); - - let release_runtime = runtime.clone(); - let release_task = tokio::spawn(async move { - release_runtime - .release_idle_cell(cell, session, generation) - .await - }); - tokio::time::timeout(std::time::Duration::from_secs(1), async { - loop { - let still_candidate = runtime - .idle_transfer_candidates() - .await - .unwrap() - .into_iter() - .any(|(candidate, _, _, _)| candidate == cell); - if !still_candidate { - break; - } - tokio::time::sleep(std::time::Duration::from_millis(5)).await; - } - }) - .await - .unwrap(); - - let result = target - .execute( - mutation_identity_window(97, 10, 10_000), - Digest::from_bytes([97; 32]), - 20, - 64, - 64, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await; - assert!(matches!( - result, - Err(crab_cell_runtime::Error::CellDraining) - )); - - assert!(blocker_task.await.unwrap().is_ok()); - release_task.await.unwrap().unwrap(); - assert_eq!(runtime.stats().active_cells(), 1); - runtime.shutdown().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn exact_idle_release_drains_work_admitted_before_transfer() { - let fixture = fixture_for(b"exact-idle-admitted-before-transfer"); - let session = SessionId::from_bytes([99; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - let (_, generation, _, _) = runtime.idle_transfer_candidates().await.unwrap()[0]; - - let (started, started_signal) = tokio::sync::oneshot::channel(); - let query_handle = handle.clone(); - let query_task = tokio::spawn(async move { - query_handle - .query(1, 1, move |_| { - let _ = started.send(()); - std::thread::sleep(std::time::Duration::from_millis(250)); - Ok(Vec::new()) - }) - .await - }); - started_signal.await.unwrap(); - - let release_runtime = runtime.clone(); - let release_task = tokio::spawn(async move { - release_runtime - .release_idle_cell(fixture.target.cell_id(), session, generation) - .await - }); - - assert!(query_task.await.unwrap().is_ok()); - release_task.await.unwrap().unwrap(); - assert_eq!(runtime.stats().active_cells(), 0); - runtime.shutdown().await.unwrap(); -} -#[tokio::test] -async fn exact_idle_release_refuses_persisted_work() { - let fixture = fixture_for(b"exact-idle-blocked"); - let session = SessionId::from_bytes([93; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_role_on( - &runtime, - &fixture, - session, - CatalogRole::Repository, - |transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (0)", - )?; - transaction.execute( - "INSERT INTO sys_effects(effect_id, destination, operation, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, created_sequence, result) VALUES (?1, ?2, ?3, 0, 0, 20, 1000, NULL, NULL, 1, NULL)", - crab_ltx::rusqlite::params![&[1_u8; 32], &[2_u8; 32], &[3_u8]], - )?; - Ok(()) - }, - ) - .await; - let (_, generation, _, _) = runtime.idle_transfer_candidates().await.unwrap()[0]; - assert!( - runtime - .release_idle_cell(fixture.target.cell_id(), session, generation) - .await - .is_err() - ); - assert_eq!(runtime.stats().active_cells(), 1); - assert!(matches!( - handle.drain().await, - Err(crab_cell_runtime::Error::CellDraining) - )); - runtime.shutdown().await.unwrap(); -} -#[tokio::test] -async fn exact_idle_release_allows_settled_queue_rows() { - let fixture = fixture_for(b"exact-idle-settled-queue"); - let session = SessionId::from_bytes([98; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let _handle = bootstrap_role_on( - &runtime, - &fixture, - session, - CatalogRole::Queue, - |transaction| { - install_queue_schema(transaction)?; - transaction.execute( - "INSERT INTO queue_messages(message_id, payload, state, attempt, due_at_ms, expires_at_ms, token, lease_until_ms, result_code) VALUES (zeroblob(16), X'', 2, 0, 0, 1000, NULL, NULL, 0)", - [], - )?; - transaction.execute( - "INSERT INTO queue_dedup VALUES (zeroblob(16), zeroblob(32), zeroblob(16), 1000)", - [], - )?; - Ok(()) - }, - ) - .await; - let (_, generation, _, _) = runtime.idle_transfer_candidates().await.unwrap()[0]; - runtime - .release_idle_cell(fixture.target.cell_id(), session, generation) - .await - .unwrap(); - assert_eq!(runtime.stats().active_cells(), 0); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership.rs deleted file mode 100644 index b96a61f98..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership.rs +++ /dev/null @@ -1,9 +0,0 @@ -//! Leases, fencing, takeover, and successor acquisition. - -use super::*; - -mod lease; -mod recovery; -mod sparse_process; -mod succession; -mod takeover; diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/lease.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/lease.rs deleted file mode 100644 index ea124da41..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/lease.rs +++ /dev/null @@ -1,143 +0,0 @@ -//! Node-lease fencing for a live Cell runtime. - -use super::*; - -#[tokio::test] -async fn fleet_runtime_stays_fenced_until_one_live_node_lease_is_installed() { - let session = SessionId::from_bytes([41; 16]); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 1_024, - session, - ReplicaHost::default(), - ) - .unwrap(); - - assert!(matches!( - runtime.try_reserve_node_bytes(1), - Err(crab_cell_runtime::Error::Fenced) - )); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - runtime.install_node_lease(lease.clone()).unwrap(); - assert!(runtime.install_node_lease(lease.clone()).is_err()); - drop(runtime.try_reserve_node_bytes(1).unwrap()); - - lease.fence(); - assert!(matches!( - runtime.try_reserve_node_bytes(1), - Err(crab_cell_runtime::Error::Fenced) - )); - runtime.shutdown().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn node_lease_expiry_hides_an_inflight_committed_command() { - let fixture = fixture_for(b"node-lease-output-gate"); - let session = SessionId::from_bytes([42; 16]); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - session, - ReplicaHost::default(), - ) - .unwrap(); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - runtime.install_node_lease(lease.clone()).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - let (entered_tx, entered_rx) = mpsc::channel(); - let (resume_tx, resume_rx) = mpsc::channel(); - - let command = tokio::spawn(async move { - handle - .execute( - mutation_identity_window(43, 10, 10_000), - Digest::from_bytes([44; 32]), - 20, - 1, - 16, - move |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - entered_tx.send(()).unwrap(); - resume_rx.recv().unwrap(); - Ok(HandlerOutcome::Success(b"hidden".to_vec())) - }, - ) - .await - }); - tokio::task::spawn_blocking(move || { - entered_rx - .recv_timeout(std::time::Duration::from_secs(5)) - .unwrap() - }) - .await - .unwrap(); - lease.fence(); - resume_tx.send(()).unwrap(); - - match command.await.unwrap() { - Err(crab_cell_runtime::Error::OutcomeUnknown { source, .. }) => { - assert!(matches!(*source, crab_cell_runtime::Error::Fenced)); - } - other => panic!("unexpected command result: {other:?}"), - } - let control = CellAuthority::new(fixture.layout.clone()) - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(control.value().root.as_ref().unwrap().commit_sequence, 0); - assert!(matches!( - runtime.shutdown().await, - Err(crab_cell_runtime::Error::Fenced) - )); -} -#[tokio::test] -async fn activation_rejects_control_owned_by_another_node_session() { - let fixture = fixture(); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .provision( - CatalogEntry::new( - &fixture.target, - CatalogRole::Repository, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .create_initial( - &proof, - IncarnationId::from_bytes([2; 16]), - Owner { - session: SessionId::from_bytes([4; 16]), - endpoint: "https://node.internal:8081".into(), - }, - ) - .await - .unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 2 * 1024 * 1024, - SessionId::from_bytes([9; 16]), - ) - .unwrap(); - assert!(matches!( - runtime - .bootstrap( - proof, - fixture.replica, - authority, - observed, - fixture.database, - |_| Ok(()), - ) - .await, - Err(crab_cell_runtime::Error::Fenced) - )); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/recovery.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/recovery.rs deleted file mode 100644 index 3f4b42e83..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/recovery.rs +++ /dev/null @@ -1,646 +0,0 @@ -//! Recovery overlays, pinned tails, and source-loss takeover. - -use super::*; - -#[tokio::test] -async fn takeover_resumes_pinned_recovery_before_serving() { - recover_retained_tail(false).await; -} -#[tokio::test] -async fn recovery_seals_already_rooted_tail_without_an_empty_manifest() { - recover_retained_tail(true).await; -} -async fn recover_retained_tail(rooted: bool) { - let fixture = fixture_for(b"recovered-takeover"); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - drop(handle); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let stale = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let predecessor = stale.value().ltx_root().unwrap(); - let leader = stale.value().owner.as_ref().unwrap().session; - - let tail_directory = tempfile::TempDir::new().unwrap(); - let tail_path = tail_directory.path().join("tail.sqlite"); - let writable = fixture - .replica - .open_root(&predecessor) - .await - .unwrap() - .paged() - .prepare_writable(&tail_path) - .await - .unwrap(); - let mut writer = writable.open_writable(&tail_path).unwrap(); - writer - .transaction(|transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - transaction.execute( - "UPDATE sys_meta SET commit_sequence = commit_sequence + 1, logical_time_ms = logical_time_ms + 1 WHERE singleton = 1", - [], - )?; - Ok(()) - }) - .unwrap(); - let capture = writer.capture().unwrap(); - let mut frames = Vec::with_capacity(capture.segments.len()); - for (index, segment) in capture.segments.iter().enumerate() { - frames.push( - crab_ltx::encode_node_frame( - crab_ltx::NodeFrameScope { - leader_session: *leader.as_bytes(), - log_epoch: 1, - node_sequence: u64::try_from(index).unwrap() + 1, - application: *fixture.layout.application_id(), - cell: *fixture.target.cell_id().as_bytes(), - incarnation: *stale.value().incarnation.as_bytes(), - cell_epoch: stale.value().epoch, - commit_sequence: predecessor.commit_sequence + 1, - }, - segment.info().clone(), - Bytes::from(std::fs::read(segment.path()).unwrap()), - Limits::default(), - ) - .unwrap() - .encoded() - .clone(), - ); - } - if rooted { - // Crash after the exact Cell root CAS but before shared node coverage. - let prepared = fixture - .replica - .prepare( - Some(&predecessor), - &capture, - predecessor.commit_sequence + 1, - stale.value().schema, - ) - .await - .unwrap(); - authority - .transition( - &stale, - stale.value().publish_prepared(&prepared, None).unwrap(), - Transition::Publish, - ) - .await - .unwrap(); - } - writer.close().unwrap(); - let follower = SessionId::from_bytes([43; 16]); - let follower_directory = tempfile::TempDir::new().unwrap(); - let follower_store = crab_cell_runtime::FollowerStore::open( - follower_directory.path().to_owned(), - Limits::default(), - crab_cell_runtime::ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let transport: Arc = Arc::new( - crab_cell_runtime::node::log_transport::LocalFollowerTransport::new( - crab_cell_runtime::identity::NodeId::from_bytes(*follower.as_bytes()), - follower_store, - ), - ); - transport - .append( - crab_cell_runtime::identity::NodeId::from_bytes(*follower.as_bytes()), - crab_cell_runtime::node::log_transport::AppendRequest { - leader_session: leader, - log_epoch: 1, - frames, - covered_through: 0, - }, - ) - .await - .unwrap(); - let manifests = crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - fixture.layout.clone(), - Limits::default(), - ); - let successor = SessionId::from_bytes([42; 16]); - let fenced = fence_log_session(&fixture.layout, leader, successor, follower, 0).await; - assert!(matches!( - fenced.direct_takeover(), - Err(crab_cell_runtime::Error::PendingPublication) - )); - let recovery = crab_cell_runtime::node::log_recovery::NodeLogRecovery::from_fenced( - Arc::clone(&transport), - &fenced, - Limits::default(), - ) - .unwrap(); - let coordinator = crab_cell_runtime::node::log_recovery::RecoveryCoordinator::new( - recovery, - manifests.clone(), - ); - let inventory = - crab_cell_runtime::node::log_recovery::recoverable_cells(&catalog, &authority, leader, 10) - .await - .unwrap(); - assert_eq!(inventory.len(), 1); - if rooted { - let directory = crab_cell_runtime::node::NodeDirectory::new( - fixture.layout.clone(), - Digest::from_bytes([90; 32]), - Digest::from_bytes([91; 32]), - Digest::from_bytes([92; 32]), - ); - let completed = coordinator - .recover_and_seal(&directory, fenced, inventory, 10_002) - .await - .unwrap(); - assert!(completed.controls.is_empty()); - assert_eq!( - completed.sealed.log().phase(), - crab_cell_runtime::node::log_state::NodeLogPhase::Sealed - ); - return; - } - let attached = coordinator - .recover(fenced.clone(), inventory) - .await - .unwrap(); - assert_eq!(attached.len(), 1); - drop(coordinator); - let resumed_recovery = crab_cell_runtime::node::log_recovery::NodeLogRecovery::from_fenced( - transport, - &fenced, - Limits::default(), - ) - .unwrap(); - let resumed = crab_cell_runtime::node::log_recovery::RecoveryCoordinator::new( - resumed_recovery, - manifests.clone(), - ); - let directory = crab_cell_runtime::node::NodeDirectory::new( - fixture.layout.clone(), - Digest::from_bytes([90; 32]), - Digest::from_bytes([91; 32]), - Digest::from_bytes([92; 32]), - ); - let completed = resumed - .recover_and_seal( - &directory, - fenced.clone(), - vec![crab_cell_runtime::node::log_recovery::RecoveryCell { - application: fixture.target.application(), - authority: authority.clone(), - observed: attached[0].clone(), - }], - 10_002, - ) - .await - .unwrap(); - assert_eq!(completed.controls[0].value(), attached[0].value()); - assert_eq!( - completed.sealed.log().phase(), - crab_cell_runtime::node::log_state::NodeLogPhase::Sealed - ); - let repeated = resumed - .recover_and_seal( - &directory, - fenced.clone(), - vec![crab_cell_runtime::node::log_recovery::RecoveryCell { - application: fixture.target.application(), - authority: authority.clone(), - observed: completed.controls[0].clone(), - }], - 10_003, - ) - .await - .unwrap(); - assert_eq!(repeated.sealed, completed.sealed); - let attached = repeated.controls.into_iter().next().unwrap(); - let takeover = repeated.takeover; - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - successor, - ) - .unwrap(); - let restored = runtime - .takeover_restored( - proof, - fixture.replica.clone(), - authority.clone(), - attached, - takeover, - manifests, - fixture._directory.path().join("recovered-takeover.sqlite"), - Owner { - session: successor, - endpoint: "https://recovered-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - - assert_eq!( - restored - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 1_i64.to_be_bytes() - ); - let serving = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(serving.value().state, ControlState::Serving); - assert!(serving.value().recovery.is_none()); - assert_eq!( - serving.value().root.as_ref().unwrap().commit_sequence, - predecessor.commit_sequence + 1 - ); - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} -#[tokio::test] -async fn unchanged_unpublished_owner_is_taken_over_then_bootstrapped() { - let fixture = fixture(); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .provision( - CatalogEntry::new( - &fixture.target, - CatalogRole::Repository, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let stale = authority - .create_initial( - &proof, - IncarnationId::from_bytes([2; 16]), - Owner { - session: SessionId::from_bytes([4; 16]), - endpoint: "https://stopped-import.internal:8081".into(), - }, - ) - .await - .unwrap(); - let fenced = fence_session( - &fixture.layout, - stale.value().owner.as_ref().unwrap().session, - SessionId::from_bytes([42; 16]), - ) - .await; - let takeover = fenced.direct_takeover().unwrap(); - let session = SessionId::from_bytes([42; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - session, - ) - .unwrap(); - let restored = runtime - .takeover_unpublished( - proof, - fixture.replica.clone(), - authority.clone(), - stale, - takeover, - fixture - ._directory - .path() - .join("takeover-unpublished.sqlite"), - Owner { - session, - endpoint: "https://import-successor.internal:8081".into(), - }, - |transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (7)", - )?; - Ok(()) - }, - ) - .await - .unwrap(); - - assert_eq!( - restored - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 7_i64.to_be_bytes() - ); - let owned = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(owned.value().epoch, 2); - assert_eq!(owned.value().owner.as_ref().unwrap().session, session); - restored.drain().await.unwrap(); -} -#[tokio::test] -async fn slow_bootstrap_renews_unpublished_ownership_before_publication() { - let fixture = fixture(); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .provision( - CatalogEntry::new( - &fixture.target, - CatalogRole::Repository, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let session = SessionId::from_bytes([43; 16]); - let observed = authority - .create_initial( - &proof, - IncarnationId::from_bytes([2; 16]), - Owner { - session, - endpoint: "https://slow-bootstrap.internal:8081".into(), - }, - ) - .await - .unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - session, - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof, - fixture.replica.clone(), - authority.clone(), - observed, - fixture._directory.path().join("slow-bootstrap.sqlite"), - |transaction| { - std::thread::sleep(std::time::Duration::from_secs(4)); - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (0)", - )?; - Ok(()) - }, - ) - .await - .unwrap(); - let published = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(published.value().state, ControlState::Serving); - assert!(published.value().revision >= 3); - handle.drain().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn source_loss_takeover_restores_exact_root_and_continues_publication() { - source_loss_takeover( - Store::new(Arc::new(InMemory::new())), - Path::from("cold-runtime"), - ) - .await; -} -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires an isolated pre-created RustFS bucket, prefix and explicit test credentials"] -async fn rustfs_source_loss_takeover_restores_exact_root_and_continues_publication() { - let required = |name| std::env::var(name).unwrap_or_else(|_| panic!("missing {name}")); - let store = build_explicit_store( - &required("CRAB_CELL_TEST_BUCKET"), - ObjectStoreCredentials::Aws { - access_key_id: required("AWS_ACCESS_KEY_ID"), - secret_access_key: required("AWS_SECRET_ACCESS_KEY"), - session_token: None, - region: "us-east-1".into(), - }, - Some(&required("CRAB_CELL_TEST_ENDPOINT")), - true, - ) - .unwrap(); - let prefix = Path::from(format!( - "{}/cold-runtime", - required("CRAB_CELL_TEST_PREFIX") - )); - source_loss_takeover(store, prefix).await; -} -async fn source_loss_takeover(store: Store, prefix: Path) { - let target = CellTarget::new( - TenantId::from_bytes([41; 16]), - ApplicationId::from_bytes([42; 16]), - NamespaceId::from_bytes([43; 16]), - b"repository-cold-start", - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([44; 16]); - let layout = CellStorageLayout::new(store, prefix, [42; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = - crab_cell_runtime::cell::catalog::CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Repository, - Digest::from_bytes([45; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - - let first_session = SessionId::from_bytes([46; 16]); - let authority = CellAuthority::new(layout.clone()); - let recovering = authority - .create_initial( - &proof, - incarnation, - Owner { - session: first_session, - endpoint: "https://node-one.internal:8081".into(), - }, - ) - .await - .unwrap(); - - let first_local = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - first_session, - ) - .unwrap(); - let first = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - recovering, - first_local.path().join("cell.sqlite"), - |transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (0)", - )?; - Ok(()) - }, - ) - .await - .unwrap(); - let first_identity = mutation_identity_window(47, 10, 10_000); - let first_digest = Digest::from_bytes([48; 32]); - let first_outcome = first - .execute( - first_identity, - first_digest, - 20, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"first".to_vec())) - }, - ) - .await - .unwrap(); - assert!(matches!( - first_outcome, - StoredOutcome::Success { - commit_sequence: 1, - .. - } - )); - first.drain().await.unwrap(); - drop(runtime); - first_local.close().unwrap(); - - let current = authority.load(cell).await.unwrap().unwrap(); - eprintln!( - "fault_seed=47 schedule=source_loss_before_owner_takeover request={first_identity:?} operation={first_digest:?} committed_sequence={} selected_follower_tickets=[] root={:?} observed_control={:?}", - first_outcome.commit_sequence(), - current.value().ltx_root(), - current.value(), - ); - let second_session = SessionId::from_bytes([49; 16]); - let mut takeover = current.value().clone(); - takeover.epoch += 1; - takeover.revision += 1; - takeover.progress += 1; - takeover.state = ControlState::Recovering; - takeover.owner = Some(Owner { - session: second_session, - endpoint: "https://node-two.internal:8081".into(), - }); - let takeover = authority - .transition(¤t, takeover, Transition::Takeover) - .await - .unwrap(); - - let second_local = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - second_session, - ) - .unwrap(); - let second = runtime - .activate_restored( - proof, - replica, - authority.clone(), - takeover, - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - layout.clone(), - Limits::default(), - ), - second_local.path().join("cell.sqlite"), - ) - .await - .unwrap(); - assert_eq!( - authority.load(cell).await.unwrap().unwrap().value().state, - ControlState::Serving - ); - assert_eq!( - second - .resolve(first_identity, first_digest, 21, 1_024) - .await - .unwrap(), - Resolution::Committed(first_outcome) - ); - assert!(matches!( - second - .execute( - mutation_identity_window(50, 10, 10_000), - Digest::from_bytes([51; 32]), - 21, - 1_024, - 1_024, - |transaction| { - let value = transaction - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(value.to_be_bytes().to_vec())) - }, - ) - .await - .unwrap(), - StoredOutcome::Success { ref result, commit_sequence: 2 } - if result == &1_i64.to_be_bytes() - )); - second.drain().await.unwrap(); - assert_eq!( - authority - .load(cell) - .await - .unwrap() - .unwrap() - .value() - .root - .as_ref() - .unwrap() - .commit_sequence, - 2 - ); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/sparse_process.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/sparse_process.rs deleted file mode 100644 index f5c680f65..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/sparse_process.rs +++ /dev/null @@ -1,297 +0,0 @@ -//! Process-kill qualification for a sparse owner after exact-root publication. - -use super::*; -use std::path::PathBuf; -use std::process::{Child, Command, Stdio}; - -const CHILD_PATH: &str = "runtime::lifecycle::ownership::sparse_process::rustfs_sparse_owner_child"; -const ACTIVE_PATH_ENV: &str = "CRAB_LTX_SPARSE_OWNER_ACTIVE"; -const READY_PATH_ENV: &str = "CRAB_LTX_SPARSE_OWNER_READY"; - -fn rustfs_store() -> Store { - let required = |name| std::env::var(name).unwrap_or_else(|_| panic!("missing {name}")); - build_explicit_store( - &required("CRAB_CELL_TEST_BUCKET"), - ObjectStoreCredentials::Aws { - access_key_id: required("AWS_ACCESS_KEY_ID"), - secret_access_key: required("AWS_SECRET_ACCESS_KEY"), - session_token: None, - region: "us-east-1".into(), - }, - Some(&required("CRAB_CELL_TEST_ENDPOINT")), - true, - ) - .unwrap() -} - -fn fixture() -> (CellTarget, CellStorageLayout, CellReplica, CellAuthority) { - let target = CellTarget::new( - TenantId::from_bytes([61; 16]), - ApplicationId::from_bytes([62; 16]), - NamespaceId::from_bytes([63; 16]), - b"sparse-process-owner", - ) - .unwrap(); - let prefix = std::env::var("CRAB_CELL_TEST_PREFIX").unwrap(); - let layout = CellStorageLayout::new( - rustfs_store(), - Path::from(format!("{prefix}/sparse-process-owner")), - [62; 16], - ); - let replica = CellReplica::new( - layout.clone(), - *target.cell_id().as_bytes(), - [64; 16], - Limits::default(), - ) - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - (target, layout, replica, authority) -} - -fn recovering_owner(control: &mut crab_cell_runtime::control::Control, session: SessionId) { - control.epoch += 1; - control.revision += 1; - control.progress += 1; - control.state = ControlState::Recovering; - control.owner = Some(Owner { - session, - endpoint: "https://sparse-owner.internal:8081".into(), - }); -} - -struct KillOnDrop(Child); - -impl Drop for KillOnDrop { - fn drop(&mut self) { - let _ = self.0.kill(); - let _ = self.0.wait(); - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn rustfs_sparse_owner_child() { - let Ok(active) = std::env::var(ACTIVE_PATH_ENV) else { - return; - }; - let ready = PathBuf::from(std::env::var(READY_PATH_ENV).unwrap()); - let active = PathBuf::from(active); - let (target, layout, replica, authority) = fixture(); - let catalog = - crab_cell_runtime::cell::catalog::CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog.lookup(target.cell_id()).await.unwrap().unwrap(); - let observed = authority.load(target.cell_id()).await.unwrap().unwrap(); - assert_eq!( - observed.value().owner.as_ref().unwrap().session, - SessionId::from_bytes([66; 16]) - ); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - SessionId::from_bytes([66; 16]), - ) - .unwrap(); - // This activation always enters through the paged root path on a fresh - // destination; its local checksum base is file-backed. - let handle = runtime - .activate_restored( - proof, - replica, - authority.clone(), - observed, - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - layout, - Limits::default(), - ), - active.clone(), - ) - .await - .unwrap(); - let mut sidecar = active.as_os_str().to_owned(); - sidecar.push(".crab-ltx-checksums"); - assert!(PathBuf::from(sidecar).is_file()); - let outcome = handle - .execute( - mutation_identity_window(67, 10, 10_000), - Digest::from_bytes([68; 32]), - 20, - 1_024, - 1_024, - |tx| { - tx.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"published".to_vec())) - }, - ) - .await - .unwrap(); - assert!(matches!( - outcome, - StoredOutcome::Success { - commit_sequence: 1, - .. - } - )); - let published = authority.load(target.cell_id()).await.unwrap().unwrap(); - assert_eq!(published.value().root.as_ref().unwrap().commit_sequence, 1); - std::fs::write(ready, b"published").unwrap(); - std::future::pending::<()>().await; -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires an isolated pre-created RustFS bucket, prefix and explicit test credentials"] -async fn rustfs_process_killed_sparse_owner_restores_published_root_and_continues() { - let (target, layout, replica, authority) = fixture(); - let catalog = - crab_cell_runtime::cell::catalog::CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision( - CatalogEntry::new( - &target, - CatalogRole::Repository, - Digest::from_bytes([65; 32]), - 1, - ) - .unwrap(), - ) - .await - .unwrap(); - let first_session = SessionId::from_bytes([69; 16]); - let initial = authority - .create_initial( - &proof, - IncarnationId::from_bytes([64; 16]), - Owner { - session: first_session, - endpoint: "https://bootstrap.internal:8081".into(), - }, - ) - .await - .unwrap(); - let bootstrap_dir = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - first_session, - ) - .unwrap(); - let bootstrap = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - initial, - bootstrap_dir.path().join("cell.sqlite"), - |tx| { - tx.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES(0)", - )?; - Ok(()) - }, - ) - .await - .unwrap(); - bootstrap.drain().await.unwrap(); - drop(runtime); - bootstrap_dir.close().unwrap(); - - let current = authority.load(target.cell_id()).await.unwrap().unwrap(); - let mut child_control = current.value().clone(); - recovering_owner(&mut child_control, SessionId::from_bytes([66; 16])); - authority - .transition(¤t, child_control, Transition::Takeover) - .await - .unwrap(); - - let child_dir = tempfile::TempDir::new().unwrap(); - let ready = child_dir.path().join("ready"); - let active = child_dir.path().join("active.sqlite"); - let child = Command::new(std::env::current_exe().unwrap()) - .args(["--exact", CHILD_PATH, "--nocapture"]) - .env(ACTIVE_PATH_ENV, &active) - .env(READY_PATH_ENV, &ready) - .stdout(Stdio::inherit()) - .stderr(Stdio::inherit()) - .spawn() - .unwrap(); - let mut child = KillOnDrop(child); - let deadline = std::time::Instant::now() + std::time::Duration::from_secs(45); - while !ready.exists() { - assert!( - child.0.try_wait().unwrap().is_none(), - "sparse owner exited before publication" - ); - assert!( - std::time::Instant::now() < deadline, - "sparse owner publication timed out" - ); - tokio::time::sleep(std::time::Duration::from_millis(50)).await; - } - child.0.kill().unwrap(); - assert!(!child.0.wait().unwrap().success()); - child_dir.close().unwrap(); - - let current = authority.load(target.cell_id()).await.unwrap().unwrap(); - assert_eq!(current.value().root.as_ref().unwrap().commit_sequence, 1); - let successor_session = SessionId::from_bytes([70; 16]); - let mut successor_control = current.value().clone(); - recovering_owner(&mut successor_control, successor_session); - let takeover = authority - .transition(¤t, successor_control, Transition::Takeover) - .await - .unwrap(); - let successor_dir = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - successor_session, - ) - .unwrap(); - let successor = runtime - .activate_restored( - proof, - replica, - authority.clone(), - takeover, - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - layout, - Limits::default(), - ), - successor_dir.path().join("cell.sqlite"), - ) - .await - .unwrap(); - let outcome = successor - .execute( - mutation_identity_window(71, 10, 10_000), - Digest::from_bytes([72; 32]), - 21, - 1_024, - 1_024, - |tx| { - let value = - tx.query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - tx.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(value.to_be_bytes().to_vec())) - }, - ) - .await - .unwrap(); - assert!(matches!( - outcome, - StoredOutcome::Success { ref result, commit_sequence: 2 } if result == &1_i64.to_be_bytes() - )); - successor.drain().await.unwrap(); - assert_eq!( - authority - .load(target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .root - .as_ref() - .unwrap() - .commit_sequence, - 2 - ); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/succession.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/succession.rs deleted file mode 100644 index 68efad6e6..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/succession.rs +++ /dev/null @@ -1,361 +0,0 @@ -//! Independent-process successor election and acquire races. -//! -//! Every test here needs the `test-support` filesystem CAS store, so the whole -//! module is gated on that feature. - -#![cfg(feature = "test-support")] - -use super::*; - -#[cfg(feature = "test-support")] -#[tokio::test(flavor = "multi_thread")] -async fn independent_processes_allow_one_idle_cell_winner() { - let object_root = tempfile::TempDir::new().unwrap(); - let fixture = filesystem_fixture(b"process-movement-race", object_root.path()); - let session = SessionId::from_bytes([83; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(idle.value().state, ControlState::Idle); - let root = idle.value().ltx_root().unwrap(); - let binary = std::env::var("CARGO_BIN_EXE_cell_movement_probe") - .expect("Cargo must provide the movement probe binary to integration tests"); - let store_root = object_root.path().to_owned(); - let partition = "process-movement-race"; - let first_destination = fixture._directory.path().join("process-first.sqlite"); - let second_destination = fixture._directory.path().join("process-second.sqlite"); - let first_store_root = store_root.clone(); - let first_binary = binary.clone(); - let first = tokio::task::spawn_blocking(move || { - std::process::Command::new(first_binary) - .stderr(std::process::Stdio::null()) - .args([ - first_store_root.as_os_str().to_string_lossy().as_ref(), - partition, - "53535353535353535353535353535353", - first_destination.to_string_lossy().as_ref(), - "750", - "drain", - ]) - .status() - .unwrap() - }); - let second_store_root = store_root; - let second_binary = binary; - let second = tokio::task::spawn_blocking(move || { - std::thread::sleep(std::time::Duration::from_millis(25)); - std::process::Command::new(second_binary) - .stderr(std::process::Stdio::null()) - .args([ - second_store_root.as_os_str().to_string_lossy().as_ref(), - partition, - "54545454545454545454545454545454", - second_destination.to_string_lossy().as_ref(), - "100", - "drain", - ]) - .status() - .unwrap() - }); - let (first, second) = tokio::join!(first, second); - let first = first.unwrap(); - let second = second.unwrap(); - assert_ne!(first.success(), second.success()); - - let final_control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(final_control.value().state, ControlState::Idle); - assert_eq!(final_control.value().ltx_root(), Some(root)); -} -#[cfg(feature = "test-support")] -#[tokio::test(flavor = "multi_thread")] -async fn independent_process_receiver_failure_returns_exact_idle_root() { - let object_root = tempfile::TempDir::new().unwrap(); - let fixture = filesystem_fixture(b"process-movement-receiver-failure", object_root.path()); - let session = SessionId::from_bytes([89; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let root = idle.value().ltx_root(); - let binary = std::env::var("CARGO_BIN_EXE_cell_movement_probe") - .expect("Cargo must provide the movement probe binary to integration tests"); - let destination = fixture - ._directory - .path() - .join("process-receiver-parent-does-not-exist") - .join("process-receiver.sqlite"); - std::fs::create_dir_all(&destination).unwrap(); - let status = tokio::task::spawn_blocking({ - let store_root = object_root.path().to_owned(); - move || { - std::process::Command::new(binary) - .args([ - store_root.as_os_str().to_string_lossy().as_ref(), - "process-movement-receiver-failure", - "59595959595959595959595959595959", - destination.to_string_lossy().as_ref(), - "0", - "fail-receiver", - ]) - .output() - .unwrap() - } - }) - .await - .unwrap(); - assert!( - status.status.success(), - "receiver-failure probe failed: {}", - String::from_utf8_lossy(&status.stderr) - ); - - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(current.value().state, ControlState::Idle); - assert!(current.value().owner.is_none()); - assert_eq!(current.value().ltx_root(), root); -} -#[cfg(feature = "test-support")] -#[tokio::test(flavor = "multi_thread")] -async fn crashed_process_is_fenced_before_successor_restore() { - let object_root = tempfile::TempDir::new().unwrap(); - let fixture = filesystem_fixture(b"process-movement-crash", object_root.path()); - let session = SessionId::from_bytes([85; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - let request = mutation_identity_window(139, 10, 10_000); - let digest = Digest::from_bytes([140; 32]); - let outcome = handle - .execute(request, digest, 20, 1_024, 1_024, |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"before-crash".to_vec())) - }) - .await - .unwrap(); - assert_eq!(outcome.commit_sequence(), 1); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - - let binary = std::env::var("CARGO_BIN_EXE_cell_movement_probe") - .expect("Cargo must provide the movement probe binary to integration tests"); - let destination = fixture._directory.path().join("process-crashed.sqlite"); - let status = tokio::task::spawn_blocking({ - let store_root = object_root.path().to_owned(); - move || { - std::process::Command::new(binary) - .stderr(std::process::Stdio::null()) - .args([ - store_root.as_os_str().to_string_lossy().as_ref(), - "process-movement-crash", - "55555555555555555555555555555555", - destination.to_string_lossy().as_ref(), - "0", - "crash", - ]) - .status() - .unwrap() - } - }) - .await - .unwrap(); - assert!(status.success()); - - let authority = CellAuthority::new(fixture.layout.clone()); - let stale = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - eprintln!( - "fault_seed=139 schedule=owner_process_crash_before_successor_restore request={request:?} operation={digest:?} committed_sequence={} selected_follower_tickets=[] root={:?} observed_control={:?}", - outcome.commit_sequence(), - stale.value().ltx_root(), - stale.value(), - ); - assert_eq!( - stale.value().owner.as_ref().map(|owner| owner.session), - Some(SessionId::from_bytes([85; 16])) - ); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let successor = SessionId::from_bytes([86; 16]); - let fenced = fence_session(&fixture.layout, SessionId::from_bytes([85; 16]), successor).await; - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 8 * 1024 * 1024, - successor, - ) - .unwrap(); - let restored = runtime - .takeover_restored( - proof, - fixture.replica.clone(), - authority.clone(), - stale, - fenced.direct_takeover().unwrap(), - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - fixture.layout.clone(), - Limits::default(), - ), - fixture._directory.path().join("process-successor.sqlite"), - Owner { - session: successor, - endpoint: "https://process-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!( - restored - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 1_i64.to_be_bytes() - ); - assert_eq!( - restored.resolve(request, digest, 21, 1_024).await.unwrap(), - Resolution::Committed(outcome) - ); - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} -#[cfg(feature = "test-support")] -#[tokio::test(flavor = "multi_thread")] -async fn lost_release_response_is_reconciled_before_successor_acquire() { - let object_root = tempfile::TempDir::new().unwrap(); - let fixture = filesystem_fixture(b"process-movement-lost-release", object_root.path()); - let session = SessionId::from_bytes([87; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - - let binary = std::env::var("CARGO_BIN_EXE_cell_movement_probe") - .expect("Cargo must provide the movement probe binary to integration tests"); - let status = tokio::task::spawn_blocking({ - let store_root = object_root.path().to_owned(); - let destination = fixture - ._directory - .path() - .join("process-lost-release.sqlite"); - move || { - std::process::Command::new(binary) - .stderr(std::process::Stdio::null()) - .args([ - store_root.as_os_str().to_string_lossy().as_ref(), - "process-movement-lost-release", - "57575757575757575757575757575757", - destination.to_string_lossy().as_ref(), - "0", - "lost-release", - ]) - .status() - .unwrap() - } - }) - .await - .unwrap(); - assert!(status.success()); - - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(idle.value().state, ControlState::Idle); - let root = idle.value().ltx_root().unwrap(); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let successor = SessionId::from_bytes([88; 16]); - let successor_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 8 * 1024 * 1024, - successor, - ) - .unwrap(); - let restored = successor_runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority.clone(), - idle, - fixture - ._directory - .path() - .join("process-lost-release-successor.sqlite"), - Owner { - session: successor, - endpoint: "https://process-lost-release-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!( - restored - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 0_i64.to_be_bytes() - ); - restored.drain().await.unwrap(); - successor_runtime.shutdown().await.unwrap(); - assert_eq!( - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root(), - Some(root) - ); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/takeover.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/takeover.rs deleted file mode 100644 index e6887b1b5..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/ownership/takeover.rs +++ /dev/null @@ -1,420 +0,0 @@ -//! Fenced takeover of idle, dead, and unpublished owners. - -use super::*; -use object_store::ObjectStoreExt; - -#[tokio::test(flavor = "multi_thread")] -async fn released_cell_is_acquired_by_one_successor_runtime() { - let fixture = fixture_for(b"successor-runtime-movement"); - let first_session = SessionId::from_bytes([81; 16]); - let second_session = SessionId::from_bytes([82; 16]); - let first_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 8 * 1024 * 1024, - first_session, - ) - .unwrap(); - let handle = bootstrap_on(&first_runtime, &fixture, first_session).await; - handle.drain().await.unwrap(); - assert_eq!(first_runtime.stats().active_cells(), 0); - - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(idle.value().state, ControlState::Idle); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let second_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 8 * 1024 * 1024, - second_session, - ) - .unwrap(); - let successor = second_runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority.clone(), - idle, - fixture._directory.path().join("successor.sqlite"), - Owner { - session: second_session, - endpoint: "https://successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - - assert_eq!(first_runtime.stats().active_cells(), 0); - assert_eq!(second_runtime.stats().active_cells(), 1); - assert_eq!( - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .owner - .as_ref() - .map(|owner| owner.session), - Some(second_session) - ); - - successor.drain().await.unwrap(); - assert_eq!(second_runtime.stats().active_cells(), 0); - second_runtime.shutdown().await.unwrap(); - first_runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn missing_authoritative_root_cannot_become_a_serving_cell() { - unavailable_authoritative_root_cannot_serve(false).await; -} - -#[tokio::test] -async fn torn_authoritative_root_cannot_become_a_serving_cell() { - unavailable_authoritative_root_cannot_serve(true).await; -} - -async fn unavailable_authoritative_root_cannot_serve(corrupt: bool) { - let backend = Arc::new(InMemory::new()); - let fixture = fixture_with_limits_and_store( - b"unavailable-authoritative-root", - Limits::default(), - Store::new(backend.clone()), - ); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - handle.drain().await.unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(idle.value().state, ControlState::Idle); - let root = idle.value().ltx_root().unwrap(); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let root_path = fixture.layout.incarnation_object_path( - fixture.target.cell_id().as_bytes(), - idle.value().incarnation.as_bytes(), - &root.digest, - CellObjectKind::Root, - ); - if corrupt { - backend - .put(&root_path, Bytes::from_static(b"torn-root").into()) - .await - .unwrap(); - } else { - backend.delete(&root_path).await.unwrap(); - } - - let session = SessionId::from_bytes([138; 16]); - let receiver = tempfile::TempDir::new().unwrap(); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let error = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority.clone(), - idle, - receiver.path().join("missing-root-restored.sqlite"), - Owner { - session, - endpoint: "https://missing-root.internal:8081".into(), - }, - ) - .await - .err() - .expect("unavailable root must not activate"); - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - eprintln!( - "fault_seed={} schedule={} request=none committed_sequence={} selected_follower_tickets=[] root={root:?} observed_control={:?} error={error:?}", - if corrupt { 139 } else { 138 }, - if corrupt { - "torn_authoritative_root" - } else { - "missing_authoritative_root" - }, - root.commit_sequence, - current.value(), - ); - assert_eq!(current.value().state, ControlState::Idle); - assert!(current.value().owner.is_none()); - assert_eq!(current.value().ltx_root(), Some(root)); - assert_eq!(runtime.stats().active_cells(), 0); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn observed_takeover_fences_the_old_cell_before_more_work() { - let fixture = fixture(); - let (runtime, handle, _pool) = activate_runtime(&fixture, 16 * 1024 * 1024).await; - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let successor = observed - .value() - .takeover(Owner { - session: SessionId::from_bytes([99; 16]), - endpoint: "https://successor.internal:8081".into(), - }) - .unwrap(); - let successor = authority - .transition(&observed, successor, Transition::Takeover) - .await - .unwrap(); - - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - match handle.query(1, 1, |_| Ok(Vec::new())).await { - Err(crab_cell_runtime::Error::Fenced) => break, - Ok(_) => tokio::time::sleep(std::time::Duration::from_millis(25)).await, - Err(error) => panic!("unexpected query outcome while awaiting fence: {error}"), - } - } - }) - .await - .unwrap(); - runtime.shutdown().await.unwrap(); - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(current.value(), successor.value()); -} -#[tokio::test] -async fn idle_control_is_acquired_before_exact_root_restore() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - handle.drain().await.unwrap(); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let session = SessionId::from_bytes([40; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - session, - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority.clone(), - idle, - fixture._directory.path().join("idle-acquire.sqlite"), - Owner { - session, - endpoint: "https://idle-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - - assert_eq!( - restored - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 0_i64.to_be_bytes() - ); - let owned = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(owned.value().state, ControlState::Serving); - assert_eq!(owned.value().epoch, 2); - assert_eq!(owned.value().owner.as_ref().unwrap().session, session); - restored.drain().await.unwrap(); -} -#[tokio::test] -async fn unchanged_dead_owner_is_taken_over_then_restored() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - drop(handle); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let stale = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let fenced = fence_session( - &fixture.layout, - stale.value().owner.as_ref().unwrap().session, - SessionId::from_bytes([41; 16]), - ) - .await; - let takeover = fenced.direct_takeover().unwrap(); - let session = SessionId::from_bytes([41; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 10).unwrap(), - 16 * 1024 * 1024, - session, - ) - .unwrap(); - let restored = runtime - .takeover_restored( - proof, - fixture.replica.clone(), - authority.clone(), - stale, - takeover, - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - fixture.layout.clone(), - Limits::default(), - ), - fixture._directory.path().join("takeover.sqlite"), - Owner { - session, - endpoint: "https://takeover-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - - assert_eq!( - restored - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 0_i64.to_be_bytes() - ); - let owned = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(owned.value().state, ControlState::Serving); - assert_eq!(owned.value().epoch, 2); - assert_eq!(owned.value().owner.as_ref().unwrap().session, session); - restored.drain().await.unwrap(); -} -#[tokio::test] -async fn failed_takeover_receiver_does_not_leave_authority_owned() { - let fixture = fixture_for(b"takeover-receiver-activation-failure"); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - drop(handle); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let stale = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let expected_root = stale.value().root.clone(); - let previous = stale.value().owner.as_ref().unwrap().session; - let successor = SessionId::from_bytes([44; 16]); - let takeover = fence_session(&fixture.layout, previous, successor).await; - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 16 * 1024 * 1024, - successor, - ) - .unwrap(); - let missing_parent = fixture - ._directory - .path() - .join("takeover-receiver-parent-does-not-exist") - .join("takeover-receiver.sqlite"); - assert!( - runtime - .takeover_restored( - proof, - fixture.replica.clone(), - authority.clone(), - stale, - takeover.direct_takeover().unwrap(), - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - fixture.layout.clone(), - Limits::default(), - ), - missing_parent, - Owner { - session: successor, - endpoint: "https://takeover-receiver-failure.internal:8081".into(), - }, - ) - .await - .is_err() - ); - - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(current.value().state, ControlState::Idle); - assert!(current.value().owner.is_none()); - assert_eq!(current.value().root, expected_root); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/read_replica.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/read_replica.rs deleted file mode 100644 index c4e6e89fe..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/read_replica.rs +++ /dev/null @@ -1,792 +0,0 @@ -//! Read-only Cell snapshots and their authority response gate. - -use super::*; -use std::{ - collections::BTreeMap, - sync::{OnceLock, atomic::AtomicU64}, -}; -use tokio::sync::oneshot; - -use crab_cell_runtime::cell::actor::CellHandle; -use crab_cell_runtime::client::CellReadReplica; -use crab_cell_runtime::client::{ - CellClient, CellDescription, InvocationError, ReadPolicy, ReplicaReadRouter, -}; -use crab_cell_runtime::node::{NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain}; -use crab_cell_runtime::peer::{ - PeerAuthorizer, PeerCellResolver, PeerDispatcher, PeerPrincipal, PeerReplicaResolver, - PeerRoundTrip, PeerSigner, PeerVerifier, ReplicaPeerClient, VerifiedPeerRequest, -}; -use crab_cell_runtime::primitives::sql::{SqlBatch, SqlStatement, SqlValue}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, MigrationDescriptor, ModuleDescriptor, NamespaceDescriptor, - OperationDescriptor, Query, QueryContext, Registry, RegistryBuilder, RetainedCodeDescriptor, -}; - -mod lifecycle; - -const MODULE: &str = "replica-counter"; -const SCHEMA: &str = "CREATE TABLE counter(value INTEGER NOT NULL)"; -const MIGRATED_SCHEMA: &str = - "ALTER TABLE counter ADD COLUMN label TEXT; UPDATE counter SET value = value + 10"; -const CODE: Digest = Digest::from_bytes([5; 32]); -const NAMESPACE: NamespaceId = NamespaceId::from_bytes([6; 16]); -static PAUSED_QUERIES: Mutex> = Mutex::new(BTreeMap::new()); - -struct PausedQuery { - entered: Arc, - released: oneshot::Receiver<()>, -} - -struct QueryPause { - id: u64, - entered: Arc, - release: Option>, -} - -impl QueryPause { - fn new() -> Self { - static NEXT_ID: AtomicU64 = AtomicU64::new(1); - let id = NEXT_ID.fetch_add(1, Ordering::Relaxed); - let entered = Arc::new(Notify::new()); - let (release, released) = oneshot::channel(); - // Typed handlers are static functions. A unique input token connects - // each callback to its fixture without letting concurrent cases rendezvous. - let previous = PAUSED_QUERIES.lock().unwrap().insert( - id, - PausedQuery { - entered: entered.clone(), - released, - }, - ); - assert!(previous.is_none()); - Self { - id, - entered, - release: Some(release), - } - } - - async fn entered(&self) { - tokio::time::timeout(std::time::Duration::from_secs(5), self.entered.notified()) - .await - .unwrap(); - } - - fn release(mut self) { - self.release.take().unwrap().send(()).unwrap(); - } -} - -impl Drop for QueryPause { - fn drop(&mut self) { - // A failed assertion must release blocked SQL, or discard a gate whose - // query never started, before the test runtime waits for its workers. - PAUSED_QUERIES.lock().unwrap().remove(&self.id); - } -} - -#[derive(Default)] -struct ControlReads(AtomicUsize); - -impl crab_cell_runtime::fleet::telemetry::CellTelemetry for ControlReads { - fn control_read(&self, _: std::time::Duration, _: bool) { - self.0.fetch_add(1, Ordering::Relaxed); - } -} - -struct ReplicaResolver(CellReadReplica); - -impl PeerReplicaResolver for ReplicaResolver { - fn resolve( - &self, - _target: CellTarget, - ) -> std::pin::Pin< - Box< - dyn std::future::Future> - + Send - + 'static, - >, - > { - let reader = self.0.clone(); - Box::pin(async move { Ok(reader) }) - } -} - -struct NoOwner; - -impl PeerCellResolver for NoOwner { - fn resolve( - &self, - _target: CellTarget, - ) -> std::pin::Pin< - Box< - dyn std::future::Future> - + Send - + 'static, - >, - > { - Box::pin(async { Err(crab_cell_runtime::Error::CellNotActive) }) - } -} - -struct ReadAuthorizer; - -impl PeerAuthorizer for ReadAuthorizer { - fn authorize(&self, request: &VerifiedPeerRequest) -> crab_cell_runtime::Result<()> { - if request.permits("repository.read") { - Ok(()) - } else { - Err(crab_cell_runtime::Error::PeerAuthorization("read denied")) - } - } -} - -struct LoopbackReplica { - verifier: Arc, - dispatcher: Arc, -} - -impl PeerRoundTrip for LoopbackReplica { - fn send( - &self, - _target: CellTarget, - _request: Vec, - _remaining_ms: u32, - ) -> std::pin::Pin< - Box>> + Send + 'static>, - > { - Box::pin(async { Err(crab_cell_runtime::Error::Peer("owner route unavailable")) }) - } - - fn send_to_node( - &self, - _target: CellTarget, - _node: NodeAdvertisement, - request: Vec, - _remaining_ms: u32, - ) -> std::pin::Pin< - Box>> + Send + 'static>, - > { - let verifier = Arc::clone(&self.verifier); - let dispatcher = Arc::clone(&self.dispatcher); - Box::pin(async move { - let now = now_ms(); - let verified = verifier.verify(&request, now)?; - dispatcher.dispatch_bytes(&verified, now).await - }) - } -} - -struct CounterModule; - -impl CellModule for CounterModule { - const NAME: &'static str = MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - static DESCRIPTOR: OnceLock = OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: MODULE, - source_digest: Digest::from_bytes([7; 32]), - retained_codes: &[RetainedCodeDescriptor { - code: CODE, - schema_min: 1, - schema_max: 2, - }], - schema_min: 1, - schema_max: 2, - migrations: Box::leak(Box::new([ - MigrationDescriptor { - version: 1, - sql: SCHEMA, - digest: Digest::from_bytes(*blake3::hash(SCHEMA.as_bytes()).as_bytes()), - }, - MigrationDescriptor { - version: 2, - sql: MIGRATED_SCHEMA, - digest: Digest::from_bytes( - *blake3::hash(MIGRATED_SCHEMA.as_bytes()).as_bytes(), - ), - }, - ])), - commands: &[], - queries: &[ - OperationDescriptor { - id: 1, - codec_version: 1, - schema_min: 1, - schema_max: 2, - input_limit: 16, - output_limit: 16, - }, - OperationDescriptor { - id: 2, - codec_version: 1, - schema_min: 1, - schema_max: 2, - input_limit: 1, - output_limit: 8, - }, - ], - workflow_definitions: &[], - activity_types: &[], - namespaces: &[NamespaceDescriptor { - id: NAMESPACE, - name: MODULE, - role: CatalogRole::Repository, - shards: 1, - effect_targets: &[], - dead_letter: None, - }], - }) - } - - fn register(self, registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - registry.bind_query::()?; - registry.bind_query::() - } -} - -struct ReadCounter; - -impl Query for ReadCounter { - const MODULE: &'static str = MODULE; - const ID: u32 = 1; - const CODEC_VERSION: u32 = 1; - type Input = u64; - type Output = u64; - - fn execute( - context: &mut QueryContext<'_>, - input: Self::Input, - ) -> crab_cell_runtime::Result { - if input != 0 { - let paused = PAUSED_QUERIES.lock().unwrap().remove(&input).unwrap(); - paused.entered.notify_one(); - paused - .released - .blocking_recv() - .map_err(|_| crab_cell_runtime::Error::Command("query pause was dropped"))?; - } - let result = context.sql(&SqlBatch { - statements: vec![SqlStatement { - sql: "SELECT value FROM counter".into(), - parameters: vec![], - }], - })?; - match result.first().and_then(|set| set.rows.first()) { - Some(row) if matches!(row.first(), Some(SqlValue::Integer(_))) => { - let Some(SqlValue::Integer(value)) = row.first() else { - return Err(crab_cell_runtime::Error::Command("counter row is invalid")); - }; - u64::try_from(*value) - .map_err(|_| crab_cell_runtime::Error::Command("counter is negative")) - } - _ => Err(crab_cell_runtime::Error::Command("counter row is missing")), - } - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn replica_reads_exact_snapshot_and_fences_after_release() { - let fixture = fixture_for(b"read-replica"); - exercise_replica_read(&fixture).await; -} - -#[tokio::test(flavor = "multi_thread")] -async fn concurrent_replica_lifecycles_keep_snapshot_and_drain_gates_separate() { - let first = fixture_for(b"concurrent-read-replica"); - let second = fixture_for(b"concurrent-read-replica"); - tokio::join!( - exercise_replica_read(&first), - exercise_replica_read(&second) - ); -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires an isolated pre-created RustFS bucket, prefix and explicit test credentials"] -async fn rustfs_replica_reads_exact_root_and_policy_cas() { - let required = |name| std::env::var(name).unwrap_or_else(|_| panic!("missing {name}")); - let store = build_explicit_store( - &required("CRAB_CELL_TEST_BUCKET"), - ObjectStoreCredentials::Aws { - access_key_id: required("AWS_ACCESS_KEY_ID"), - secret_access_key: required("AWS_SECRET_ACCESS_KEY"), - session_token: None, - region: "us-east-1".into(), - }, - Some(&required("CRAB_CELL_TEST_ENDPOINT")), - true, - ) - .unwrap(); - let prefix = Path::from(required("CRAB_CELL_TEST_PREFIX")); - let fixture = fixture_with_limits_and_store_at_prefix( - b"read-replica", - Limits::default(), - store.clone(), - prefix.clone(), - ); - exercise_replica_read(&fixture).await; - let policy = crab_cell_runtime::read_policy::ReadPolicyStore::new(fixture.layout.clone()); - let first = policy - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!( - policy - .update(&first, 2) - .await - .unwrap() - .value() - .desired_readers(), - 2 - ); - lifecycle::exercise_schema_change(fixture_with_limits_and_store_at_prefix( - b"read-replica-migration", - Limits::default(), - store, - prefix, - )) - .await; -} - -fn compiled_reader_registry() -> Arc { - let mut builder = RegistryBuilder::new(BuildDescriptor { - source_revision: "replica-test".into(), - cargo_lock_digest: Digest::from_bytes([8; 32]), - }); - builder.register(CounterModule).unwrap(); - Arc::new(builder.finish().unwrap()) -} - -async fn owner_directory( - fixture: &Fixture, - session: SessionId, - registry: &Registry, -) -> NodeDirectory { - let fleet = Digest::from_bytes([9; 32]); - let image = Digest::from_bytes([10; 32]); - let release = registry.release_digest(); - let directory = NodeDirectory::new(fixture.layout.clone(), fleet, image, release); - let now = now_ms(); - directory - .create( - NodeAdvertisement::sign( - NodeId::from_bytes([11; 16]), - session, - "https://node.internal:8081".into(), - fleet, - Digest::from_bytes([12; 32]), - image, - release, - &ed25519_dalek::SigningKey::from_bytes(&[13; 32]), - 1, - now, - now + 15_000, - registry.module_digests(), - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 1 << 20, - free_disk_bytes: 1 << 20, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .unwrap(), - now, - ) - .await - .unwrap(); - - directory -} - -async fn exercise_replica_read(fixture: &Fixture) { - let session = SessionId::from_bytes([4; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 << 20, session).unwrap(); - // Hold one old-view query while a second admitted SQL job opens its replacement. - let reader_runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(2, 2) - .unwrap() - .with_native_memory_limit(32 << 20) - .unwrap(), - 8 << 20, - SessionId::from_bytes([14; 16]), - crab_ltx::Host::default().with_local_disk_budget(crab_ltx::DiskBudget::new(8 << 20)), - ) - .unwrap(); - let handle = bootstrap_on(&runtime, fixture, session).await; - let active = runtime.active_catalog_entries().await.unwrap(); - assert_eq!(active.len(), 1); - assert_eq!(active[0].cell(), fixture.target.cell_id()); - - let registry = compiled_reader_registry(); - let fleet = Digest::from_bytes([9; 32]); - let image = Digest::from_bytes([10; 32]); - let release = registry.release_digest(); - let directory = owner_directory(fixture, session, ®istry).await; - - let reader_path = fixture._directory.path().join("reader.sqlite"); - let constrained = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 8 << 20, - SessionId::from_bytes([15; 16]), - ) - .unwrap(); - assert!(matches!( - CellReadReplica::open( - constrained.clone(), - Arc::clone(®istry), - CellAuthority::new(fixture.layout.clone()), - directory.clone(), - fixture.replica.clone(), - fixture.target.clone(), - &reader_path, - ) - .await, - Err(crab_cell_runtime::Error::Capacity(_)) - )); - assert!(!reader_path.exists()); - constrained.shutdown().await.unwrap(); - let reader = CellReadReplica::open( - reader_runtime.clone(), - Arc::clone(®istry), - CellAuthority::new(fixture.layout.clone()), - directory.clone(), - fixture.replica.clone(), - fixture.target.clone(), - &reader_path, - ) - .await - .unwrap(); - assert_eq!(reader_runtime.stats().file_descriptors(), 4); - assert_eq!(reader_runtime.stats().resident_bytes(), 12 << 20); - let disk_bytes = reader_runtime.stats().local_disk_reserved_bytes(); - assert_eq!(disk_bytes, 0); - assert_eq!( - reader.query::(None, 0).await.unwrap().output, - 0 - ); - let authority = CellAuthority::new(fixture.layout.clone()); - let control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let expected = CellDescription { - cell: fixture.target.cell_id(), - incarnation: control.value().incarnation, - code: control.value().code, - schema: control.value().schema, - }; - let peer_session = SessionId::from_bytes([21; 16]); - let peer_key = ed25519_dalek::SigningKey::from_bytes(&[22; 32]); - let dispatcher = PeerDispatcher::new( - Arc::clone(®istry), - Arc::new(NoOwner), - Arc::new(ReadAuthorizer), - ) - .with_replica_resolver(Arc::new(ReplicaResolver(reader.clone()))); - let transport = LoopbackReplica { - verifier: Arc::new(PeerVerifier::new( - peer_session, - release, - peer_key.verifying_key(), - )), - dispatcher: Arc::new(dispatcher), - }; - let peer_client = ReplicaPeerClient::new( - Arc::clone(®istry), - Arc::new(PeerSigner::new(peer_session, release, peer_key)), - PeerPrincipal { - issuer: "test".into(), - subject: "reader".into(), - actions: vec!["repository.read".into()], - }, - Arc::new(transport), - ); - let reader_node = NodeAdvertisement::sign( - NodeId::from_bytes([23; 16]), - SessionId::from_bytes([14; 16]), - "https://reader.internal:8081".into(), - fleet, - Digest::from_bytes([24; 32]), - image, - release, - &ed25519_dalek::SigningKey::from_bytes(&[25; 32]), - 1, - now_ms(), - now_ms() + 15_000, - vec![CODE], - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 32 << 20, - free_disk_bytes: 1 << 20, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .unwrap(); - assert_eq!( - peer_client - .query::(&fixture.target, reader_node.clone(), expected, None, 0) - .await - .unwrap() - .output, - 0 - ); - // Peer fencing is encoded as unavailable and decoded as CellNotActive. - // A mismatched incarnation must be rejected before executing the query. - assert!(matches!( - peer_client - .query::( - &fixture.target, - reader_node.clone(), - CellDescription { - incarnation: IncarnationId::from_bytes([99; 16]), - ..expected - }, - None, - 0, - ) - .await, - Err(crab_cell_runtime::Error::CellNotActive) - )); - directory - .create(reader_node.clone(), now_ms()) - .await - .unwrap(); - let policy = crab_cell_runtime::read_policy::ReadPolicyStore::new(fixture.layout.clone()); - let selected_policy = policy - .create(fixture.target.cell_id(), expected.incarnation, 1) - .await - .unwrap(); - let control_reads = Arc::new(ControlReads::default()); - let routing_authority = CellAuthority::with_telemetry( - fixture.layout.clone(), - crab_cell_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(control_reads.clone()), - ); - let router = ReplicaReadRouter::new(routing_authority, directory.clone()); - let (_, selected) = router.selected(&fixture.target).await.unwrap(); - assert_eq!(selected.len(), 1); - assert_eq!(control_reads.0.load(Ordering::Relaxed), 1); - let owner_client = CellClient::local(Arc::clone(®istry), handle.clone()); - let configured = owner_client - .with_read_replicas(router.clone(), peer_client.clone(), None) - .unwrap(); - let replica_client = configured.with_read_policy(ReadPolicy::Replica); - assert_eq!( - replica_client - .query::(&fixture.target, None, 0) - .await - .unwrap() - .output, - 0 - ); - let local_client = owner_client - .with_read_replicas( - router, - peer_client.clone(), - Some(( - reader_node.session(), - Arc::new(ReplicaResolver(reader.clone())), - )), - ) - .unwrap() - .with_read_policy(ReadPolicy::Replica); - assert_eq!( - local_client - .query::(&fixture.target, None, 0) - .await - .unwrap() - .output, - 0 - ); - let committed = handle - .execute( - crate::support::fixtures::mutation_identity(91), - Digest::from_bytes([92; 32]), - now_ms(), - 64, - 64, - |transaction| { - transaction.execute("UPDATE counter SET value = 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - assert!(committed.commit_sequence() > reader.receipt().await.commit_sequence); - assert_eq!( - reader.query::(None, 0).await.unwrap().output, - 0 - ); - let minimum = crab_cell_runtime::Receipt { - commit_sequence: committed.commit_sequence(), - ..reader.receipt().await - }; - assert!(matches!( - reader.query::(Some(minimum), 0).await, - Err(crab_cell_runtime::Error::ReplicaBehind { observed_sequence, minimum_sequence }) - if observed_sequence < minimum_sequence && minimum_sequence == minimum.commit_sequence - )); - assert!(matches!( - peer_client.query::(&fixture.target, reader_node, expected, Some(minimum), 0).await, - Err(crab_cell_runtime::Error::ReplicaBehind { observed_sequence, minimum_sequence }) - if observed_sequence < minimum_sequence && minimum_sequence == minimum.commit_sequence - )); - assert_eq!( - configured - .query::(&fixture.target, None, 0) - .await - .unwrap() - .output, - 1 - ); - assert_eq!( - replica_client - .query::(&fixture.target, None, 0) - .await - .unwrap() - .output, - 0 - ); - assert!(matches!( - replica_client - .query::(&fixture.target, Some(minimum), 0) - .await, - Err(InvocationError::NotStarted( - crab_cell_runtime::Error::ReplicaBehind { .. } - )) - )); - let withdrawn_policy = policy.update(&selected_policy, 0).await.unwrap(); - assert!(matches!( - replica_client - .query::(&fixture.target, None, 0) - .await, - Err(InvocationError::NotStarted( - crab_cell_runtime::Error::ReplicaUnavailable - )) - )); - policy.update(&withdrawn_policy, 1).await.unwrap(); - // Owner-ordered streams keep their watermark contract even on a capability - // whose ordinary queries explicitly select lagging snapshots. - let mut stream = replica_client - .open_state_stream::( - &fixture.target, - std::time::Instant::now() + std::time::Duration::from_secs(5), - ) - .await - .unwrap(); - assert_eq!(stream.emit(0).await.unwrap().output, 1); - drop(stream); - drop(peer_client); - - let refreshed_path = fixture._directory.path().join("refreshed.sqlite"); - std::fs::write(&refreshed_path, b"occupied").unwrap(); - assert!(reader.refresh(&refreshed_path).await.is_err()); - assert_eq!( - reader.query::(None, 0).await.unwrap().output, - 0 - ); - std::fs::remove_file(&refreshed_path).unwrap(); - let pause = QueryPause::new(); - let pause_id = pause.id; - let pending_reader = reader.clone(); - let pending = - tokio::spawn(async move { pending_reader.query::(None, pause_id).await }); - pause.entered().await; - let refreshed = reader.clone(); - let refresh = tokio::time::timeout( - std::time::Duration::from_secs(5), - refreshed.refresh(&refreshed_path), - ) - .await; - let old_view_retained = reader_path.exists(); - let during_refresh = reader_runtime.stats(); - // Release the SQL callback before asserting progress, so an admission - // regression fails instead of leaving the test's blocking task parked. - pause.release(); - let old = pending.await.unwrap().unwrap(); - assert_eq!( - refresh.unwrap().unwrap().commit_sequence, - committed.commit_sequence() - ); - assert!(old_view_retained); - assert_eq!(during_refresh.worker_jobs(), 1); - assert_eq!(reader_runtime.stats().worker_jobs(), 0); - assert_eq!(during_refresh.file_descriptors(), 8); - assert_eq!(old.output, 0); - assert!(old.receipt.commit_sequence < refreshed.receipt().await.commit_sequence); - assert!(!reader_path.exists()); - assert_eq!(reader_runtime.stats().file_descriptors(), 4); - assert_eq!(during_refresh.local_disk_reserved_bytes(), disk_bytes); - assert_eq!(during_refresh.resident_bytes(), 24 << 20); - assert_eq!( - reader.query::(None, 0).await.unwrap().output, - 1 - ); - assert_eq!( - refreshed - .query::(None, 0) - .await - .unwrap() - .output, - 1 - ); - - lifecycle::drain_waits_for_replica_sql(fixture, ®istry, &directory).await; - - let advertised = directory.load(session, now_ms()).await.unwrap().unwrap(); - directory.withdraw(&advertised, now_ms()).await.unwrap(); - let (warm, ready) = reader.readiness().await.unwrap(); - assert!(!ready); - assert_eq!(warm.commit_sequence, committed.commit_sequence()); - assert!(matches!( - reader.query::(None, 0).await, - Err(crab_cell_runtime::Error::Fenced) - )); - - handle.drain().await.unwrap(); - assert!(runtime.active_catalog_entries().await.unwrap().is_empty()); - assert!(matches!( - reader.query::(None, 0).await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert!(matches!( - refreshed.query::(None, 0).await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert!(matches!( - reader.readiness().await, - Err(crab_cell_runtime::Error::Fenced) - )); - reader.close(); - assert!(matches!( - refreshed.readiness().await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert!(matches!( - replica_client - .query::(&fixture.target, None, 0) - .await, - Err(InvocationError::NotStarted( - crab_cell_runtime::Error::Fenced - )) - )); - drop(replica_client); - drop(local_client); - drop(configured); - drop(reader); - drop(refreshed); - assert!(!reader_path.exists()); - assert!(!refreshed_path.exists()); - assert_eq!(reader_runtime.stats().file_descriptors(), 0); - assert_eq!(reader_runtime.stats().resident_bytes(), 0); - assert_eq!(reader_runtime.stats().local_disk_reserved_bytes(), 0); - reader_runtime.shutdown().await.unwrap(); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/read_replica/lifecycle.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/read_replica/lifecycle.rs deleted file mode 100644 index 6d157ca4c..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/read_replica/lifecycle.rs +++ /dev/null @@ -1,609 +0,0 @@ -//! In-flight snapshot work at the node drain and schema boundaries. - -use super::*; - -pub(super) struct ReadLogicalTime; - -impl Query for ReadLogicalTime { - const MODULE: &'static str = MODULE; - const ID: u32 = 2; - const CODEC_VERSION: u32 = 1; - type Input = (); - type Output = i64; - - fn execute(context: &mut QueryContext<'_>, _: ()) -> crab_cell_runtime::Result { - Ok(context.now_ms()) - } -} - -#[tokio::test] -async fn owner_and_replica_queries_do_not_precede_committed_time() { - let fixture = fixture(); - let owner = SessionId::from_bytes([44; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 << 20, owner).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, owner).await; - let committed_time = now_ms() + 10_000; - handle - .execute( - crate::support::fixtures::mutation_identity(46), - Digest::from_bytes([47; 32]), - committed_time, - 64, - 64, - |_| Ok(HandlerOutcome::Success(Vec::new())), - ) - .await - .unwrap(); - let registry = compiled_reader_registry(); - let directory = owner_directory(&fixture, owner, ®istry).await; - let reader_runtime = CellRuntime::new( - SqlWorkerPool::new(2, 512).unwrap(), - 8 << 20, - SessionId::from_bytes([45; 16]), - ) - .unwrap(); - let reader = CellReadReplica::open( - reader_runtime.clone(), - Arc::clone(®istry), - CellAuthority::new(fixture.layout.clone()), - directory, - fixture.replica.clone(), - fixture.target.clone(), - &fixture._directory.path().join("clock-reader.sqlite"), - ) - .await - .unwrap(); - let local = CellClient::local(registry, handle.clone()); - let local_time = local - .query::(&fixture.target, None, ()) - .await - .unwrap() - .output; - let replica_time = reader - .query::(None, ()) - .await - .unwrap() - .output; - assert_eq!((local_time, replica_time), (committed_time, committed_time)); - drop(reader); - reader_runtime.shutdown().await.unwrap(); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn reader_routing_overlaps_independent_control_and_policy_reads() { - let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let reads = Arc::new(AtomicUsize::new(0)); - let counted = reads.clone(); - let fixture = fixture_with_limits_and_store( - b"parallel-reader-routing", - Limits::default(), - Store::new(store.clone()).with_read_request_observer(Arc::new(move |_| { - counted.fetch_add(1, Ordering::Relaxed); - })), - ); - let registry = compiled_reader_registry(); - let directory = NodeDirectory::new( - fixture.layout.clone(), - Digest::from_bytes([9; 32]), - Digest::from_bytes([10; 32]), - registry.release_digest(), - ); - let router = ReplicaReadRouter::new(CellAuthority::new(fixture.layout.clone()), directory); - reads.store(0, Ordering::Relaxed); - store.arm_gets(); - let selected = router.selected(&fixture.target); - tokio::pin!(selected); - assert!(futures_util::poll!(selected.as_mut()).is_pending()); - let started = reads.load(Ordering::Relaxed); - store.release_gets(); - assert!(matches!( - selected.await, - Err(crab_cell_runtime::Error::ReplicaUnavailable) - )); - assert_eq!( - started, 2, - "policy discovery waited for the blocked control read" - ); -} - -#[tokio::test(flavor = "multi_thread")] -async fn concurrent_refreshes_coalesce_after_the_first_authority_read() { - let store = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let fixture = fixture_with_limits_and_store( - b"coalesced-reader-refresh", - Limits::default(), - Store::new(store.clone()), - ); - let owner = SessionId::from_bytes([44; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 << 20, owner).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, owner).await; - let registry = compiled_reader_registry(); - let directory = owner_directory(&fixture, owner, ®istry).await; - let reads = Arc::new(ControlReads::default()); - let authority = CellAuthority::with_telemetry( - fixture.layout.clone(), - crab_cell_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(reads.clone()), - ); - let reader_runtime = CellRuntime::new( - SqlWorkerPool::new(2, 512).unwrap(), - 8 << 20, - SessionId::from_bytes([45; 16]), - ) - .unwrap(); - let original = fixture._directory.path().join("original.sqlite"); - let reader = CellReadReplica::open( - reader_runtime.clone(), - registry, - authority, - directory, - fixture.replica.clone(), - fixture.target.clone(), - &original, - ) - .await - .unwrap(); - let committed = handle - .execute( - crate::support::fixtures::mutation_identity(46), - Digest::from_bytes([47; 32]), - now_ms(), - 64, - 64, - |transaction| { - transaction.execute("UPDATE counter SET value = 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - store.arm_gets(); - let first_path = fixture._directory.path().join("first.sqlite"); - let second_path = fixture._directory.path().join("second.sqlite"); - let first_reader = reader.clone(); - let destination = first_path.clone(); - let first = tokio::spawn(async move { first_reader.refresh(&destination).await }); - tokio::time::timeout( - std::time::Duration::from_secs(3), - store.wait_until_get_blocked(), - ) - .await - .unwrap(); - let before = reads.0.load(Ordering::Relaxed); - let (first, second, pending, reads_while_blocked) = { - let second = reader.refresh(&second_path); - tokio::pin!(second); - let pending = futures_util::poll!(second.as_mut()).is_pending(); - let reads_while_blocked = reads.0.load(Ordering::Relaxed) - before; - // Release the injected I/O stall before assertions so a failed - // coalescing invariant cannot leave a runtime task parked forever. - store.release_gets(); - let (first, second) = tokio::join!(first, second); - ( - first.unwrap().unwrap(), - second.unwrap(), - pending, - reads_while_blocked, - ) - }; - assert!(pending); - assert_eq!(reads_while_blocked, 0); - assert_eq!(first, second); - assert_eq!(first.commit_sequence, committed.commit_sequence()); - assert!(!original.exists()); - assert!(first_path.exists()); - assert!(!second_path.exists()); - assert_eq!(reader_runtime.stats().resident_bytes(), 12 << 20); - assert_eq!( - reader.query::(None, 0).await.unwrap().output, - 1 - ); - reader.close(); - drop(reader); - assert!(!first_path.exists()); - assert_eq!(reader_runtime.stats().resident_bytes(), 0); - reader_runtime.shutdown().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn schema_change_fences_an_inflight_old_snapshot_query() { - exercise_schema_change(fixture_for(b"read-replica-migration")).await; -} - -pub(super) async fn exercise_schema_change(fixture: Fixture) { - let session = SessionId::from_bytes([34; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 << 20, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - let registry = compiled_reader_registry(); - let directory = owner_directory(&fixture, session, ®istry).await; - let authority = CellAuthority::new(fixture.layout.clone()); - let reader_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 512).unwrap(), - 8 << 20, - SessionId::from_bytes([35; 16]), - ) - .unwrap(); - let old_path = fixture._directory.path().join("old-schema.sqlite"); - let reader = CellReadReplica::open( - reader_runtime.clone(), - Arc::clone(®istry), - authority.clone(), - directory.clone(), - fixture.replica.clone(), - fixture.target.clone(), - &old_path, - ) - .await - .unwrap(); - let before = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let pause = QueryPause::new(); - let pause_id = pause.id; - let pending_reader = reader.clone(); - let pending = - tokio::spawn(async move { pending_reader.query::(None, pause_id).await }); - pause.entered().await; - let plan = registry - .next_migration(NAMESPACE, handle.code(), handle.schema()) - .unwrap() - .unwrap(); - let migrated = tokio::time::timeout( - std::time::Duration::from_secs(3), - handle.migrate(plan, now_ms()), - ) - .await; - pause.release(); - let migrated = migrated.unwrap().unwrap(); - assert!(matches!( - pending.await.unwrap(), - Err(crab_cell_runtime::Error::Fenced) - )); - let after = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(after.value().schema, 2); - assert_eq!(after.value().code, registry.module_code(MODULE).unwrap()); - assert_eq!(after.value().owner, before.value().owner); - assert_eq!(after.value().epoch, before.value().epoch); - assert!( - after.value().root.as_ref().unwrap().commit_sequence - > before.value().root.as_ref().unwrap().commit_sequence - ); - let rejected_path = fixture._directory.path().join("stale-refresh.sqlite"); - assert!(matches!( - reader.refresh(&rejected_path).await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert!(!rejected_path.exists()); - let new_path = fixture._directory.path().join("new-schema.sqlite"); - let replacement = CellReadReplica::open( - reader_runtime.clone(), - registry, - authority, - directory, - fixture.replica, - fixture.target, - &new_path, - ) - .await - .unwrap(); - assert_eq!( - replacement - .query::(None, 0) - .await - .unwrap() - .output, - 10 - ); - reader.close(); - replacement.close(); - drop(reader); - drop(replacement); - assert!(!old_path.exists()); - assert!(!new_path.exists()); - migrated.handle.drain().await.unwrap(); - reader_runtime.shutdown().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -pub(super) async fn drain_waits_for_replica_sql( - fixture: &Fixture, - registry: &Arc, - directory: &NodeDirectory, -) { - for cancel in [false, true] { - drain_query(fixture, registry, directory, cancel).await; - } -} - -async fn drain_query( - fixture: &Fixture, - registry: &Arc, - directory: &NodeDirectory, - cancel: bool, -) { - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 256).unwrap(), - 8 << 20, - SessionId::from_bytes([33; 16]), - ) - .unwrap(); - let path = fixture._directory.path().join("draining-reader.sqlite"); - let reader = CellReadReplica::open( - runtime.clone(), - Arc::clone(registry), - CellAuthority::new(fixture.layout.clone()), - directory.clone(), - fixture.replica.clone(), - fixture.target.clone(), - &path, - ) - .await - .unwrap(); - let pause = QueryPause::new(); - let pause_id = pause.id; - let pending_reader = reader.clone(); - let mut pending = - tokio::spawn(async move { pending_reader.query::(None, pause_id).await }); - pause.entered().await; - reader.close(); - drop(reader); - let cancelled = if cancel { - pending.abort(); - Some((&mut pending).await.unwrap_err()) - } else { - None - }; - let retained_while_running = runtime.stats().resident_bytes(); - let descriptors_while_running = runtime.stats().file_descriptors(); - let draining = runtime.clone(); - let mut shutdown = tokio::spawn(async move { draining.shutdown().await }); - let early = tokio::time::timeout(std::time::Duration::from_millis(50), &mut shutdown) - .await - .ok(); - let completed_early = early.is_some(); - let jobs_during_drain = runtime.stats().worker_jobs(); - // Always release the SQL callback before asserting the drain result, so a - // regression cannot strand a blocking worker and hang the test process. - pause.release(); - let output = match cancelled { - Some(error) => { - assert!(error.is_cancelled()); - None - } - None => Some(pending.await.unwrap()), - }; - match early { - Some(result) => result.unwrap().unwrap(), - None => shutdown.await.unwrap().unwrap(), - } - assert!(matches!( - output, - None | Some(Err(crab_cell_runtime::Error::Fenced)) - )); - assert_eq!(retained_while_running, 12 << 20); - assert_eq!(descriptors_while_running, 4); - assert_eq!(jobs_during_drain, 1); - assert_eq!(runtime.stats().worker_jobs(), 0); - assert_eq!(runtime.stats().resident_bytes(), 0); - assert!(!path.exists()); - assert!( - !completed_early, - "node drain returned while replica SQL was running" - ); -} - -struct StalledRead(Arc); - -impl Drop for StalledRead { - fn drop(&mut self) { - self.0.fetch_add(1, Ordering::Relaxed); - } -} - -impl PeerReplicaResolver for StalledRead { - fn resolve( - &self, - _: CellTarget, - ) -> std::pin::Pin< - Box< - dyn std::future::Future> - + Send - + 'static, - >, - > { - let guard = Self(self.0.clone()); - Box::pin(async move { - let _guard = guard; - std::future::pending().await - }) - } -} - -struct StalledPeer { - session: Option, - dropped: Arc, - healthy: LoopbackReplica, -} - -impl PeerRoundTrip for StalledPeer { - fn send( - &self, - target: CellTarget, - request: Vec, - remaining_ms: u32, - ) -> std::pin::Pin< - Box>> + Send + 'static>, - > { - self.healthy.send(target, request, remaining_ms) - } - - fn send_to_node( - &self, - target: CellTarget, - node: NodeAdvertisement, - request: Vec, - remaining_ms: u32, - ) -> std::pin::Pin< - Box>> + Send + 'static>, - > { - if Some(node.session()) != self.session { - return self - .healthy - .send_to_node(target, node, request, remaining_ms); - } - let guard = StalledRead(self.dropped.clone()); - Box::pin(async move { - let _guard = guard; - std::future::pending().await - }) - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn stalled_remote_reader_leaves_time_for_a_healthy_replica() { - stalled_reader_is_skipped(false).await; -} - -#[tokio::test(flavor = "multi_thread")] -async fn stalled_local_resolver_leaves_time_for_a_healthy_replica() { - stalled_reader_is_skipped(true).await; -} - -async fn stalled_reader_is_skipped(local: bool) { - let fixture = fixture(); - let owner = SessionId::from_bytes([44; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 << 20, owner).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, owner).await; - let registry = compiled_reader_registry(); - let directory = owner_directory(&fixture, owner, ®istry).await; - for node in [14, 15] { - let now = now_ms(); - directory - .create( - NodeAdvertisement::sign( - NodeId::from_bytes([node; 16]), - SessionId::from_bytes([node; 16]), - format!("https://reader-{node}.internal:8081"), - directory.fleet(), - Digest::from_bytes([12; 32]), - Digest::from_bytes([10; 32]), - registry.release_digest(), - &ed25519_dalek::SigningKey::from_bytes(&[node; 32]), - 1, - now, - now + 15_000, - vec![CODE], - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 32 << 20, - free_disk_bytes: 1 << 20, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .unwrap(), - now, - ) - .await - .unwrap(); - } - let reader_runtime = CellRuntime::new( - SqlWorkerPool::new(2, 2) - .unwrap() - .with_native_memory_limit(32 << 20) - .unwrap(), - 8 << 20, - SessionId::from_bytes([14; 16]), - ) - .unwrap(); - let reader = CellReadReplica::open( - reader_runtime.clone(), - registry.clone(), - CellAuthority::new(fixture.layout.clone()), - directory.clone(), - fixture.replica.clone(), - fixture.target.clone(), - &fixture._directory.path().join("routing-reader.sqlite"), - ) - .await - .unwrap(); - let receipt = reader.receipt().await; - crab_cell_runtime::read_policy::ReadPolicyStore::new(fixture.layout.clone()) - .create(fixture.target.cell_id(), receipt.incarnation, 2) - .await - .unwrap(); - let router = ReplicaReadRouter::new(CellAuthority::new(fixture.layout.clone()), directory); - let (_, selected) = router.selected(&fixture.target).await.unwrap(); - assert_eq!(selected.len(), 2); - let blocked = selected[0].session(); - let dropped = Arc::new(AtomicUsize::new(0)); - let local_resolver = StalledRead(dropped.clone()); - let peer_session = SessionId::from_bytes([21; 16]); - let key = ed25519_dalek::SigningKey::from_bytes(&[22; 32]); - let dispatcher = PeerDispatcher::new( - registry.clone(), - Arc::new(NoOwner), - Arc::new(ReadAuthorizer), - ) - .with_replica_resolver(Arc::new(ReplicaResolver(reader.clone()))); - let transport = StalledPeer { - session: (!local).then_some(blocked), - dropped: dropped.clone(), - healthy: LoopbackReplica { - verifier: Arc::new(PeerVerifier::new( - peer_session, - registry.release_digest(), - key.verifying_key(), - )), - dispatcher: Arc::new(dispatcher), - }, - }; - let peer = ReplicaPeerClient::new( - registry.clone(), - Arc::new(PeerSigner::new( - peer_session, - registry.release_digest(), - key, - )), - PeerPrincipal { - issuer: "test".into(), - subject: "reader".into(), - actions: vec!["repository.read".into()], - }, - Arc::new(transport), - ); - let queried = router - .query::( - &peer, - local.then_some((blocked, &local_resolver as &dyn PeerReplicaResolver)), - &fixture.target, - Some(receipt), - 0, - ) - .await; - // Drain even on the red run, so a failed routing assertion cannot strand SQL workers. - drop(peer); - drop(reader); - reader_runtime.shutdown().await.unwrap(); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - let (observed, served_by) = queried.unwrap(); - assert_eq!( - (observed.output, observed.receipt, served_by), - (0, receipt, selected[1].node()) - ); - assert_eq!( - dropped.load(Ordering::Relaxed), - 1, - "stalled attempt was not cancelled" - ); -} diff --git a/crates/crab-cell-runtime/tests/runtime/lifecycle/residency.rs b/crates/crab-cell-runtime/tests/runtime/lifecycle/residency.rs deleted file mode 100644 index 6ff75e3e6..000000000 --- a/crates/crab-cell-runtime/tests/runtime/lifecycle/residency.rs +++ /dev/null @@ -1,1912 +0,0 @@ -//! Resident routes, hydration, and owner progress. - -use super::*; - -#[tokio::test] -async fn activation_opens_one_cache_at_the_database_directory_off_async_worker() { - let fixture = fixture_for(b"activation-cache-owner"); - let filesystem = Arc::new(crate::runtime::fault_fs::FaultFileSystem::new()); - let host = ReplicaHost::default() - .with_filesystem(filesystem.clone()) - .with_local_disk_budget(DiskBudget::new(1 << 30)); - let session = SessionId::from_bytes([181; 16]); - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 1).unwrap(), - 16 << 20, - session, - host.clone(), - ) - .unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - drop(handle); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let authority = CellAuthority::new(fixture.layout.clone()); - let directory = cold_node_directory(&fixture); - // Cold recovery and then clean local resume must share the publisher's - // one cache owner instead of reopening it at a parent directory. - for (number, name) in [(182, "cold"), (183, "resumed")] { - let session = SessionId::from_bytes([number; 16]); - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 1).unwrap(), - 16 << 20, - session, - host.clone(), - ) - .unwrap(); - let handle = runtime - .acquire_idle_restored( - catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(), - fixture.replica.clone(), - authority.clone(), - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(), - directory.join(format!("{name}.sqlite")), - Owner { - session, - endpoint: "https://cache-owner.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!( - handle - .query(64, 64, |db| { - let value: i64 = - db.query_row("SELECT value FROM counter", [], |row| row.get(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(), - 0_i64.to_be_bytes() - ); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - } - let opens = filesystem.cache_opens(); - assert_eq!( - opens - .iter() - .map(|(path, _)| path.clone()) - .collect::>(), - [ - fixture - .database - .parent() - .unwrap() - .join(".crab-cell-directory-cache"), - directory.join(".crab-cell-directory-cache"), - directory.join(".crab-cell-directory-cache"), - ] - ); - assert!( - opens - .iter() - .all(|(_, thread)| *thread != std::thread::current().id()) - ); -} - -/// Counts the metadata and origin reads one cold route performs. -#[derive(Default)] -struct ColdPathRecorder { - catalog_reads: AtomicUsize, - control_reads: AtomicUsize, - origin_requests: AtomicUsize, - ltx_phases: AtomicUsize, - activation_phases: std::sync::Mutex>, -} - -impl ColdPathRecorder { - fn catalog_reads(&self) -> usize { - self.catalog_reads.load(Ordering::Acquire) - } - - fn control_reads(&self) -> usize { - self.control_reads.load(Ordering::Acquire) - } - - fn origin_requests(&self) -> usize { - self.origin_requests.load(Ordering::Acquire) - } - - fn ltx_phases(&self) -> usize { - self.ltx_phases.load(Ordering::Acquire) - } - - fn activation_phases(&self) -> Vec { - self.activation_phases.lock().unwrap().clone() - } -} - -impl crab_cell_runtime::fleet::telemetry::CellTelemetry for ColdPathRecorder { - fn catalog_read( - &self, - _kind: crab_cell_runtime::fleet::telemetry::CatalogReadKind, - _elapsed: std::time::Duration, - _succeeded: bool, - ) { - self.catalog_reads.fetch_add(1, Ordering::AcqRel); - } - - fn control_read(&self, _elapsed: std::time::Duration, _succeeded: bool) { - self.control_reads.fetch_add(1, Ordering::AcqRel); - } - - fn ltx_phase( - &self, - _phase: crab_cell_runtime::ltx::LtxPhase, - _elapsed: std::time::Duration, - _succeeded: bool, - ) { - self.ltx_phases.fetch_add(1, Ordering::AcqRel); - } - - fn ltx_origin_request( - &self, - _origin: crab_cell_runtime::ltx::LtxReadOrigin, - _outcome: crab_cell_runtime::ltx::LtxRequestOutcome, - _bytes: u64, - ) { - self.origin_requests.fetch_add(1, Ordering::AcqRel); - } - - fn activation_phase( - &self, - phase: crab_cell_runtime::fleet::telemetry::ActivationPhase, - _elapsed: std::time::Duration, - ) { - self.activation_phases.lock().unwrap().push(phase); - } -} - -#[tokio::test] -async fn cold_activation_reports_metadata_and_origin_reads() { - let fixture = fixture_for(b"cold-path-counts"); - let session = SessionId::from_bytes([112; 16]); - let first_runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&first_runtime, &fixture, session).await; - handle.drain().await.unwrap(); - first_runtime.shutdown().await.unwrap(); - - let recorder = Arc::new(ColdPathRecorder::default()); - let sink = - crab_cell_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(recorder.clone()); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::with_telemetry( - fixture.layout.clone(), - fixture.target.tenant(), - sink.clone(), - ); - let authority = CellAuthority::with_telemetry(fixture.layout.clone(), sink.clone()); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let successor = SessionId::from_bytes([113; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 16 * 1024 * 1024, - successor, - ) - .unwrap(); - runtime.install_telemetry(recorder.clone()).unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - idle, - cold_node_directory(&fixture).join("cold-path.sqlite"), - Owner { - session: successor, - endpoint: "https://cold-path.internal:8081".into(), - }, - ) - .await - .unwrap(); - - // The window above is one cold route: catalog locator, control record, and - // the root open that fetches origin objects. A resident route issues none - // of them, so these counts are the cold-start cost. Four of the seven - // object-store requests are metadata; changing them changes what a cold - // start costs, so a change here must be deliberate. - assert_eq!(recorder.catalog_reads(), 2); - // The caller's decision read plus the acquisition's own confirmation. - assert_eq!(recorder.control_reads(), 2); - assert_eq!(recorder.origin_requests(), 3); - assert!(recorder.ltx_phases() >= 1); - // The phases decompose the cold route: claim, verify the root, restore it - // locally, then activate. A warm route would record none of them. - assert_eq!( - recorder.activation_phases(), - [ - crab_cell_runtime::fleet::telemetry::ActivationPhase::Ownership, - crab_cell_runtime::fleet::telemetry::ActivationPhase::RootOpen, - crab_cell_runtime::fleet::telemetry::ActivationPhase::Restore, - crab_cell_runtime::fleet::telemetry::ActivationPhase::Activate, - ] - ); - - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -/// Returns a node directory that holds no local image of the fixture's Cell. -/// -/// A cold route is what a node without a readable resume record pays, so tests -/// that measure it must not reuse the directory the released database is in. -fn cold_node_directory(fixture: &Fixture) -> std::path::PathBuf { - let directory = fixture._directory.path().join("cold-node"); - std::fs::create_dir_all(&directory).unwrap(); - directory -} - -/// Boots one Cell, commits once, and releases it, leaving a resume record. -async fn released_cell(fixture: &Fixture, session: SessionId, start_ms: i64) -> i64 { - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, fixture, session).await; - handle - .execute( - mutation_identity_window(161, start_ms, start_ms + 10_000), - Digest::from_bytes([162; 32]), - start_ms, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - start_ms -} - -/// Wakes a released Cell on the same node and returns what it read and ran. -async fn wake_released_cell( - fixture: &Fixture, - successor: SessionId, - destination: std::path::PathBuf, -) -> ( - Arc, - crab_cell_runtime::cell::actor::CellHandle, - CellRuntime, -) { - let recorder = Arc::new(ColdPathRecorder::default()); - let sink = - crab_cell_runtime::fleet::telemetry::CellTelemetryHandle::from_sink(recorder.clone()); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::with_telemetry( - fixture.layout.clone(), - fixture.target.tenant(), - sink.clone(), - ); - let authority = CellAuthority::with_telemetry(fixture.layout.clone(), sink.clone()); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 16 * 1024 * 1024, - successor, - ) - .unwrap(); - runtime.install_telemetry(recorder.clone()).unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - idle, - destination, - Owner { - session: successor, - endpoint: "https://warm-wake.internal:8081".into(), - }, - ) - .await - .unwrap(); - (recorder, restored, runtime) -} - -#[tokio::test] -async fn a_warm_wake_continues_the_local_database_without_the_origin() { - let fixture = fixture_for(b"warm-wake"); - let start_ms = now_ms(); - released_cell(&fixture, SessionId::from_bytes([163; 16]), start_ms).await; - - // The same node wakes the same root. The release record makes the local - // image the fast path, so this activation reads no origin object at all - // and never verifies or restores the immutable root graph. - let successor = SessionId::from_bytes([164; 16]); - let (recorder, restored, runtime) = wake_released_cell( - &fixture, - successor, - fixture._directory.path().join("warm.sqlite"), - ) - .await; - assert_eq!(recorder.origin_requests(), 0); - assert_eq!( - recorder.activation_phases(), - [ - crab_cell_runtime::fleet::telemetry::ActivationPhase::Ownership, - crab_cell_runtime::fleet::telemetry::ActivationPhase::Resume, - crab_cell_runtime::fleet::telemetry::ActivationPhase::Activate, - ] - ); - - // The resumed database continues the lineage rather than resetting it: the - // commit from the released session is readable, and the next commit lands - // on the same chain. - let outcome = restored - .execute( - mutation_identity_window(165, start_ms + 20_000, start_ms + 30_000), - Digest::from_bytes([166; 32]), - start_ms + 20_000, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - let value: i64 = - transaction.query_row("SELECT value FROM counter", [], |row| row.get(0))?; - Ok(HandlerOutcome::Success(value.to_be_bytes().to_vec())) - }, - ) - .await - .unwrap(); - assert_eq!(outcome.result(), 2i64.to_be_bytes()); - - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn a_resume_record_that_names_another_root_is_discarded() { - let fixture = fixture_for(b"stale-resume"); - let start_ms = now_ms(); - released_cell(&fixture, SessionId::from_bytes([167; 16]), start_ms).await; - let record = fixture._directory.path().join("cell.sqlite.resume"); - let mut bytes = std::fs::read(&record).unwrap(); - // Field layout: magic, version, schema, code, cell, incarnation, then the - // root digest the record has to match against the observed control. - bytes[96] ^= 0xff; - std::fs::write(&record, bytes).unwrap(); - - // A record that no longer names the observed root must be discarded with - // the database it names, and the wake must fall back to the exact restore. - let successor = SessionId::from_bytes([168; 16]); - let (recorder, restored, runtime) = wake_released_cell( - &fixture, - successor, - fixture._directory.path().join("cold-again.sqlite"), - ) - .await; - assert!(recorder.origin_requests() > 0); - assert_eq!( - recorder.activation_phases(), - [ - crab_cell_runtime::fleet::telemetry::ActivationPhase::Ownership, - crab_cell_runtime::fleet::telemetry::ActivationPhase::RootOpen, - crab_cell_runtime::fleet::telemetry::ActivationPhase::Restore, - crab_cell_runtime::fleet::telemetry::ActivationPhase::Activate, - ] - ); - assert!(!fixture.database.exists()); - - let outcome = restored - .execute( - mutation_identity_window(169, start_ms + 20_000, start_ms + 30_000), - Digest::from_bytes([170; 32]), - start_ms + 20_000, - 1_024, - 1_024, - |transaction| { - let value: i64 = - transaction.query_row("SELECT value FROM counter", [], |row| row.get(0))?; - Ok(HandlerOutcome::Success(value.to_be_bytes().to_vec())) - }, - ) - .await - .unwrap(); - assert_eq!(outcome.result(), 1i64.to_be_bytes()); - - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn clean_drain_publishes_and_consumes_one_due_hint() { - let fixture = fixture_for(b"due-hint"); - let start_ms = now_ms(); - let session = SessionId::from_bytes([118; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - handle - .execute( - mutation_identity_window(119, start_ms, start_ms + 10_000), - Digest::from_bytes([120; 32]), - start_ms, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - - let authority = CellAuthority::new(fixture.layout.clone()); - let control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let due_ms = control.value().next_due_ms.unwrap(); - // Eviction releases ownership from a spawned task, so the hint lands just - // after control turns Idle: wait for the key instead of racing it. - let bucket = crab_cell_runtime::cell::due::bucket_for(due_ms).unwrap(); - let expected = fixture - .layout - .due_hint_path(bucket, fixture.target.cell_id().as_bytes()); - tokio::time::timeout(std::time::Duration::from_secs(3), async { - loop { - if fixture - .layout - .store() - .get_with_etag_bounded(&expected, 8) - .await - .is_ok() - { - break; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); - assert_eq!( - crab_cell_runtime::cell::due::take(&fixture.layout, due_ms, 8) - .await - .unwrap(), - vec![fixture.target.cell_id()] - ); - // Consuming a hint deletes it; the backstop scan is what makes that safe. - assert!( - crab_cell_runtime::cell::due::take(&fixture.layout, due_ms, 8) - .await - .unwrap() - .is_empty() - ); -} - -#[tokio::test] -async fn idle_eviction_publishes_the_same_due_hint() { - let fixture = fixture_for(b"due-hint-eviction"); - let start_ms = now_ms(); - let session = SessionId::from_bytes([122; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - handle - .execute( - mutation_identity_window(123, start_ms, start_ms + 10_000), - Digest::from_bytes([124; 32]), - start_ms, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - // Pressure eviction, not a clean drain: it must leave the same hint, or - // the thirty-cycle backstop would carry every evicted deadline. - tokio::time::timeout(std::time::Duration::from_secs(3), async { - loop { - if runtime.evict_idle(1).await.unwrap() == 1 { - break; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let due_ms = control.value().next_due_ms.unwrap(); - // The release runs in a spawned task, so the hint lands just after control - // turns Idle: wait for the key instead of racing it. - let bucket = crab_cell_runtime::cell::due::bucket_for(due_ms).unwrap(); - let expected = fixture - .layout - .due_hint_path(bucket, fixture.target.cell_id().as_bytes()); - tokio::time::timeout(std::time::Duration::from_secs(3), async { - loop { - if fixture - .layout - .store() - .get_with_etag_bounded(&expected, 8) - .await - .is_ok() - { - break; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); - assert!( - fixture - .layout - .store() - .get_with_etag_bounded(&expected, 8) - .await - .is_ok(), - "a pressure eviction leaves a hint in the released head's bucket" - ); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn a_cleanly_closed_database_survives_a_rename_to_a_fresh_path() { - let fixture = fixture_for(b"spike-rename"); - let session = SessionId::from_bytes([131; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - - // A clean session's file, renamed to a path nothing has ever opened: the - // capture-directory fence is per path, so this is the candidate mechanism - // for a warm wake that reuses local bytes. - let old = fixture.database.clone(); - let fresh = fixture._directory.path().join("fresh-name.sqlite"); - std::fs::rename(&old, &fresh).unwrap(); - let mut db = crab_ltx::Db::open_with_host( - &fresh, - crab_ltx::Limits::default(), - crab_ltx::Host::default(), - ) - .unwrap(); - let rows = db - .query_with(|connection| { - connection.query_row("SELECT count(*) FROM counter", [], |row| { - row.get::<_, i64>(0) - }) - }) - .unwrap(); - assert_eq!(rows, 1, "the renamed database still holds its rows"); - db.close().unwrap(); -} - -#[tokio::test] -async fn due_hint_publication_skips_a_deadline_outside_the_window() { - let fixture = fixture_for(b"due-hint-window"); - let bucket_ms = crab_cell_runtime::cell::due::HINT_BUCKET_MS; - let now_ms = bucket_ms * 100; - // Older than the listing window: no listing would see this key, so writing - // it would leave an object behind that nothing ever consumes. - let stale_ms = now_ms - bucket_ms * 10; - crab_cell_runtime::cell::due::publish( - &fixture.layout, - fixture.target.cell_id(), - stale_ms, - now_ms, - ) - .await - .unwrap(); - let stale_group = crab_cell_runtime::cell::due::bucket_for(stale_ms).unwrap(); - assert!( - fixture - .layout - .store() - .get_with_etag_bounded( - &fixture - .layout - .due_hint_path(stale_group, fixture.target.cell_id().as_bytes()), - 8 - ) - .await - .is_err(), - "a deadline outside the window publishes no key" - ); - - // Inside the window the same deadline is published for the next listing. - let inside_ms = now_ms - bucket_ms; - crab_cell_runtime::cell::due::publish( - &fixture.layout, - fixture.target.cell_id(), - inside_ms, - now_ms, - ) - .await - .unwrap(); - let inside_group = crab_cell_runtime::cell::due::bucket_for(inside_ms).unwrap(); - assert!( - fixture - .layout - .store() - .get_with_etag_bounded( - &fixture - .layout - .due_hint_path(inside_group, fixture.target.cell_id().as_bytes()), - 8 - ) - .await - .is_ok(), - "a deadline inside the window publishes its key" - ); -} - -#[tokio::test] -async fn due_hint_listing_clears_foreign_keys() { - let fixture = fixture_for(b"due-hint-junk"); - let bucket = crab_cell_runtime::cell::due::bucket_for(1_000).unwrap(); - let foreign = object_store::path::Path::from(format!( - "{}/not-a-cell.json", - fixture.layout.due_hint_prefix(bucket) - )); - fixture - .layout - .store() - .put(&foreign, bytes::Bytes::new()) - .await - .unwrap(); - assert!( - fixture - .layout - .store() - .get_with_etag_bounded(&foreign, 1_024) - .await - .is_ok(), - "the foreign key must exist before the listing" - ); - assert!( - crab_cell_runtime::cell::due::take(&fixture.layout, 1_000, 8) - .await - .unwrap() - .is_empty(), - "a foreign key is not a candidate" - ); - assert!( - fixture - .layout - .store() - .get_with_etag_bounded(&foreign, 1_024) - .await - .is_err(), - "a foreign key must not be listed on every cycle" - ); -} - -#[tokio::test] -async fn due_hint_listing_rejects_nested_cell_keys() { - let fixture = fixture_for(b"due-hint-nested"); - let due_ms = 1_000; - let bucket = crab_cell_runtime::cell::due::bucket_for(due_ms).unwrap(); - let canonical = fixture - .layout - .due_hint_path(bucket, fixture.target.cell_id().as_bytes()); - let nested = object_store::path::Path::from(format!( - "{}/nested/{}", - fixture.layout.due_hint_prefix(bucket), - canonical.filename().unwrap() - )); - fixture - .layout - .store() - .put(&nested, bytes::Bytes::new()) - .await - .unwrap(); - - assert!( - crab_cell_runtime::cell::due::take(&fixture.layout, due_ms, 8) - .await - .unwrap() - .is_empty(), - "a valid Cell filename outside its canonical path is not a due hint" - ); - assert!( - fixture - .layout - .store() - .get_with_etag_bounded(&nested, 8) - .await - .is_err() - ); - - crab_cell_runtime::cell::due::publish( - &fixture.layout, - fixture.target.cell_id(), - due_ms, - due_ms, - ) - .await - .unwrap(); - assert_eq!( - crab_cell_runtime::cell::due::take(&fixture.layout, due_ms, 8) - .await - .unwrap(), - vec![fixture.target.cell_id()] - ); -} - -#[tokio::test] -async fn due_hint_listing_uses_remaining_capacity_across_buckets() { - let fixture = fixture_for(b"due-hint-multiple-buckets"); - let previous = fixture.target.cell_id(); - let current = crab_cell_runtime::identity::CellId::from_bytes([17; 32]); - let now_ms = crab_cell_runtime::cell::due::HINT_BUCKET_MS * 10; - crab_cell_runtime::cell::due::publish(&fixture.layout, previous, now_ms - 1, now_ms) - .await - .unwrap(); - crab_cell_runtime::cell::due::publish(&fixture.layout, current, now_ms, now_ms) - .await - .unwrap(); - - assert_eq!( - crab_cell_runtime::cell::due::take(&fixture.layout, now_ms, 2) - .await - .unwrap(), - vec![current, previous] - ); -} - -#[tokio::test] -async fn resident_due_list_mirrors_the_published_head() { - let fixture = fixture_for(b"resident-due-list"); - let (runtime, handle, _pool) = activate_runtime(&fixture, 2 * 1024 * 1024).await; - // A bootstrap that arms no deadline publishes no due time. - assert!(runtime.due_resident(i64::MAX, 4).await.unwrap().is_empty()); - let now_ms = 1_000; - // An accepted command leaves a retention row, so the Cell publishes an - // earliest due class. The due list must answer from that mirror without - // reading the catalog entry or the control record. - let committed = handle - .execute( - mutation_identity_window(97, 1_000, 10_000), - Digest::from_bytes([98; 32]), - now_ms, - 1_024, - 1_024, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - assert!(runtime.due_resident(now_ms, 4).await.unwrap().is_empty()); - let due = runtime.due_resident(i64::MAX, 4).await.unwrap(); - assert_eq!(due.len(), 1); - assert_eq!(due[0].handle().cell_id(), handle.cell_id()); - assert_eq!( - due[0].expected_commit_sequence(), - committed.commit_sequence() - ); - assert!(due[0].next_due_ms() > now_ms); - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn runtime_stats_follow_active_cell_lifecycle() { - let fixture = fixture_for(b"runtime-stats"); - let (runtime, handle, _pool) = activate_runtime(&fixture, 2 * 1024 * 1024).await; - - assert_eq!(runtime.stats().active_cells(), 1); - assert_eq!( - runtime.stats().resident_bytes(), - crab_cell_runtime::cell::actor::ACTIVE_CELL_NATIVE_BYTES as usize - ); - assert_eq!( - runtime.stats().resident_capacity_bytes(), - 10 * crab_cell_runtime::cell::actor::ACTIVE_CELL_NATIVE_BYTES as usize - ); - assert_eq!( - runtime.stats().file_descriptors(), - ACTIVE_CELL_FILE_DESCRIPTORS - ); - assert_eq!( - runtime.stats().file_descriptor_capacity(), - 10 * ACTIVE_CELL_FILE_DESCRIPTORS - ); - handle.drain().await.unwrap(); - assert_eq!(runtime.stats().active_cells(), 0); - assert_eq!(runtime.stats().resident_bytes(), 0); - assert_eq!(runtime.stats().file_descriptors(), 0); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn resident_lookup_is_invalidated_before_drain_releases_the_cell() { - let fixture = fixture_for(b"resident-drain-race"); - let (runtime, handle, _pool) = activate_runtime(&fixture, 2 * 1024 * 1024).await; - - assert!( - runtime - .resident_handle(&fixture.target, CatalogRole::Repository) - .await - .unwrap() - .is_some() - ); - - handle.drain().await.unwrap(); - - assert!( - runtime - .resident_handle(&fixture.target, CatalogRole::Repository) - .await - .unwrap() - .is_none() - ); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn resident_route_reports_zero_origin_reads_and_latency_percentiles() { - let origin_reads = Arc::new(std::sync::atomic::AtomicUsize::new(0)); - let observed_reads = Arc::clone(&origin_reads); - let store = - Store::new(Arc::new(InMemory::new())).with_read_request_observer(Arc::new(move |_kind| { - observed_reads.fetch_add(1, Ordering::AcqRel); - })); - let fixture = - fixture_with_limits_and_store(b"resident-warm-qualification", Limits::default(), store); - let (runtime, handle, _pool) = activate_runtime(&fixture, 2 * 1024 * 1024).await; - origin_reads.store(0, Ordering::Release); - - let mut samples = Vec::with_capacity(64); - for _ in 0..64 { - let started = std::time::Instant::now(); - let resident = runtime - .resident_handle(&fixture.target, CatalogRole::Repository) - .await - .unwrap() - .expect("bootstrapped Cell must remain resident"); - let value = resident - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(); - assert_eq!(i64::from_be_bytes(value.try_into().unwrap()), 0); - samples.push(started.elapsed()); - } - - samples.sort_unstable(); - let percentile = |percent: usize| { - let index = ((samples.len() - 1) * percent).div_ceil(100); - samples[index] - }; - println!( - "resident warm route: samples={} p50_us={} p95_us={} p99_us={} max_us={}", - samples.len(), - percentile(50).as_micros(), - percentile(95).as_micros(), - percentile(99).as_micros(), - samples.last().unwrap().as_micros() - ); - assert_eq!(origin_reads.load(Ordering::Acquire), 0); - - handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn restored_sparse_route_promotes_before_zero_origin_reads() { - let origin_reads = Arc::new(std::sync::atomic::AtomicUsize::new(0)); - let observed_reads = Arc::clone(&origin_reads); - let store = - Store::new(Arc::new(InMemory::new())).with_read_request_observer(Arc::new(move |_kind| { - observed_reads.fetch_add(1, Ordering::AcqRel); - })); - let fixture = fixture_with_limits_and_store( - b"resident-warm-restart-qualification", - Limits::default(), - store, - ); - let session = SessionId::from_bytes([110; 16]); - let first_runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&first_runtime, &fixture, session).await; - handle.drain().await.unwrap(); - first_runtime.shutdown().await.unwrap(); - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let successor = SessionId::from_bytes([111; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 16 * 1024 * 1024, - successor, - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - idle, - fixture._directory.path().join("warm-restart.sqlite"), - Owner { - session: successor, - endpoint: "https://warm-restart.internal:8081".into(), - }, - ) - .await - .unwrap(); - - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - if runtime - .resident_handle(&fixture.target, CatalogRole::Repository) - .await - .unwrap() - .is_some() - { - break; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); - origin_reads.store(0, Ordering::Release); - - let resident = runtime - .resident_handle(&fixture.target, CatalogRole::Repository) - .await - .unwrap() - .expect("verified sparse restore must promote to resident"); - let value = resident - .query(64, 64, |connection| { - let value = connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - .await - .unwrap(); - assert_eq!(i64::from_be_bytes(value.try_into().unwrap()), 0); - assert_eq!(origin_reads.load(Ordering::Acquire), 0); - - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -async fn released_large_cell_with_store( - seed: &'static [u8], - session: SessionId, -) -> (Arc, Fixture) { - let pausing = Arc::new(PausingStore::new(Arc::new(InMemory::new()))); - let store = Store::with_retry( - pausing.clone(), - RetryPolicy { - // Expose transport failures to the runtime's hydration retry policy. - max_attempts: 1, - ..RetryPolicy::DEFAULT - }, - ); - let fixture = fixture_with_limits_and_store(seed, Limits::default(), store); - let first_runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 64 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_role_on( - &first_runtime, - &fixture, - session, - CatalogRole::Repository, - |transaction| { - transaction.execute_batch( - "CREATE TABLE resident(value INTEGER NOT NULL);\ - INSERT INTO resident VALUES(0);\ - CREATE TABLE payload(value BLOB NOT NULL);\ - WITH RECURSIVE numbers(value) AS (\ - SELECT 1 UNION ALL SELECT value + 1 FROM numbers WHERE value < 512\ - )\ - INSERT INTO payload(value) SELECT zeroblob(16384) FROM numbers;", - )?; - Ok(()) - }, - ) - .await; - handle.drain().await.unwrap(); - first_runtime.shutdown().await.unwrap(); - (pausing, fixture) -} - -#[tokio::test(flavor = "multi_thread")] -async fn foreground_query_and_mutation_progress_during_same_cell_hydration() { - let (pausing, fixture) = released_large_cell_with_store( - b"foreground-during-hydration", - SessionId::from_bytes([181; 16]), - ) - .await; - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let session = SessionId::from_bytes([182; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 64 << 20, session).unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority.clone(), - observed, - cold_node_directory(&fixture).join("foreground.sqlite"), - Owner { - session, - endpoint: "https://foreground.internal:8081".into(), - }, - ) - .await - .unwrap(); - let read = || { - restored.query(64, 64, |connection| { - let value: i64 = - connection.query_row("SELECT value FROM resident", [], |row| row.get(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - }; - assert_eq!(read().await.unwrap(), 0_i64.to_be_bytes()); - pausing.arm_gets(); - tokio::time::timeout( - std::time::Duration::from_secs(3), - pausing.wait_until_get_blocked(), - ) - .await - .unwrap(); - let foreground = tokio::time::timeout(std::time::Duration::from_secs(1), async { - let before = read().await?; - let now = now_ms(); - restored - .execute( - mutation_identity_window(183, now, now + 10_000), - Digest::from_bytes([184; 32]), - now, - 64, - 64, - |transaction| { - transaction.execute("UPDATE resident SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await?; - Ok::<_, crab_cell_runtime::Error>((before, read().await?)) - }) - .await; - // Release before checking the deadline so the failing baseline can drain. - pausing.release_gets(); - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - let (before, after) = foreground - .expect("background fetch blocked foreground work on its own Cell") - .unwrap(); - assert_eq!( - (before, after), - (0_i64.to_be_bytes().to_vec(), 1_i64.to_be_bytes().to_vec()) - ); - let root = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root() - .unwrap(); - let destination = fixture._directory.path().join("foreground-restored.sqlite"); - fixture - .replica - .open_root(&root) - .await - .unwrap() - .restore(&destination) - .await - .unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(destination).unwrap(); - let value: i64 = connection - .query_row("SELECT value FROM resident", [], |row| row.get(0)) - .unwrap(); - assert_eq!(value, 1); -} - -#[tokio::test(flavor = "multi_thread")] -async fn retryable_hydration_fetch_failure_preserves_the_serving_owner() { - let (pausing, fixture) = released_large_cell_with_store( - b"retryable-hydration-fetch", - SessionId::from_bytes([185; 16]), - ) - .await; - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let session = SessionId::from_bytes([186; 16]); - let runtime = CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 64 << 20, session).unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - observed, - cold_node_directory(&fixture).join("retryable-fetch.sqlite"), - Owner { - session, - endpoint: "https://retryable-fetch.internal:8081".into(), - }, - ) - .await - .unwrap(); - let read = || { - restored.query(64, 64, |connection| { - let value: i64 = - connection.query_row("SELECT value FROM resident", [], |row| row.get(0))?; - Ok(value.to_be_bytes().to_vec()) - }) - }; - read().await.unwrap(); - pausing.arm_gets(); - tokio::time::timeout( - std::time::Duration::from_secs(3), - pausing.wait_until_get_blocked(), - ) - .await - .unwrap(); - // One failed transport attempt reaches the runtime before its fetch - // deadline, proving the retryable-error branch independently of timeout. - pausing.transient_get_failures.store(1, Ordering::Release); - pausing.release_gets(); - tokio::time::timeout(std::time::Duration::from_secs(6), async { - while runtime.stats().hydration_jobs() != 0 { - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - assert_eq!(pausing.transient_get_failures.load(Ordering::Acquire), 0); - assert_eq!(runtime.stats().active_cells(), 1); - assert_eq!(read().await.unwrap(), 0_i64.to_be_bytes()); - tokio::time::timeout(std::time::Duration::from_secs(8), async { - loop { - if runtime - .resident_handle(&fixture.target, CatalogRole::Repository) - .await - .unwrap() - .is_some() - { - break; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn shutdown_releases_a_hydration_reservation_after_an_origin_wait() { - let (pausing, fixture) = released_large_cell_with_store( - b"hydration-shutdown-cancellation", - SessionId::from_bytes([113; 16]), - ) - .await; - - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let successor = SessionId::from_bytes([114; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 64 * 1024 * 1024, - successor, - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - observed, - cold_node_directory(&fixture).join("hydration-shutdown.sqlite"), - Owner { - session: successor, - endpoint: "https://hydration-shutdown.internal:8081".into(), - }, - ) - .await - .unwrap(); - - pausing.arm_gets(); - tokio::time::timeout( - std::time::Duration::from_secs(3), - pausing.wait_until_get_blocked(), - ) - .await - .unwrap(); - assert_eq!(runtime.stats().hydration_jobs(), 1); - - let shutdown_runtime = runtime.clone(); - let shutdown = tokio::spawn(async move { shutdown_runtime.shutdown().await }); - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - pausing.release_gets(); - shutdown.await.unwrap().unwrap(); - assert_eq!(runtime.stats().hydration_jobs(), 0); - drop(restored); -} - -#[tokio::test(flavor = "multi_thread")] -async fn origin_failure_during_hydration_fences_without_serving_unverified_pages() { - let (pausing, fixture) = released_large_cell_with_store( - b"hydration-origin-failure", - SessionId::from_bytes([145; 16]), - ) - .await; - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let root = observed.value().ltx_root().unwrap(); - let session = SessionId::from_bytes([146; 16]); - // The default host's disk budget is process-wide; isolate this Cell's - // reservation so concurrent tests cannot change its zero-use assertion. - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 1).unwrap(), - 64 * 1024 * 1024, - session, - ReplicaHost::default().with_local_disk_budget(DiskBudget::new(1 << 30)), - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority.clone(), - observed, - cold_node_directory(&fixture).join("hydration-origin-failure.sqlite"), - Owner { - session, - endpoint: "https://hydration-origin-failure.internal:8081".into(), - }, - ) - .await - .unwrap(); - - pausing.arm_gets(); - tokio::time::timeout( - std::time::Duration::from_secs(3), - pausing.wait_until_get_blocked(), - ) - .await - .expect("seed=145: hydration did not reach the origin"); - assert_eq!(runtime.stats().hydration_jobs(), 1); - pausing.fail_next_get(); - pausing.release_gets(); - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - if runtime.stats().active_cells() == 0 && runtime.stats().hydration_jobs() == 0 { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .expect("seed=145: failed hydration did not fence and release its reservation"); - assert!(!pausing.fail_next_get.load(Ordering::Acquire)); - assert!( - runtime - .resident_handle(&fixture.target, CatalogRole::Repository) - .await - .unwrap() - .is_none() - ); - assert!(matches!( - restored.query(64, 64, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert_eq!( - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root(), - Some(root) - ); - let verified = fixture.replica.open_root(&root).await.unwrap(); - let recovered = fixture - ._directory - .path() - .join("origin-failure-recovered.sqlite"); - assert_eq!(verified.restore(&recovered).await.unwrap(), root.position); - drop(restored); - runtime.shutdown().await.unwrap(); - assert_eq!(runtime.local_disk_budget().used(), 0); -} - -#[tokio::test(flavor = "multi_thread")] -async fn disk_exhaustion_during_hydration_fences_and_releases_capacity() { - const DISK_CAPACITY: u64 = 4 << 20; - let (_, fixture) = released_large_cell_with_store( - b"hydration-disk-exhaustion", - SessionId::from_bytes([147; 16]), - ) - .await; - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let root = observed.value().ltx_root().unwrap(); - let session = SessionId::from_bytes([148; 16]); - let runtime = CellRuntime::new_with_replica_host( - SqlWorkerPool::new(1, 1).unwrap(), - 64 * 1024 * 1024, - session, - ReplicaHost::default().with_local_disk_budget(DiskBudget::new(DISK_CAPACITY)), - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority.clone(), - observed, - cold_node_directory(&fixture).join("hydration-disk-exhaustion.sqlite"), - Owner { - session, - endpoint: "https://hydration-disk-exhaustion.internal:8081".into(), - }, - ) - .await - .unwrap(); - - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - if runtime.stats().active_cells() == 0 && runtime.stats().hydration_jobs() == 0 { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .expect("seed=147: disk-limited hydration did not fence"); - assert!( - runtime - .resident_handle(&fixture.target, CatalogRole::Repository) - .await - .unwrap() - .is_none() - ); - assert!(matches!( - restored.query(64, 64, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert_eq!( - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root(), - Some(root) - ); - let verified = fixture.replica.open_root(&root).await.unwrap(); - let recovered = fixture - ._directory - .path() - .join("disk-exhaustion-recovered.sqlite"); - assert_eq!(verified.restore(&recovered).await.unwrap(), root.position); - assert!(std::fs::metadata(&recovered).unwrap().len() > DISK_CAPACITY); - drop(restored); - runtime.shutdown().await.unwrap(); - assert_eq!(runtime.local_disk_budget().used(), 0); -} - -#[tokio::test(flavor = "multi_thread")] -async fn node_lease_loss_during_hydration_cannot_promote_a_stale_owner() { - let (pausing, fixture) = released_large_cell_with_store( - b"hydration-node-lease-loss", - SessionId::from_bytes([149; 16]), - ) - .await; - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let observed = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let root = observed.value().ltx_root().unwrap(); - let session = SessionId::from_bytes([150; 16]); - let runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 64 * 1024 * 1024, - session, - ReplicaHost::default().with_local_disk_budget(DiskBudget::new(1 << 30)), - ) - .unwrap(); - let lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - runtime.install_node_lease(lease.clone()).unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority.clone(), - observed, - cold_node_directory(&fixture).join("hydration-lease-loss.sqlite"), - Owner { - session, - endpoint: "https://hydration-lease-loss.internal:8081".into(), - }, - ) - .await - .unwrap(); - - pausing.arm_gets(); - tokio::time::timeout( - std::time::Duration::from_secs(3), - pausing.wait_until_get_blocked(), - ) - .await - .expect("seed=149: hydration did not reach the origin"); - assert_eq!(runtime.stats().hydration_jobs(), 1); - lease.fence(); - pausing.release_gets(); - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - if runtime.stats().active_cells() == 0 && runtime.stats().hydration_jobs() == 0 { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .expect("seed=149: lost node lease did not stop hydration and release the Cell"); - assert!(matches!( - runtime - .resident_handle(&fixture.target, CatalogRole::Repository) - .await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert!(matches!( - restored.query(64, 64, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert_eq!( - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .ltx_root(), - Some(root) - ); - let verified = fixture.replica.open_root(&root).await.unwrap(); - let recovered = fixture - ._directory - .path() - .join("lease-loss-recovered.sqlite"); - assert_eq!(verified.restore(&recovered).await.unwrap(), root.position); - drop(restored); - runtime.shutdown().await.unwrap(); - assert_eq!(runtime.local_disk_budget().used(), 0); -} - -#[tokio::test(flavor = "multi_thread")] -async fn successor_takeover_while_hydration_waits_keeps_the_old_owner_fenced() { - let (pausing, fixture) = released_large_cell_with_store( - b"hydration-concurrent-takeover", - SessionId::from_bytes([151; 16]), - ) - .await; - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let old_session = SessionId::from_bytes([152; 16]); - let old_runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 64 * 1024 * 1024, - old_session, - ReplicaHost::default().with_local_disk_budget(DiskBudget::new(1 << 30)), - ) - .unwrap(); - let old_lease = NodeLeaseGuard::new(0, 60_000).unwrap(); - old_runtime.install_node_lease(old_lease.clone()).unwrap(); - let old_handle = old_runtime - .acquire_idle_restored( - proof.clone(), - fixture.replica.clone(), - authority.clone(), - idle, - cold_node_directory(&fixture).join("hydrating-owner.sqlite"), - Owner { - session: old_session, - endpoint: "https://hydrating-owner.internal:8081".into(), - }, - ) - .await - .unwrap(); - pausing.arm_gets(); - tokio::time::timeout( - std::time::Duration::from_secs(3), - pausing.wait_until_get_blocked(), - ) - .await - .expect("seed=151: old owner hydration did not reach the origin"); - assert_eq!(old_runtime.stats().hydration_jobs(), 1); - - let old_control = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let root = old_control.value().ltx_root().unwrap(); - let successor_session = SessionId::from_bytes([153; 16]); - let fenced = fence_session(&fixture.layout, old_session, successor_session).await; - old_lease.fence(); - let successor_runtime = CellRuntime::new_with_replica_host_requiring_node_lease( - SqlWorkerPool::new(1, 1).unwrap(), - 64 * 1024 * 1024, - successor_session, - ReplicaHost::default().with_local_disk_budget(DiskBudget::new(1 << 30)), - ) - .unwrap(); - successor_runtime - .install_node_lease(NodeLeaseGuard::new(0, 60_000).unwrap()) - .unwrap(); - let successor_directory = fixture._directory.path().join("takeover-node"); - std::fs::create_dir_all(&successor_directory).unwrap(); - let successor = successor_runtime - .takeover_restored( - proof, - fixture.replica.clone(), - authority.clone(), - old_control.clone(), - fenced.direct_takeover().unwrap(), - crab_cell_runtime::recovery::manifest::RecoveryManifestStore::new( - fixture.layout.clone(), - Limits::default(), - ), - successor_directory.join("successor.sqlite"), - Owner { - session: successor_session, - endpoint: "https://hydration-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(current.value().ltx_root(), Some(root)); - assert!(current.value().epoch > old_control.value().epoch); - assert_eq!( - current.value().owner.as_ref().unwrap().session, - successor_session - ); - let payload = successor - .query(64, 64, |connection| { - let (rows, exact) = connection.query_row( - "SELECT COUNT(*), SUM(value = zeroblob(16384)) FROM payload", - [], - |row| Ok((row.get::<_, i64>(0)?, row.get::<_, i64>(1)?)), - )?; - Ok([rows.to_be_bytes(), exact.to_be_bytes()].concat()) - }) - .await - .unwrap(); - assert_eq!( - payload, - [512_i64.to_be_bytes(), 512_i64.to_be_bytes()].concat() - ); - assert!(!pausing.get_released.load(Ordering::Acquire)); - - pausing.release_gets(); - tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - if old_runtime.stats().active_cells() == 0 && old_runtime.stats().hydration_jobs() == 0 - { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .expect("seed=151: old hydration did not release after successor takeover"); - assert!(matches!( - old_handle.query(64, 64, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert_eq!( - authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap() - .value() - .owner - .as_ref() - .unwrap() - .session, - successor_session - ); - drop(old_handle); - old_runtime.shutdown().await.unwrap(); - assert_eq!(old_runtime.local_disk_budget().used(), 0); - successor.drain().await.unwrap(); - successor_runtime.shutdown().await.unwrap(); - assert_eq!(successor_runtime.local_disk_budget().used(), 0); -} - -#[tokio::test] -async fn retained_request_outcome_moves_with_exact_root() { - let fixture = fixture_for(b"eviction-persisted-work"); - let session = SessionId::from_bytes([78; 16]); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 8 * 1024 * 1024, session).unwrap(); - let handle = bootstrap_on(&runtime, &fixture, session).await; - assert!(matches!( - handle - .execute( - mutation_identity_window(79, 10, 10_000), - Digest::from_bytes([79; 32]), - 20, - 64, - 64, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(), - StoredOutcome::Success { - commit_sequence: 1, - .. - } - )); - - let generation = tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - if let Some((_, generation, _, _)) = - runtime.idle_transfer_candidates().await.unwrap().first() - { - break *generation; - } - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - }) - .await - .unwrap(); - runtime - .release_idle_cell(fixture.target.cell_id(), session, generation) - .await - .unwrap(); - assert_eq!(runtime.stats().active_cells(), 0); - let authority = CellAuthority::new(fixture.layout.clone()); - let idle = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - assert_eq!(idle.value().state, ControlState::Idle); - let catalog = crab_cell_runtime::cell::catalog::CellCatalog::new( - fixture.layout.clone(), - fixture.target.tenant(), - ); - let proof = catalog - .lookup(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let successor_session = SessionId::from_bytes([95; 16]); - let successor_runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 8 * 1024 * 1024, - successor_session, - ) - .unwrap(); - let successor = successor_runtime - .acquire_idle_restored( - proof, - fixture.replica.clone(), - authority, - idle, - fixture._directory.path().join("outcome-successor.sqlite"), - Owner { - session: successor_session, - endpoint: "https://outcome-successor.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert!(matches!( - successor - .resolve( - mutation_identity_window(79, 10, 10_000), - Digest::from_bytes([79; 32]), - 20, - 64 - ) - .await - .unwrap(), - Resolution::Committed(StoredOutcome::Success { - commit_sequence: 1, - .. - }) - )); - successor.drain().await.unwrap(); - successor_runtime.shutdown().await.unwrap(); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn idle_owner_progress_is_renewed_without_a_per_cell_task() { - let fixture = fixture(); - let handle = activate(&fixture, 16 * 1024 * 1024).await; - let authority = CellAuthority::new(fixture.layout.clone()); - let initial = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - let initial_progress = initial.value().progress; - let initial_root = initial.value().root.clone(); - - let renewed = tokio::time::timeout(std::time::Duration::from_secs(5), async { - loop { - let current = authority - .load(fixture.target.cell_id()) - .await - .unwrap() - .unwrap(); - if current.value().progress > initial_progress { - break current; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - }) - .await - .unwrap(); - - assert_eq!(renewed.value().root, initial_root); - assert_eq!(renewed.value().owner, initial.value().owner); - assert_eq!(renewed.value().revision, initial.value().revision + 1); - handle.drain().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/migration.rs b/crates/crab-cell-runtime/tests/runtime/migration.rs deleted file mode 100644 index 42fe4fb9f..000000000 --- a/crates/crab-cell-runtime/tests/runtime/migration.rs +++ /dev/null @@ -1,304 +0,0 @@ -use std::{ - fmt, - future::Future, - pin::Pin, - sync::{ - Arc, OnceLock, - atomic::{AtomicBool, Ordering}, - }, -}; - -use crab_cell_runtime::cell::actor::CellRuntime; -use crab_cell_runtime::cell::catalog::CatalogRole; -use crab_cell_runtime::cell::catalog::{CatalogEntry, CellCatalog}; -use crab_cell_runtime::cell::executor::HandlerOutcome; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::control::authority::CellAuthority; -use crab_cell_runtime::control::{ControlState, Owner}; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::identity::{IncarnationId, NodeId}; -use crab_cell_runtime::node::durability::{NodeDurability, NodeLogAuthority}; -use crab_cell_runtime::node::lease::NodeLeaseGuard; -use crab_cell_runtime::node::log::{DurabilityGate, NodeLogRotationBarrier}; -use crab_cell_runtime::node::log_shipper::NodeLogShipper; -use crab_cell_runtime::node::log_transport::{LocalFollowerTransport, NodeLogTransport}; -use crab_cell_runtime::peer::{ - MigrationPeerClient, PeerAuthorizer, PeerCellResolver, PeerDispatcher, PeerPrincipal, - PeerRoundTrip, PeerSigner, PeerVerifier, VerifiedPeerRequest, -}; -use crab_cell_runtime::registry::{ - BuildDescriptor, CellModule, ModuleDescriptor, NamespaceDescriptor, Registry, RegistryBuilder, -}; -use crab_cell_runtime::registry::{MigrationDescriptor, RetainedCodeDescriptor}; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Limits}; -use crab_storage::Store; -use futures_util::stream::BoxStream; -use object_store::{ - CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, - PutMultipartOptions, PutOptions, PutPayload, PutResult, memory::InMemory, path::Path, -}; - -const MODULE: &str = "migration-test"; -const NAMESPACE: NamespaceId = NamespaceId::from_bytes([61; 16]); -const MIGRATION_ONE: &str = - "CREATE TABLE records(id INTEGER PRIMARY KEY, value BLOB NOT NULL) STRICT"; -const MIGRATION_TWO: &str = "ALTER TABLE records ADD COLUMN label TEXT"; -const PREDECESSOR_CODE: Digest = Digest::from_bytes([60; 32]); - -struct MigrationModule; - -struct MigrationNodeAuthority; - -impl NodeLogAuthority for MigrationNodeAuthority { - fn activate<'a>( - &'a self, - _log_epoch: u64, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result<()>> { - Box::pin(async { Ok(()) }) - } - - fn advance_coverage<'a>( - &'a self, - _log_epoch: u64, - _tiered_through: u64, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result<()>> { - Box::pin(async { Ok(()) }) - } - - fn close<'a>( - &'a self, - _barrier: &'a NodeLogRotationBarrier, - ) -> futures_util::future::BoxFuture<'a, crab_cell_runtime::Result<()>> { - Box::pin(async { Ok(()) }) - } -} - -#[derive(Debug)] -struct PausedPutStore { - inner: Arc, - pause_next: AtomicBool, - started: Arc, - release: Arc, -} - -impl PausedPutStore { - fn new() -> Self { - Self { - inner: Arc::new(InMemory::new()), - pause_next: AtomicBool::new(false), - started: Arc::new(tokio::sync::Barrier::new(2)), - release: Arc::new(tokio::sync::Barrier::new(2)), - } - } - - fn pause_next(&self) { - self.pause_next.store(true, Ordering::Release); - } -} - -impl fmt::Display for PausedPutStore { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter.write_str("paused-put-store") - } -} - -#[async_trait::async_trait] -impl ObjectStore for PausedPutStore { - async fn put_opts( - &self, - location: &Path, - payload: PutPayload, - options: PutOptions, - ) -> object_store::Result { - if self.pause_next.swap(false, Ordering::AcqRel) { - self.started.wait().await; - self.release.wait().await; - } - self.inner.put_opts(location, payload, options).await - } - - async fn put_multipart_opts( - &self, - location: &Path, - options: PutMultipartOptions, - ) -> object_store::Result> { - self.inner.put_multipart_opts(location, options).await - } - - async fn get_opts( - &self, - location: &Path, - options: GetOptions, - ) -> object_store::Result { - self.inner.get_opts(location, options).await - } - - fn delete_stream( - &self, - locations: BoxStream<'static, object_store::Result>, - ) -> BoxStream<'static, object_store::Result> { - self.inner.delete_stream(locations) - } - - fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { - self.inner.list(prefix) - } - - async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { - self.inner.list_with_delimiter(prefix).await - } - - async fn copy_opts( - &self, - from: &Path, - to: &Path, - options: CopyOptions, - ) -> object_store::Result<()> { - self.inner.copy_opts(from, to, options).await - } -} - -impl CellModule for MigrationModule { - const NAME: &'static str = MODULE; - - fn descriptor(&self) -> &'static ModuleDescriptor { - descriptor() - } - - fn register(self, _registry: &mut RegistryBuilder) -> crab_cell_runtime::Result<()> { - Ok(()) - } -} - -fn descriptor() -> &'static ModuleDescriptor { - static DESCRIPTOR: OnceLock = OnceLock::new(); - DESCRIPTOR.get_or_init(|| ModuleDescriptor { - name: MODULE, - source_digest: Digest::from_bytes([62; 32]), - retained_codes: &[RetainedCodeDescriptor { - code: PREDECESSOR_CODE, - schema_min: 1, - schema_max: 2, - }], - schema_min: 1, - schema_max: 2, - migrations: Box::leak(Box::new([ - MigrationDescriptor { - version: 1, - sql: MIGRATION_ONE, - digest: Digest::from_bytes(*blake3::hash(MIGRATION_ONE.as_bytes()).as_bytes()), - }, - MigrationDescriptor { - version: 2, - sql: MIGRATION_TWO, - digest: Digest::from_bytes(*blake3::hash(MIGRATION_TWO.as_bytes()).as_bytes()), - }, - ])), - commands: &[], - queries: &[], - workflow_definitions: &[], - activity_types: &[], - namespaces: Box::leak(Box::new([NamespaceDescriptor { - id: NAMESPACE, - name: MODULE, - role: CatalogRole::Sql, - shards: 1, - effect_targets: &[], - dead_letter: None, - }])), - }) -} - -fn compiled_registry() -> Registry { - let mut registry = RegistryBuilder::new(BuildDescriptor { - source_revision: "migration-test".into(), - cargo_lock_digest: Digest::from_bytes([63; 32]), - }); - registry.register(MigrationModule).unwrap(); - registry.finish().unwrap() -} - -struct RuntimeResolver { - target: CellTarget, - proof: crab_cell_runtime::cell::catalog::CatalogProof, - authority: CellAuthority, - runtime: CellRuntime, -} - -impl PeerCellResolver for RuntimeResolver { - fn resolve( - &self, - target: CellTarget, - ) -> Pin< - Box< - dyn Future< - Output = crab_cell_runtime::Result, - > + Send - + 'static, - >, - > { - let matches = target == self.target; - let proof = self.proof.clone(); - let authority = self.authority.clone(); - let runtime = self.runtime.clone(); - Box::pin(async move { - if !matches { - return Err(crab_cell_runtime::Error::CellNotActive); - } - let control = authority - .load(proof.entry().cell()) - .await? - .ok_or(crab_cell_runtime::Error::CellNotActive)?; - runtime - .local_handle(proof, &control) - .await? - .ok_or(crab_cell_runtime::Error::CellNotActive) - }) - } -} - -struct MigrationAuthorizer; - -impl PeerAuthorizer for MigrationAuthorizer { - fn authorize(&self, request: &VerifiedPeerRequest) -> crab_cell_runtime::Result<()> { - if request.permits("cell.release.migrate") { - Ok(()) - } else { - Err(crab_cell_runtime::Error::PeerAuthorization( - "missing release migration action", - )) - } - } -} - -struct LoopbackRoundTrip { - verifier: Arc, - dispatcher: Arc, -} - -impl PeerRoundTrip for LoopbackRoundTrip { - fn send( - &self, - target: CellTarget, - request: Vec, - _remaining_ms: u32, - ) -> Pin>> + Send + 'static>> { - let verifier = Arc::clone(&self.verifier); - let dispatcher = Arc::clone(&self.dispatcher); - Box::pin(async move { - let verified = verifier.verify(&request, 10)?; - if verified.target() != &target { - return Err(crab_cell_runtime::Error::Peer( - "loopback migration target changed", - )); - } - dispatcher.dispatch_bytes(&verified, 10).await - }) - } -} - -mod peer; -mod schema; diff --git a/crates/crab-cell-runtime/tests/runtime/migration/peer.rs b/crates/crab-cell-runtime/tests/runtime/migration/peer.rs deleted file mode 100644 index f57ab2a25..000000000 --- a/crates/crab-cell-runtime/tests/runtime/migration/peer.rs +++ /dev/null @@ -1,229 +0,0 @@ -//! Authenticated peer migration plans and digest conflicts. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn authenticated_peer_migration_derives_plan_and_reconciles_retry() { - let registry = Arc::new(compiled_registry()); - let target = CellTarget::new( - TenantId::from_bytes([80; 16]), - ApplicationId::from_bytes([81; 16]), - NAMESPACE, - b"peer-code-only", - ) - .unwrap(); - let incarnation = IncarnationId::from_bytes([82; 16]); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("peer-code-only-migration"), - [81; 16], - ); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision(CatalogEntry::new(&target, CatalogRole::Sql, PREDECESSOR_CODE, 2).unwrap()) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let owner_session = SessionId::from_bytes([83; 16]); - let initial = authority - .create_initial( - &proof, - incarnation, - Owner { - session: owner_session, - endpoint: "https://peer-migration.internal:8081".into(), - }, - ) - .await - .unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 2).unwrap(), - 16 * 1024 * 1024, - owner_session, - ) - .unwrap(); - let files = tempfile::TempDir::new().unwrap(); - let replica = CellReplica::new( - layout, - *target.cell_id().as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof.clone(), - replica, - authority.clone(), - initial, - files.path().join("peer-code-only.sqlite"), - |transaction| { - transaction.execute_batch(MIGRATION_ONE)?; - transaction.execute_batch(MIGRATION_TWO)?; - Ok(()) - }, - ) - .await - .unwrap(); - let expected = crab_cell_runtime::client::CellDescription { - cell: target.cell_id(), - incarnation, - code: handle.code(), - schema: handle.schema(), - }; - let plan = registry - .next_migration(target.namespace(), expected.code, expected.schema) - .unwrap() - .unwrap(); - let signing_session = SessionId::from_bytes([84; 16]); - let signer = PeerSigner::new( - signing_session, - registry.release_digest(), - ed25519_dalek::SigningKey::from_bytes(&[85; 32]), - ); - let verifier = Arc::new(PeerVerifier::new( - signing_session, - registry.release_digest(), - signer.verifying_key(), - )); - let dispatcher = Arc::new(PeerDispatcher::new( - Arc::clone(®istry), - Arc::new(RuntimeResolver { - target: target.clone(), - proof: proof.clone(), - authority: authority.clone(), - runtime: runtime.clone(), - }), - Arc::new(MigrationAuthorizer), - )); - let client = MigrationPeerClient::new( - Arc::new(signer), - PeerPrincipal { - issuer: "crab-runtime:test".into(), - subject: "release-operator".into(), - actions: vec!["cell.release.migrate".into()], - }, - Arc::new(LoopbackRoundTrip { - verifier, - dispatcher, - }), - ); - - let migrated = client - .migrate(target.clone(), expected, plan, 10) - .await - .unwrap(); - assert_eq!(migrated.code, registry.module_code(MODULE).unwrap()); - assert_eq!(migrated.schema, 2); - assert_eq!( - client.migrate(target, expected, plan, 10).await.unwrap(), - migrated - ); - - let control = authority.load(proof.entry().cell()).await.unwrap().unwrap(); - runtime - .local_handle(proof, &control) - .await - .unwrap() - .unwrap() - .drain() - .await - .unwrap(); - runtime.shutdown().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn migration_digest_conflict_blocks_activation() { - let registry = compiled_registry(); - let target = CellTarget::new( - TenantId::from_bytes([71; 16]), - ApplicationId::from_bytes([72; 16]), - NAMESPACE, - b"migration-conflict", - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([73; 16]); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("migration-conflict"), - [72; 16], - ); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision(CatalogEntry::new(&target, CatalogRole::Sql, PREDECESSOR_CODE, 1).unwrap()) - .await - .unwrap(); - let authority = CellAuthority::new(layout); - let session = SessionId::from_bytes([74; 16]); - let initial = authority - .create_initial( - &proof, - incarnation, - Owner { - session, - endpoint: "https://migration-conflict.internal:8081".into(), - }, - ) - .await - .unwrap(); - let files = tempfile::TempDir::new().unwrap(); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 1).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let handle = runtime - .bootstrap( - proof, - replica, - authority.clone(), - initial, - files.path().join("conflict.sqlite"), - |transaction| { - transaction.execute_batch(MIGRATION_ONE)?; - transaction.execute( - "INSERT INTO sys_migrations(version, digest, applied_sequence) VALUES (2, ?1, 0)", - [[99_u8; 32].as_slice()], - )?; - Ok(()) - }, - ) - .await - .unwrap(); - let before = authority.load(cell).await.unwrap().unwrap(); - let plan = registry - .next_migration(NAMESPACE, handle.code(), handle.schema()) - .unwrap() - .unwrap(); - assert!(matches!( - handle.migrate(plan, 10).await, - Err(crab_cell_runtime::Error::Registry( - "migration digest conflicts with SQLite history" - )) - )); - let after = tokio::time::timeout(std::time::Duration::from_secs(2), async { - loop { - let current = authority.load(cell).await.unwrap().unwrap(); - if current.value().state == ControlState::Idle { - break current; - } - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - }) - .await - .unwrap(); - assert_eq!(after.value().incarnation, before.value().incarnation); - assert_eq!(after.value().code, before.value().code); - assert_eq!(after.value().schema, before.value().schema); - assert_eq!(after.value().root, before.value().root); - assert!(after.value().owner.is_none()); - assert!(matches!( - handle.query(1, 1, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::CellDraining) | Err(crab_cell_runtime::Error::Fenced) - )); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/migration/schema.rs b/crates/crab-cell-runtime/tests/runtime/migration/schema.rs deleted file mode 100644 index 3ba256f78..000000000 --- a/crates/crab-cell-runtime/tests/runtime/migration/schema.rs +++ /dev/null @@ -1,362 +0,0 @@ -//! Capability and code-only migration publication with exact-root restore. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn migration_replaces_capability_publishes_schema_and_restores_exact_root() { - let registry = compiled_registry(); - let code = registry.module_code(MODULE).unwrap(); - assert!(registry.supports_cell(NAMESPACE, CatalogRole::Sql, PREDECESSOR_CODE, 1)); - assert!(registry.supports_cell(NAMESPACE, CatalogRole::Sql, PREDECESSOR_CODE, 2)); - assert!(!registry.is_current_cell(NAMESPACE, CatalogRole::Sql, PREDECESSOR_CODE, 2)); - assert!(registry.is_current_cell(NAMESPACE, CatalogRole::Sql, code, 2)); - assert!(registry.module_digests().contains(&PREDECESSOR_CODE)); - assert!(registry.module_digests().contains(&code)); - let target = CellTarget::new( - TenantId::from_bytes([64; 16]), - ApplicationId::from_bytes([65; 16]), - NAMESPACE, - b"schema-step", - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([66; 16]); - let object_store = Arc::new(PausedPutStore::new()); - let layout = CellStorageLayout::new( - Store::new(object_store.clone()), - Path::from("migration-runtime"), - [65; 16], - ); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision(CatalogEntry::new(&target, CatalogRole::Sql, PREDECESSOR_CODE, 1).unwrap()) - .await - .unwrap(); - let authority = CellAuthority::new(layout); - let first_session = SessionId::from_bytes([67; 16]); - let control = authority - .create_initial( - &proof, - incarnation, - Owner { - session: first_session, - endpoint: "https://migration-first.internal:8081".into(), - }, - ) - .await - .unwrap(); - let files = tempfile::TempDir::new().unwrap(); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 2).unwrap(), - 16 * 1024 * 1024, - first_session, - ) - .unwrap(); - let handle = runtime - .bootstrap( - proof.clone(), - replica.clone(), - authority.clone(), - control, - files.path().join("schema-one.sqlite"), - |transaction| { - transaction.execute_batch(MIGRATION_ONE)?; - Ok(()) - }, - ) - .await - .unwrap(); - let follower_directory = tempfile::TempDir::new().unwrap(); - let follower = NodeId::from_bytes([84; 16]); - let follower_store = crab_cell_runtime::FollowerStore::open( - follower_directory.path().to_owned(), - Limits::default(), - crab_cell_runtime::ltx::DiskBudget::new(1 << 30), - ) - .unwrap(); - let transport: Arc = - Arc::new(LocalFollowerTransport::new(follower, follower_store)); - let gate = - DurabilityGate::new(first_session, NodeId::from_bytes([85; 16]), 1, [follower]).unwrap(); - let shipper = - NodeLogShipper::new(gate.clone(), Arc::clone(&transport), Limits::default()).unwrap(); - let durability = Arc::new(NodeDurability::new( - gate, - shipper, - Arc::new(MigrationNodeAuthority), - transport, - NodeLeaseGuard::new(0, 60_000).unwrap(), - )); - runtime - .install_node_durability(target.application(), durability) - .unwrap(); - let old_handle = handle.clone(); - let plan = registry - .next_migration(NAMESPACE, handle.code(), handle.schema()) - .unwrap() - .unwrap(); - assert_eq!(plan.from_code(), PREDECESSOR_CODE); - assert_eq!(plan.to_code(), code); - let migration_digest = plan.digest().unwrap(); - object_store.pause_next(); - let migrated = - tokio::time::timeout(std::time::Duration::from_secs(5), handle.migrate(plan, 10)) - .await - .unwrap() - .unwrap(); - object_store.started.wait().await; - let queued_handle = migrated.handle.clone(); - let mut queued = tokio::spawn(async move { - queued_handle - .query(1, 8, |connection| { - Ok(connection - .query_row("SELECT schema_version FROM sys_meta", [], |row| { - row.get::<_, u32>(0) - })? - .to_be_bytes() - .to_vec()) - }) - .await - }); - assert!( - tokio::time::timeout(std::time::Duration::from_millis(50), &mut queued) - .await - .is_err() - ); - object_store.release.wait().await; - assert_eq!( - tokio::time::timeout(std::time::Duration::from_secs(5), queued) - .await - .unwrap() - .unwrap() - .unwrap(), - 2_u32.to_be_bytes() - ); - assert_eq!(migrated.outcome.code, code); - assert_eq!(migrated.outcome.schema, 2); - assert_eq!(migrated.outcome.commit_sequence, 1); - assert!(matches!( - old_handle.query(1, 1, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::CellDraining) - )); - assert!( - registry - .next_migration(NAMESPACE, migrated.handle.code(), migrated.handle.schema()) - .unwrap() - .is_none() - ); - let code_only = registry - .next_migration(NAMESPACE, PREDECESSOR_CODE, 2) - .unwrap() - .unwrap(); - assert_eq!(code_only.from_schema(), code_only.to_schema()); - assert_eq!(code_only.from_code(), PREDECESSOR_CODE); - assert_eq!(code_only.to_code(), code); - assert!(code_only.sql().is_none()); - let metadata = migrated - .handle - .query(64, 128, |connection| { - let schema = connection.query_row( - "SELECT schema_version FROM sys_meta WHERE singleton = 1", - [], - |row| row.get::<_, u32>(0), - )?; - let (digest, sequence) = connection.query_row( - "SELECT digest, applied_sequence FROM sys_migrations WHERE version = 2", - [], - |row| Ok((row.get::<_, Vec>(0)?, row.get::<_, u64>(1)?)), - )?; - let mut result = schema.to_be_bytes().to_vec(); - result.extend_from_slice(&sequence.to_be_bytes()); - result.extend_from_slice(&digest); - Ok(result) - }) - .await - .unwrap(); - assert_eq!(&metadata[..4], &2_u32.to_be_bytes()); - assert_eq!(&metadata[4..12], &1_u64.to_be_bytes()); - assert_eq!(&metadata[12..], migration_digest.as_bytes()); - let written = migrated - .handle - .execute( - crab_cell_runtime::cell::executor::MutationIdentity { - request_id: crab_cell_runtime::identity::RequestId::from_bytes([68; 16]), - issued_at_ms: 10, - expires_at_ms: 1_000, - }, - Digest::from_bytes([69; 32]), - 20, - 128, - 64, - |transaction| { - transaction.execute( - "INSERT INTO records(id, value, label) VALUES (1, x'01', 'migrated')", - [], - )?; - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - .unwrap(); - assert_eq!(written.commit_sequence(), 2); - migrated.handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); - - let idle = authority.load(cell).await.unwrap().unwrap(); - assert_eq!(idle.value().schema, 2); - assert_eq!(idle.value().code, code); - assert_eq!(idle.value().root.as_ref().unwrap().commit_sequence, 2); - let second_session = SessionId::from_bytes([70; 16]); - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 2).unwrap(), - 16 * 1024 * 1024, - second_session, - ) - .unwrap(); - let restored = runtime - .acquire_idle_restored( - proof, - replica, - authority, - idle, - files.path().join("schema-two.sqlite"), - Owner { - session: second_session, - endpoint: "https://migration-second.internal:8081".into(), - }, - ) - .await - .unwrap(); - assert_eq!(restored.schema(), 2); - assert_eq!( - restored - .query(64, 64, |connection| { - Ok( - connection.query_row("SELECT label FROM records WHERE id = 1", [], |row| { - row.get::<_, String>(0).map(String::into_bytes) - })?, - ) - }) - .await - .unwrap(), - b"migrated" - ); - restored.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn code_only_migration_publishes_new_code_without_schema_ledger_entry() { - let registry = compiled_registry(); - let code = registry.module_code(MODULE).unwrap(); - let target = CellTarget::new( - TenantId::from_bytes([75; 16]), - ApplicationId::from_bytes([76; 16]), - NAMESPACE, - b"code-only", - ) - .unwrap(); - let cell = target.cell_id(); - let incarnation = IncarnationId::from_bytes([77; 16]); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("code-only-migration-runtime"), - [76; 16], - ); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let catalog = CellCatalog::new(layout.clone(), target.tenant()); - let proof = catalog - .provision(CatalogEntry::new(&target, CatalogRole::Sql, PREDECESSOR_CODE, 2).unwrap()) - .await - .unwrap(); - let authority = CellAuthority::new(layout); - let session = SessionId::from_bytes([78; 16]); - let control = authority - .create_initial( - &proof, - incarnation, - Owner { - session, - endpoint: "https://code-only.internal:8081".into(), - }, - ) - .await - .unwrap(); - let files = tempfile::TempDir::new().unwrap(); - let runtime = - CellRuntime::new(SqlWorkerPool::new(1, 2).unwrap(), 16 * 1024 * 1024, session).unwrap(); - let handle = runtime - .bootstrap( - proof, - replica, - authority.clone(), - control, - files.path().join("code-only.sqlite"), - |transaction| { - transaction.execute_batch(MIGRATION_ONE)?; - transaction.execute_batch(MIGRATION_TWO)?; - Ok(()) - }, - ) - .await - .unwrap(); - let old_handle = handle.clone(); - let plan = registry - .next_migration(NAMESPACE, handle.code(), handle.schema()) - .unwrap() - .unwrap(); - assert_eq!(plan.from_code(), PREDECESSOR_CODE); - assert_eq!(plan.to_code(), code); - assert_eq!(plan.from_schema(), 2); - assert_eq!(plan.to_schema(), 2); - assert_eq!(plan.digest(), None); - assert_eq!(plan.sql(), None); - - let migrated = handle.migrate(plan, 10).await.unwrap(); - assert_eq!(migrated.outcome.code, code); - assert_eq!(migrated.outcome.schema, 2); - assert_eq!(migrated.outcome.commit_sequence, 1); - assert!(matches!( - old_handle.query(1, 1, |_| Ok(Vec::new())).await, - Err(crab_cell_runtime::Error::CellDraining) - )); - let metadata = migrated - .handle - .query(64, 64, |connection| { - let schema = connection.query_row( - "SELECT schema_version FROM sys_meta WHERE singleton = 1", - [], - |row| row.get::<_, u32>(0), - )?; - let migrations = - connection.query_row("SELECT COUNT(*) FROM sys_migrations", [], |row| { - row.get::<_, u32>(0) - })?; - let mut result = schema.to_be_bytes().to_vec(); - result.extend_from_slice(&migrations.to_be_bytes()); - Ok(result) - }) - .await - .unwrap(); - assert_eq!(&metadata[..4], &2_u32.to_be_bytes()); - assert_eq!(&metadata[4..], &0_u32.to_be_bytes()); - let published = authority.load(cell).await.unwrap().unwrap(); - assert_eq!(published.value().code, code); - assert_eq!(published.value().schema, 2); - assert_eq!(published.value().root.as_ref().unwrap().commit_sequence, 1); - - migrated.handle.drain().await.unwrap(); - runtime.shutdown().await.unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/publication.rs b/crates/crab-cell-runtime/tests/runtime/publication.rs deleted file mode 100644 index 173da108a..000000000 --- a/crates/crab-cell-runtime/tests/runtime/publication.rs +++ /dev/null @@ -1,694 +0,0 @@ -use std::{ - fmt, - sync::{ - Arc, - atomic::{AtomicUsize, Ordering}, - }, -}; - -use bytes::Bytes; -use crab_cell_runtime::cell::executor::{ - CellExecutor, CommandExecution, HandlerOutcome, MutationIdentity, StoredOutcome, -}; -use crab_cell_runtime::cell::schema::install_runtime_schema; -use crab_cell_runtime::control::authority::{CellAuthority, VersionedControl}; -use crab_cell_runtime::control::{Control, ControlState, Owner, Transition}; -use crab_cell_runtime::identity::{CellId, Digest, SessionId}; -use crab_cell_runtime::identity::{IncarnationId, RequestId}; -use crab_cell_runtime::publication::CellPublisher; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Db, Host, Limits}; -use crab_storage::Store; -use futures_util::stream::BoxStream; -use object_store::{ - CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, - PutMode, PutMultipartOptions, PutOptions, PutPayload, PutResult, memory::InMemory, path::Path, -}; - -use crate::runtime::fault_fs::FaultFileSystem; - -const RESULT_LIMIT: usize = 1 << 20; - -struct Fixture { - _directory: tempfile::TempDir, - database: std::path::PathBuf, - cell: CellId, - incarnation: IncarnationId, - layout: CellStorageLayout, - replica: CellReplica, - executor: CellExecutor, -} - -fn fixture() -> Fixture { - fixture_with_store(Store::new(Arc::new(InMemory::new()))) -} - -fn fixture_with_store(store: Store) -> Fixture { - fixture_with_store_and_host(store, Host::default()) -} - -fn fixture_with_store_and_host(store: Store, host: Host) -> Fixture { - let cell = CellId::from_bytes([1; 32]); - let incarnation = IncarnationId::from_bytes([2; 16]); - let layout = CellStorageLayout::new(store, Path::from("runtime"), [3; 16]); - let replica = CellReplica::new( - layout.clone(), - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let database = directory.path().join("cell.sqlite"); - let mut connection = crab_ltx::rusqlite::Connection::open(&database).unwrap(); - install_runtime_schema(&mut connection, cell, incarnation, 1).unwrap(); - connection - .execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (0)", - ) - .unwrap(); - drop(connection); - let writer = Db::open_with_host(&database, Limits::default(), host).unwrap(); - Fixture { - _directory: directory, - database, - cell, - incarnation, - layout, - replica, - executor: CellExecutor::new(writer, cell, incarnation, 1), - } -} - -#[derive(Debug)] -struct LostUpdateResponseStore { - inner: Arc, - remaining_failures: AtomicUsize, - updates: AtomicUsize, -} - -impl LostUpdateResponseStore { - fn new(inner: Arc) -> Self { - Self { - inner, - remaining_failures: AtomicUsize::new(1), - updates: AtomicUsize::new(0), - } - } -} - -impl fmt::Display for LostUpdateResponseStore { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter.write_str("lost-update-response-store") - } -} - -#[async_trait::async_trait] -impl ObjectStore for LostUpdateResponseStore { - async fn put_opts( - &self, - location: &Path, - payload: PutPayload, - options: PutOptions, - ) -> object_store::Result { - let update = matches!(&options.mode, PutMode::Update(_)); - let result = self.inner.put_opts(location, payload, options).await?; - if !update { - return Ok(result); - } - self.updates.fetch_add(1, Ordering::SeqCst); - if self - .remaining_failures - .fetch_update(Ordering::SeqCst, Ordering::SeqCst, |remaining| { - remaining.checked_sub(1) - }) - .is_ok() - { - return Err(object_store::Error::Generic { - store: "lost-update-response-store", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::ConnectionReset, - "control CAS response lost after commit", - )), - }); - } - Ok(result) - } - - async fn put_multipart_opts( - &self, - location: &Path, - options: PutMultipartOptions, - ) -> object_store::Result> { - self.inner.put_multipart_opts(location, options).await - } - - async fn get_opts( - &self, - location: &Path, - options: GetOptions, - ) -> object_store::Result { - self.inner.get_opts(location, options).await - } - - fn delete_stream( - &self, - locations: BoxStream<'static, object_store::Result>, - ) -> BoxStream<'static, object_store::Result> { - self.inner.delete_stream(locations) - } - - fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { - self.inner.list(prefix) - } - - async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { - self.inner.list_with_delimiter(prefix).await - } - - async fn copy_opts( - &self, - from: &Path, - to: &Path, - options: CopyOptions, - ) -> object_store::Result<()> { - self.inner.copy_opts(from, to, options).await - } -} - -async fn initialized_authority( - layout: &CellStorageLayout, - cell: CellId, - incarnation: IncarnationId, -) -> (Control, CellAuthority, VersionedControl) { - let control = Control::initial( - cell, - incarnation, - Owner { - session: SessionId::from_bytes([4; 16]), - endpoint: "https://node.internal:8081".into(), - }, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(); - layout - .store() - .create_strict( - &layout.control_path(cell.as_bytes()), - Bytes::from(control.encode().unwrap()), - ) - .await - .unwrap(); - let authority = CellAuthority::new(layout.clone()); - let observed = authority.load(cell).await.unwrap().unwrap(); - (control, authority, observed) -} - -#[tokio::test] -async fn prepared_root_becomes_one_valid_control_successor() { - let Fixture { - _directory, - database: _, - cell, - incarnation, - layout: _, - replica, - mut executor, - } = fixture(); - let identity = MutationIdentity { - request_id: RequestId::from_bytes([6; 16]), - issued_at_ms: 10, - expires_at_ms: 10_000, - }; - let operation_digest = Digest::from_bytes([7; 32]); - let calls = Arc::new(AtomicUsize::new(0)); - let observed = calls.clone(); - assert_eq!( - executor - .execute( - identity, - operation_digest, - 20, - RESULT_LIMIT, - move |transaction| { - observed.fetch_add(1, Ordering::SeqCst); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"one".to_vec())) - } - ) - .unwrap(), - CommandExecution::Pending - ); - assert!(matches!( - executor.execute(identity, operation_digest, 20, RESULT_LIMIT, |_| { - Ok(HandlerOutcome::Success(Vec::new())) - }), - Err(crab_cell_runtime::Error::PendingPublication) - )); - let pending = executor.pending().unwrap(); - assert_eq!(pending.cuts().timing.fsync_nanos, 0); - let prepared = replica - .prepare(None, pending.cuts(), pending.outcome().commit_sequence(), 1) - .await - .unwrap(); - let control = Control::initial( - cell, - incarnation, - Owner { - session: SessionId::from_bytes([4; 16]), - endpoint: "https://node.internal:8081".into(), - }, - Digest::from_bytes([5; 32]), - 1, - ) - .unwrap(); - let published = control.publish_prepared(&prepared, Some(42)).unwrap(); - assert_eq!(published.ltx_root(), Some(prepared.root())); - assert_eq!(published.next_due_ms, Some(42)); - assert_eq!(published.revision, 2); - executor.bind_prepared(&prepared).unwrap(); - let wrong_root = crab_ltx::RootRef { - digest: [11; 32], - ..prepared.root() - }; - assert!(executor.confirm_published(&wrong_root).is_err()); - assert_eq!( - executor.pending().unwrap().prepared(), - Some(prepared.root()) - ); - assert_eq!( - executor.confirm_published(&prepared.root()).unwrap(), - StoredOutcome::Success { - result: b"one".to_vec(), - commit_sequence: 1, - } - ); - assert!(matches!( - executor - .execute(identity, operation_digest, 21, RESULT_LIMIT, |_| { - calls.fetch_add(1, Ordering::SeqCst); - Ok(HandlerOutcome::Success(b"two".to_vec())) - }) - .unwrap(), - CommandExecution::Recorded(StoredOutcome::Success { ref result, commit_sequence: 1 }) - if result == b"one" - )); - assert_eq!(calls.load(Ordering::SeqCst), 1); - - let compacted = replica - .prepare_compaction(&prepared.root(), 0..1, 9, _directory.path()) - .await - .unwrap(); - let compacted_control = published.publish_prepared(&compacted, Some(42)).unwrap(); - assert_eq!(compacted.root().position, prepared.root().position); - assert_eq!( - compacted.root().commit_sequence, - prepared.root().commit_sequence - ); - assert_ne!(compacted.root().digest, prepared.root().digest); - assert_eq!(compacted_control.ltx_root(), Some(compacted.root())); - executor.close().unwrap(); -} - -#[tokio::test] -async fn business_rejection_rolls_back_domain_writes_and_publishes_the_outcome() { - let Fixture { - _directory, - database, - cell: _, - incarnation: _, - layout: _, - replica, - mut executor, - } = fixture(); - let identity = MutationIdentity { - request_id: RequestId::from_bytes([8; 16]), - issued_at_ms: 100, - expires_at_ms: 20_000, - }; - let digest = Digest::from_bytes([9; 32]); - assert_eq!( - executor - .execute(identity, digest, 110, RESULT_LIMIT, |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Rejected(b"insufficient quota".to_vec())) - }) - .unwrap(), - CommandExecution::Pending - ); - let pending = executor.pending().unwrap(); - assert!(matches!(pending.outcome(), StoredOutcome::Rejected { .. })); - let prepared = replica - .prepare(None, pending.cuts(), pending.outcome().commit_sequence(), 1) - .await - .unwrap(); - executor.bind_prepared(&prepared).unwrap(); - assert!(matches!( - executor.confirm_published(&prepared.root()).unwrap(), - StoredOutcome::Rejected { ref result, commit_sequence: 1 } - if result == b"insufficient quota" - )); - assert!(matches!( - executor.execute( - identity, - Digest::from_bytes([10; 32]), - 111, - RESULT_LIMIT, - |_| { Ok(HandlerOutcome::Success(Vec::new())) }, - ), - Err(crab_cell_runtime::Error::RequestConflict) - )); - executor.close().unwrap(); - - let connection = crab_ltx::rusqlite::Connection::open(database).unwrap(); - assert_eq!( - connection - .query_row("SELECT value FROM counter", [], |row| row.get::<_, i64>(0)) - .unwrap(), - 0 - ); - assert_eq!( - connection - .query_row( - "SELECT outcome FROM sys_requests WHERE request_id = ?1", - [identity.request_id.as_bytes().as_slice()], - |row| row.get::<_, i64>(0), - ) - .unwrap(), - 2 - ); -} - -#[tokio::test] -async fn publisher_uploads_cas_and_releases_one_result() { - let Fixture { - _directory, - database: _, - cell, - incarnation, - layout, - replica, - mut executor, - } = fixture(); - let (_, authority, observed) = initialized_authority(&layout, cell, incarnation).await; - let identity = MutationIdentity { - request_id: RequestId::from_bytes([14; 16]), - issued_at_ms: 100, - expires_at_ms: 20_000, - }; - executor - .execute( - identity, - Digest::from_bytes([15; 32]), - 110, - RESULT_LIMIT, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"committed".to_vec())) - }, - ) - .unwrap(); - let mut publisher = - CellPublisher::new(replica, authority, observed, _directory.path().to_owned()); - assert!(matches!( - publisher - .publish_pending(&mut executor) - .await - .unwrap(), - StoredOutcome::Success { ref result, commit_sequence: 1 } if result == b"committed" - )); - assert_eq!( - publisher.control().value().next_due_ms, - Some(20_000 + 24 * 60 * 60 * 1000) - ); - assert!(executor.pending().is_none()); - executor.close().unwrap(); -} - -#[tokio::test] -async fn lost_publication_response_reconciles_without_replaying_sql() { - let fault_store = Arc::new(LostUpdateResponseStore::new(Arc::new(InMemory::new()))); - let Fixture { - _directory, - database: _, - cell, - incarnation, - layout, - replica, - mut executor, - } = fixture_with_store(Store::new(fault_store.clone())); - let (_, authority, observed) = initialized_authority(&layout, cell, incarnation).await; - let identity = MutationIdentity { - request_id: RequestId::from_bytes([12; 16]), - issued_at_ms: 100, - expires_at_ms: 20_000, - }; - let digest = Digest::from_bytes([13; 32]); - let calls = Arc::new(AtomicUsize::new(0)); - let observed_calls = calls.clone(); - executor - .execute(identity, digest, 110, RESULT_LIMIT, move |transaction| { - observed_calls.fetch_add(1, Ordering::SeqCst); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"published".to_vec())) - }) - .unwrap(); - - let mut publisher = - CellPublisher::new(replica, authority, observed, _directory.path().to_owned()); - assert!(matches!( - publisher - .publish_pending(&mut executor) - .await - .unwrap(), - StoredOutcome::Success { ref result, commit_sequence: 1 } if result == b"published" - )); - assert_eq!(fault_store.updates.load(Ordering::SeqCst), 1); - assert_eq!(fault_store.remaining_failures.load(Ordering::SeqCst), 0); - assert!(executor.pending().is_none()); - assert!(matches!( - executor - .execute(identity, digest, 111, RESULT_LIMIT, |_| { - calls.fetch_add(1, Ordering::SeqCst); - Ok(HandlerOutcome::Success(b"replayed".to_vec())) - }) - .unwrap(), - CommandExecution::Recorded(StoredOutcome::Success { ref result, commit_sequence: 1 }) - if result == b"published" - )); - assert_eq!(calls.load(Ordering::SeqCst), 1); - executor.close().unwrap(); -} - -#[tokio::test] -async fn published_root_survives_local_prune_failure_without_replaying_sql() { - let filesystem = Arc::new(FaultFileSystem::new()); - let host = Host::default().with_filesystem(filesystem.clone()); - let Fixture { - _directory, - database: _, - cell, - incarnation, - layout, - replica, - mut executor, - } = fixture_with_store_and_host(Store::new(Arc::new(InMemory::new())), host); - let (_, authority, observed) = initialized_authority(&layout, cell, incarnation).await; - let recovery_replica = replica.clone(); - let identity = MutationIdentity { - request_id: RequestId::from_bytes([21; 16]), - issued_at_ms: 100, - expires_at_ms: 20_000, - }; - let digest = Digest::from_bytes([22; 32]); - let calls = Arc::new(AtomicUsize::new(0)); - let observed_calls = calls.clone(); - assert_eq!( - executor - .execute(identity, digest, 110, RESULT_LIMIT, move |transaction| { - observed_calls.fetch_add(1, Ordering::SeqCst); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"committed".to_vec())) - }) - .unwrap(), - CommandExecution::Pending - ); - - filesystem.fail_next_prune(); - let mut publisher = - CellPublisher::new(replica, authority, observed, _directory.path().to_owned()); - assert!(matches!( - publisher.publish_pending(&mut executor).await, - Err(crab_cell_runtime::Error::Ltx(_)) - )); - assert!(filesystem.prune_failure_consumed()); - assert!(executor.pending().is_some()); - assert_eq!(calls.load(Ordering::SeqCst), 1); - - let root = publisher.control().value().ltx_root().unwrap(); - let observed_control = CellAuthority::new(layout) - .load(cell) - .await - .unwrap() - .unwrap(); - assert_eq!(observed_control.value().ltx_root(), Some(root)); - assert_eq!(root.commit_sequence, 1); - assert_eq!(root.position.txid, 1); - assert!(matches!( - executor.execute(identity, digest, 111, RESULT_LIMIT, |_| { - calls.fetch_add(1, Ordering::SeqCst); - Ok(HandlerOutcome::Success(b"duplicate".to_vec())) - }), - Err(crab_cell_runtime::Error::PendingPublication) - )); - drop(executor); - - let restored = _directory.path().join("restored.sqlite"); - let verified = recovery_replica.open_root(&root).await.unwrap(); - assert_eq!(verified.root(), root); - assert_eq!(verified.restore(&restored).await.unwrap(), root.position); - let mut recovered = CellExecutor::new( - Db::open(&restored, Limits::default()).unwrap(), - cell, - incarnation, - 1, - ); - assert!(matches!( - recovered - .execute(identity, digest, 112, RESULT_LIMIT, |_| { - calls.fetch_add(1, Ordering::SeqCst); - Ok(HandlerOutcome::Success(b"duplicate".to_vec())) - }) - .unwrap(), - CommandExecution::Recorded(StoredOutcome::Success { ref result, commit_sequence: 1 }) - if result == b"committed" - )); - assert_eq!(calls.load(Ordering::SeqCst), 1); - recovered.close().unwrap(); -} - -#[tokio::test] -async fn published_root_observed_after_takeover_fences_the_old_executor() { - let Fixture { - _directory, - database: _, - cell, - incarnation, - layout, - replica, - mut executor, - } = fixture(); - let (initial, authority, stale) = initialized_authority(&layout, cell, incarnation).await; - let identity = MutationIdentity { - request_id: RequestId::from_bytes([18; 16]), - issued_at_ms: 100, - expires_at_ms: 20_000, - }; - executor - .execute( - identity, - Digest::from_bytes([19; 32]), - 110, - RESULT_LIMIT, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success( - b"published-before-takeover".to_vec(), - )) - }, - ) - .unwrap(); - let pending = executor.pending().unwrap(); - let prepared = replica - .prepare(None, pending.cuts(), pending.outcome().commit_sequence(), 1) - .await - .unwrap(); - let published = authority - .transition( - &stale, - initial.publish_prepared(&prepared, None).unwrap(), - Transition::Publish, - ) - .await - .unwrap(); - let mut takeover = published.value().clone(); - takeover.epoch += 1; - takeover.revision += 1; - takeover.progress += 1; - takeover.state = ControlState::Recovering; - takeover.owner = Some(Owner { - session: SessionId::from_bytes([20; 16]), - endpoint: "https://replacement.internal:8081".into(), - }); - authority - .transition(&published, takeover, Transition::Takeover) - .await - .unwrap(); - - let mut publisher = CellPublisher::new(replica, authority, stale, _directory.path().to_owned()); - assert!(matches!( - publisher.publish_pending(&mut executor).await, - Err(crab_cell_runtime::Error::Fenced) - )); - assert!(matches!( - executor.execute( - identity, - Digest::from_bytes([19; 32]), - 111, - RESULT_LIMIT, - |_| Ok(HandlerOutcome::Success(Vec::new())), - ), - Err(crab_cell_runtime::Error::Fenced) - )); -} - -#[tokio::test] -async fn publication_rebases_over_a_pure_lease_renewal_without_sql_replay() { - let Fixture { - _directory, - database: _, - cell, - incarnation, - layout, - replica, - mut executor, - } = fixture(); - let (initial, authority, stale) = initialized_authority(&layout, cell, incarnation).await; - let identity = MutationIdentity { - request_id: RequestId::from_bytes([16; 16]), - issued_at_ms: 100, - expires_at_ms: 20_000, - }; - executor - .execute( - identity, - Digest::from_bytes([17; 32]), - 110, - RESULT_LIMIT, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"renewed".to_vec())) - }, - ) - .unwrap(); - - let mut renewed = initial; - renewed.revision += 1; - renewed.progress += 1; - authority - .transition(&stale, renewed, Transition::Renew) - .await - .unwrap(); - - let mut publisher = CellPublisher::new(replica, authority, stale, _directory.path().to_owned()); - assert!(matches!( - publisher - .publish_pending(&mut executor) - .await - .unwrap(), - StoredOutcome::Success { ref result, commit_sequence: 1 } if result == b"renewed" - )); - assert_eq!(publisher.control().value().revision, 3); - executor.close().unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/runtime/release_progress.rs b/crates/crab-cell-runtime/tests/runtime/release_progress.rs deleted file mode 100644 index 2fc6d8257..000000000 --- a/crates/crab-cell-runtime/tests/runtime/release_progress.rs +++ /dev/null @@ -1,91 +0,0 @@ -//! Release progress tests extracted from `src/release_progress.rs`. -use crab_cell_runtime::cell::application::ApplicationIdentity; -use crab_cell_runtime::identity::RequestId; -use crab_cell_runtime::ltx::CellStorageLayout; -use crab_cell_runtime::recovery::release_progress::{ - MigrationFailure, MigrationProgressAttempt, MigrationProgressState, MigrationProgressStore, -}; - -use std::sync::Arc; - -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use crab_cell_runtime::identity::{ApplicationId, TenantId}; -use crab_cell_runtime::*; - -fn fixture() -> (MigrationProgressStore, MigrationProgressAttempt) { - let identity = ApplicationIdentity::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - ); - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("progress"), - *identity.application().as_bytes(), - ); - let attempt = MigrationProgressAttempt::new( - RequestId::from_bytes([3; 16]), - Digest::from_bytes([4; 32]), - SessionId::from_bytes([5; 16]), - CellId::from_bytes([6; 32]), - (Digest::from_bytes([7; 32]), 1), - (Digest::from_bytes([8; 32]), 2), - ) - .unwrap(); - ( - MigrationProgressStore::new(layout, identity).unwrap(), - attempt, - ) -} - -#[tokio::test] -async fn failure_retry_and_completion_are_canonical_and_monotonic() { - let (store, attempt) = fixture(); - let failed = store - .failed(attempt, MigrationFailure::Unavailable, 10) - .await - .unwrap(); - assert_eq!(failed.revision(), 1); - assert_eq!(failed.attempts(), 1); - assert_eq!(failed.state(), MigrationProgressState::Failed); - assert_eq!(failed.failure(), Some(MigrationFailure::Unavailable)); - - let completed = store.completed(attempt, 20).await.unwrap(); - assert_eq!(completed.revision(), 2); - assert_eq!(completed.attempts(), 2); - assert_eq!(completed.state(), MigrationProgressState::Completed); - assert_eq!(completed.failure(), None); - assert_eq!(store.completed(attempt, 30).await.unwrap(), completed); - assert_eq!( - store - .failed(attempt, MigrationFailure::Internal, 40) - .await - .unwrap(), - completed - ); - assert_eq!( - store - .load(attempt.cell(), attempt.operation()) - .await - .unwrap() - .unwrap(), - completed - ); -} - -#[tokio::test] -async fn one_operation_path_rejects_version_scope_drift() { - let (store, attempt) = fixture(); - store.completed(attempt, 10).await.unwrap(); - let changed = MigrationProgressAttempt::new( - attempt.operation(), - attempt.release(), - attempt.session(), - attempt.cell(), - attempt.from(), - (Digest::from_bytes([9; 32]), attempt.to().1), - ) - .unwrap(); - assert!(store.completed(changed, 20).await.is_err()); -} diff --git a/crates/crab-cell-runtime/tests/runtime/scheduler.rs b/crates/crab-cell-runtime/tests/runtime/scheduler.rs deleted file mode 100644 index c238ba1cd..000000000 --- a/crates/crab-cell-runtime/tests/runtime/scheduler.rs +++ /dev/null @@ -1,115 +0,0 @@ -use crab_cell_runtime::cell::schema::install_runtime_schema; -use crab_cell_runtime::fleet::scheduler::{ - SchedulerFleet, preferred_scanner, scheduler_next_due_ms, scheduler_tick, -}; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, Digest, NamespaceId, SessionId, TenantId, -}; -use crab_cell_runtime::node::{NodeAdvertisement, NodeCapacity}; -use crab_cell_runtime::primitives::blob::install_blob_schema; -use crab_cell_runtime::primitives::cron::{CronTarget, install_cron_schema}; -use crab_cell_runtime::primitives::effects::{EffectCommandIntent, effect_id}; -use crab_cell_runtime::primitives::kv::install_kv_schema; -use crab_cell_runtime::primitives::queue::{QueueDeadLetterTarget, install_queue_schema}; -use crab_cell_runtime::primitives::workflow::{ - WorkflowAction, WorkflowContext, WorkflowDecision, WorkflowDefinition, WorkflowStatus, - install_workflow_schema, -}; -use ed25519_dalek::SigningKey; - -struct ExpiryDefinition; - -impl WorkflowDefinition for ExpiryDefinition { - fn digest(&self) -> crab_cell_runtime::Digest { - crab_cell_runtime::Digest::from_bytes([9; 32]) - } - - fn transition( - &self, - _state: &[u8], - event: &[u8], - _context: WorkflowContext, - ) -> crab_cell_runtime::Result { - assert!(event.starts_with(b"activity\0\x01")); - Ok(WorkflowDecision { - status: WorkflowStatus::Failed, - state: b"activity-expired".to_vec(), - result: Some(event.to_vec()), - actions: Vec::new(), - }) - } -} - -static EXPIRY_DEFINITION: ExpiryDefinition = ExpiryDefinition; - -struct EffectDefinition; - -const EFFECT_NAMESPACE: NamespaceId = NamespaceId::from_bytes([8; 16]); -static EFFECT_TARGETS: [NamespaceId; 1] = [EFFECT_NAMESPACE]; - -impl WorkflowDefinition for EffectDefinition { - fn digest(&self) -> crab_cell_runtime::Digest { - crab_cell_runtime::Digest::from_bytes([10; 32]) - } - - fn effect_targets(&self) -> &'static [NamespaceId] { - &EFFECT_TARGETS - } - - fn transition( - &self, - _state: &[u8], - event: &[u8], - context: WorkflowContext, - ) -> crab_cell_runtime::Result { - Ok(WorkflowDecision { - status: WorkflowStatus::Failed, - state: b"activity-expired".to_vec(), - result: Some(event.to_vec()), - actions: vec![WorkflowAction::Effect { - intent: EffectCommandIntent { - target: CellTarget::new( - context.source().tenant(), - context.source().application(), - EFFECT_NAMESPACE, - b"destination", - )?, - command_id: 7, - codec_version: 1, - input: event.to_vec(), - expires_at_ms: 20_000, - }, - }], - }) - } -} - -static EFFECT_DEFINITION: EffectDefinition = EffectDefinition; - -fn connection() -> crab_ltx::rusqlite::Connection { - let mut connection = crab_ltx::rusqlite::Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - source_target().cell_id(), - IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - connection -} - -fn source_target() -> CellTarget { - CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NamespaceId::from_bytes([3; 16]), - b"scheduler", - ) - .unwrap() -} - -mod public_tick; -mod scanner; -mod summary; -mod tick; diff --git a/crates/crab-cell-runtime/tests/runtime/scheduler/public_tick.rs b/crates/crab-cell-runtime/tests/runtime/scheduler/public_tick.rs deleted file mode 100644 index 862e0e0cd..000000000 --- a/crates/crab-cell-runtime/tests/runtime/scheduler/public_tick.rs +++ /dev/null @@ -1,85 +0,0 @@ -//! Public tick cron firing and queue dead-letter routing. - -use super::*; - -#[test] -fn public_tick_fires_due_cron_schedules_through_its_targets() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_cron_schema(&transaction).unwrap(); - let target = CronTarget::new("cron-fixture", NamespaceId::from_bytes([9; 16]), 1, 1, 4096); - transaction - .execute( - "INSERT INTO cron_schedules(schedule_id, target_index, target_partition, payload, \ - interval_ms, next_due_ms, occurrence, enabled, generation, updated_at_ms) \ - VALUES (?1, 0, X'00', ?2, 1000, 1000, 0, 1, 1, 0)", - crab_ltx::rusqlite::params![ - crab_ltx::rusqlite::types::Value::Blob(vec![4; 16]), - crab_ltx::rusqlite::types::Value::Blob(b"payload".to_vec()), - ], - ) - .unwrap(); - - let outcome = - scheduler_tick(&transaction, &source_target(), 2_000, &[], None, &[target]).unwrap(); - - assert_eq!( - outcome.processed, 2, - "both due occurrences reach the depth-first class through the public entry" - ); - let schedule = transaction - .query_row( - "SELECT next_due_ms, occurrence FROM cron_schedules", - [], - |row| Ok((row.get::<_, i64>(0)?, row.get::<_, i64>(1)?)), - ) - .unwrap(); - assert_eq!(schedule, (3_000, 2)); - let effects: i64 = transaction - .query_row("SELECT count(*) FROM sys_effects", [], |row| row.get(0)) - .unwrap(); - assert_eq!(effects, 2, "each occurrence publishes its own effect"); -} -#[test] -fn public_tick_dead_letters_exhausted_queue_work_through_its_target() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - let dead_letter_namespace = NamespaceId::from_bytes([10; 16]); - let dead_letter = QueueDeadLetterTarget::new("dead-letter", dead_letter_namespace, 1, 7, 1); - transaction - .execute( - "INSERT INTO queue_messages(message_id, payload, state, attempt, due_at_ms, \ - expires_at_ms, token, lease_until_ms, result_code, dead_letter_effect_id) \ - VALUES (?1, X'', 0, 20, 1000, 10000, NULL, NULL, NULL, NULL)", - [crab_ltx::rusqlite::types::Value::Blob(vec![5; 16])], - ) - .unwrap(); - - scheduler_tick( - &transaction, - &source_target(), - 2_000, - &[], - Some(dead_letter), - &[], - ) - .unwrap(); - - let (state, dead_lettered): (i64, bool) = transaction - .query_row( - "SELECT state, dead_letter_effect_id IS NOT NULL FROM queue_messages", - [], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .unwrap(); - assert_eq!( - (state, dead_lettered), - (3, true), - "an exhausted ready message becomes dead and names its dead-letter effect" - ); - let effects: i64 = transaction - .query_row("SELECT count(*) FROM sys_effects", [], |row| row.get(0)) - .unwrap(); - assert_eq!(effects, 1, "the dead-letter target receives one effect"); -} diff --git a/crates/crab-cell-runtime/tests/runtime/scheduler/scanner.rs b/crates/crab-cell-runtime/tests/runtime/scheduler/scanner.rs deleted file mode 100644 index d1b8b46ae..000000000 --- a/crates/crab-cell-runtime/tests/runtime/scheduler/scanner.rs +++ /dev/null @@ -1,150 +0,0 @@ -//! Rendezvous scanner selection and scanner liveness. - -use super::*; - -#[test] -fn rendezvous_scanner_choice_is_order_independent_and_uses_both_nodes() { - let first = SessionId::from_bytes([1; 16]); - let second = SessionId::from_bytes([2; 16]); - let mut winners = std::collections::HashSet::new(); - for shard in 0_u8..=u8::MAX { - let forward = preferred_scanner(shard, &[first, second]).unwrap().unwrap(); - let reverse = preferred_scanner(shard, &[second, first]).unwrap().unwrap(); - assert_eq!(forward, reverse); - winners.insert(*forward.as_bytes()); - } - assert_eq!(winners.len(), 2); - assert!(preferred_scanner(0, &[first, first]).is_err()); -} -#[test] -fn stalled_scanner_is_removed_until_its_advertised_progress_advances() { - let key = SigningKey::from_bytes(&[7; 32]); - let first = advertisement(1, 1, 0, &key); - let second = advertisement(2, 1, 0, &key); - let mut fleet = SchedulerFleet::default(); - assert_eq!( - fleet - .eligible_sessions(&[first.clone(), second], 0, 100) - .unwrap() - .len(), - 2 - ); - - let second = advertisement(2, 2, 50, &key); - assert_eq!( - fleet - .eligible_sessions(&[first.clone(), second.clone()], 101, 100) - .unwrap(), - vec![SessionId::from_bytes([2; 16])] - ); - assert_eq!( - fleet - .eligible_sessions(&[advertisement(1, 2, 102, &key), second], 102, 100) - .unwrap(), - vec![ - SessionId::from_bytes([1; 16]), - SessionId::from_bytes([2; 16]) - ] - ); - assert!( - fleet - .eligible_sessions(&[advertisement(1, 1, 103, &key)], 103, 100) - .is_err() - ); -} -#[test] -fn exhausted_scanner_is_removed_until_its_capacity_recovers() { - let key = SigningKey::from_bytes(&[9; 32]); - let exhausted = [ - NodeCapacity { - free_memory_bytes: 0, - free_disk_bytes: 1, - job_credits: 1, - ..NodeCapacity::default() - }, - NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 0, - job_credits: 1, - ..NodeCapacity::default() - }, - NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 1, - job_credits: 0, - ..NodeCapacity::default() - }, - ]; - let available = advertisement(2, 1, 0, &key); - let mut fleet = SchedulerFleet::default(); - - for capacity in exhausted { - assert_eq!( - fleet - .eligible_sessions( - &[ - advertisement_with_capacity(1, 1, 0, &key, capacity), - available.clone(), - ], - 0, - 100, - ) - .unwrap(), - vec![SessionId::from_bytes([2; 16])] - ); - } - assert_eq!( - fleet - .eligible_sessions(&[advertisement(1, 1, 50, &key), available], 50, 100) - .unwrap(), - vec![ - SessionId::from_bytes([1; 16]), - SessionId::from_bytes([2; 16]) - ] - ); -} -fn advertisement( - session: u8, - progress: u64, - issued_at_ms: i64, - key: &SigningKey, -) -> NodeAdvertisement { - advertisement_with_capacity( - session, - progress, - issued_at_ms, - key, - NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 1, - job_credits: 1, - ..NodeCapacity::default() - }, - ) -} -fn advertisement_with_capacity( - session: u8, - progress: u64, - issued_at_ms: i64, - key: &SigningKey, - capacity: NodeCapacity, -) -> NodeAdvertisement { - NodeAdvertisement::sign( - crab_cell_runtime::identity::NodeId::from_bytes([session; 16]), - SessionId::from_bytes([session; 16]), - format!("https://node-{session}.internal:8789"), - Digest::from_bytes([1; 32]), - Digest::from_bytes([2; 32]), - Digest::from_bytes([3; 32]), - Digest::from_bytes([4; 32]), - key, - progress, - issued_at_ms, - issued_at_ms + 10_000, - vec![Digest::from_bytes([5; 32])], - vec![1], - crab_cell_runtime::node::NodeFailureDomain::default(), - capacity, - ) - .unwrap() -} diff --git a/crates/crab-cell-runtime/tests/runtime/scheduler/summary.rs b/crates/crab-cell-runtime/tests/runtime/scheduler/summary.rs deleted file mode 100644 index 0c2f0b755..000000000 --- a/crates/crab-cell-runtime/tests/runtime/scheduler/summary.rs +++ /dev/null @@ -1,144 +0,0 @@ -//! Scheduler summaries for deadline classes and primitive deadlines. - -use super::*; - -#[test] -fn runtime_only_summary_clamps_overdue_work() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - assert_eq!(scheduler_next_due_ms(&transaction, 10).unwrap(), None); - transaction - .execute( - "INSERT INTO sys_requests VALUES (?1, ?2, 1, X'', 1, 20, 30)", - ([3_u8; 16].as_slice(), [4_u8; 32].as_slice()), - ) - .unwrap(); - assert_eq!(scheduler_next_due_ms(&transaction, 10).unwrap(), Some(30)); - assert_eq!(scheduler_next_due_ms(&transaction, 40).unwrap(), Some(40)); -} -#[test] -fn summary_covers_all_installed_primitive_deadline_classes() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_kv_schema(&transaction).unwrap(); - install_queue_schema(&transaction).unwrap(); - install_workflow_schema(&transaction).unwrap(); - transaction - .execute( - "INSERT INTO kv_entries VALUES (X'01', X'02', zeroblob(28), X'03', 90)", - [], - ) - .unwrap(); - transaction - .execute( - "INSERT INTO queue_messages VALUES (zeroblob(16), X'04', 1, 1, 50, 100, zeroblob(16), 80, NULL, NULL)", - [], - ) - .unwrap(); - transaction - .execute( - "INSERT INTO sys_effects VALUES (zeroblob(32), zeroblob(32), X'05', 0, 0, 70, 110, NULL, NULL, 1, NULL)", - [], - ) - .unwrap(); - transaction - .execute( - "INSERT INTO workflow_runs VALUES (X'06', zeroblob(16), zeroblob(32), 0, X'', 0, NULL, NULL)", - [], - ) - .unwrap(); - transaction - .execute( - "INSERT INTO workflow_activities VALUES (zeroblob(16), zeroblob(16), 'job', X'', 0, 0, 60, 120, NULL, NULL, NULL, NULL, NULL)", - [], - ) - .unwrap(); - transaction - .execute( - "INSERT INTO workflow_timers VALUES (zeroblob(16), X'01010101010101010101010101010101', 40, 0)", - [], - ) - .unwrap(); - assert_eq!(scheduler_next_due_ms(&transaction, 10).unwrap(), Some(40)); - transaction - .execute("UPDATE workflow_timers SET state = 1", []) - .unwrap(); - assert_eq!(scheduler_next_due_ms(&transaction, 10).unwrap(), Some(60)); -} -#[test] -fn summary_ignores_ready_queue_available_time() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - transaction - .execute( - "INSERT INTO queue_messages VALUES (X'01010101010101010101010101010101', X'04', 0, 0, 10, 100, NULL, NULL, NULL, NULL)", - [], - ) - .unwrap(); - - assert_eq!(scheduler_next_due_ms(&transaction, 10).unwrap(), Some(100)); -} -#[test] -fn summary_wakes_immediately_for_exhausted_ready_queue_work() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - transaction - .execute( - "INSERT INTO queue_messages VALUES (X'02020202020202020202020202020202', X'04', 0, 20, 100, 1000, NULL, NULL, NULL, NULL)", - [], - ) - .unwrap(); - - assert_eq!(scheduler_next_due_ms(&transaction, 10).unwrap(), Some(10)); -} -#[test] -fn exhausted_ready_deadline_uses_the_attempt_index() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - let details = transaction - .prepare( - "EXPLAIN QUERY PLAN SELECT EXISTS(SELECT 1 FROM queue_messages INDEXED BY queue_attempts WHERE state = 0 AND attempt >= ?1)", - ) - .unwrap() - .query_map([i64::from(20_u32)], |row| row.get::<_, String>(3)) - .unwrap() - .collect::>>() - .unwrap(); - - assert!( - details - .iter() - .any(|detail| detail.contains("SEARCH") && detail.contains("queue_attempts")) - ); -} -#[test] -fn summary_uses_queue_lease_deadline_not_ready_available_time() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - transaction - .execute( - "INSERT INTO queue_messages VALUES (X'03030303030303030303030303030303', X'04', 1, 1, 10, 1000, zeroblob(16), 80, NULL, NULL)", - [], - ) - .unwrap(); - - assert_eq!(scheduler_next_due_ms(&transaction, 10).unwrap(), Some(80)); -} -#[test] -fn summary_uses_cron_deadline_when_the_schedule_schema_is_installed() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_cron_schema(&transaction).unwrap(); - transaction - .execute( - "INSERT INTO cron_schedules(schedule_id, target_index, target_partition, payload, interval_ms, next_due_ms, occurrence, enabled, generation, updated_at_ms) VALUES (X'02020202020202020202020202020202', 0, X'00', X'', 30000, 80, 0, 1, 1, 10)", - [], - ) - .unwrap(); - - assert_eq!(scheduler_next_due_ms(&transaction, 10).unwrap(), Some(80)); -} diff --git a/crates/crab-cell-runtime/tests/runtime/scheduler/tick.rs b/crates/crab-cell-runtime/tests/runtime/scheduler/tick.rs deleted file mode 100644 index 287c890cb..000000000 --- a/crates/crab-cell-runtime/tests/runtime/scheduler/tick.rs +++ /dev/null @@ -1,212 +0,0 @@ -//! Tick budgets, cleanup ordering, expiry, and effect ordinals. - -use super::*; - -#[test] -fn tick_processes_at_most_128_due_items() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - for ordinal in 0_u16..140 { - let mut request_id = [0; 16]; - request_id[..2].copy_from_slice(&ordinal.to_be_bytes()); - transaction - .execute( - "INSERT INTO sys_requests VALUES (?1, ?2, 1, X'', 1, 1, 2)", - (request_id.as_slice(), [4_u8; 32].as_slice()), - ) - .unwrap(); - } - assert_eq!( - scheduler_tick(&transaction, &source_target(), 10, &[], None, &[]) - .unwrap() - .processed, - 128 - ); - let remaining: usize = transaction - .query_row("SELECT count(*) FROM sys_requests", [], |row| row.get(0)) - .unwrap(); - assert_eq!(remaining, 12); - assert_eq!( - scheduler_tick(&transaction, &source_target(), 10, &[], None, &[]) - .unwrap() - .processed, - 12 - ); -} -#[test] -fn tick_request_cleanup_cannot_starve_queue_expiry() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - for ordinal in 0_u16..200 { - let mut request_id = [0; 16]; - request_id[..2].copy_from_slice(&ordinal.to_be_bytes()); - transaction - .execute( - "INSERT INTO sys_requests VALUES (?1, ?2, 1, X'', 1, 1, 2)", - (request_id.as_slice(), [4_u8; 32].as_slice()), - ) - .unwrap(); - } - transaction - .execute( - "INSERT INTO queue_messages VALUES (zeroblob(16), X'02', 0, 0, 1, 5, NULL, NULL, NULL, NULL)", - [], - ) - .unwrap(); - - let outcome = scheduler_tick(&transaction, &source_target(), 10, &[], None, &[]).unwrap(); - - assert_eq!(outcome.processed, 128); - assert_eq!( - transaction - .query_row("SELECT state FROM queue_messages", [], |row| row - .get::<_, i64>(0)) - .unwrap(), - 3 - ); - assert!( - transaction - .query_row("SELECT count(*) FROM sys_requests", [], |row| row - .get::<_, usize>(0)) - .unwrap() - > 0 - ); -} -#[test] -fn tick_terminalizes_expired_ready_work_and_runs_workflow_failure_transition() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_queue_schema(&transaction).unwrap(); - install_workflow_schema(&transaction).unwrap(); - transaction - .execute( - "INSERT INTO sys_effects VALUES (zeroblob(32), zeroblob(32), X'01', 0, 0, 1, 5, NULL, NULL, 1, NULL)", - [], - ) - .unwrap(); - transaction - .execute( - "INSERT INTO queue_messages VALUES (zeroblob(16), X'02', 0, 0, 1, 5, NULL, NULL, NULL, NULL)", - [], - ) - .unwrap(); - transaction - .execute( - "INSERT INTO workflow_runs VALUES (X'03', zeroblob(16), ?1, 0, X'', 1, NULL, NULL)", - [EXPIRY_DEFINITION.digest().as_bytes().as_slice()], - ) - .unwrap(); - transaction - .execute( - "INSERT INTO workflow_activities VALUES (zeroblob(16), X'01010101010101010101010101010101', 'job', X'', 0, 0, 1, 5, NULL, NULL, NULL, NULL, NULL)", - [], - ) - .unwrap(); - - let outcome = scheduler_tick( - &transaction, - &source_target(), - 10, - &[&EXPIRY_DEFINITION], - None, - &[], - ) - .unwrap(); - assert_eq!(outcome.processed, 3); - let states: (i64, i64, i64) = transaction - .query_row( - "SELECT (SELECT state FROM sys_effects), (SELECT state FROM queue_messages), (SELECT status FROM workflow_runs)", - [], - |row| Ok((row.get(0)?, row.get(1)?, row.get(2)?)), - ) - .unwrap(); - assert_eq!(states, (3, 3, 2)); -} -#[test] -fn tick_assigns_unique_command_ordinals_to_effects_from_multiple_runs() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_workflow_schema(&transaction).unwrap(); - for byte in [3_u8, 4_u8] { - transaction - .execute( - "INSERT INTO workflow_runs VALUES (?1, ?2, ?3, 0, X'', 1, NULL, NULL)", - ( - [byte].as_slice(), - [byte; 16].as_slice(), - EFFECT_DEFINITION.digest().as_bytes().as_slice(), - ), - ) - .unwrap(); - transaction - .execute( - "INSERT INTO workflow_activities VALUES (?1, ?2, 'job', X'', 0, 0, 1, 5, NULL, NULL, NULL, NULL, NULL)", - ([byte; 16].as_slice(), [byte + 10; 16].as_slice()), - ) - .unwrap(); - } - - assert_eq!( - scheduler_tick( - &transaction, - &source_target(), - 10, - &[&EFFECT_DEFINITION], - None, - &[] - ) - .unwrap() - .processed, - 2 - ); - let mut statement = transaction - .prepare("SELECT effect_id FROM sys_effects ORDER BY effect_id") - .unwrap(); - let mut stored = statement - .query_map([], |row| row.get::<_, Vec>(0)) - .unwrap() - .collect::, _>>() - .unwrap(); - let mut expected = vec![ - effect_id( - source_target().cell_id(), - IncarnationId::from_bytes([2; 16]), - 1, - 0, - ) - .to_vec(), - effect_id( - source_target().cell_id(), - IncarnationId::from_bytes([2; 16]), - 1, - 1, - ) - .to_vec(), - ]; - expected.sort(); - stored.sort(); - assert_eq!(stored, expected); -} -#[test] -fn tick_cleans_expired_blob_uploads_that_never_completed() { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_blob_schema(&transaction).unwrap(); - transaction - .execute( - "INSERT INTO blob_uploads VALUES (X'01010101010101010101010101010101', X'02', zeroblob(32), 0, NULL, NULL, X'', 10, 20, 0, NULL, 0, 0)", - [], - ) - .unwrap(); - - scheduler_tick(&transaction, &source_target(), 25, &[], None, &[]).unwrap(); - - let remaining: usize = transaction - .query_row("SELECT count(*) FROM blob_uploads", [], |row| row.get(0)) - .unwrap(); - assert_eq!( - remaining, 0, - "an expired upload without an object is garbage" - ); -} diff --git a/crates/crab-cell-runtime/tests/runtime/scheduler_properties.proptest-regressions b/crates/crab-cell-runtime/tests/runtime/scheduler_properties.proptest-regressions deleted file mode 100644 index 9398155ed..000000000 --- a/crates/crab-cell-runtime/tests/runtime/scheduler_properties.proptest-regressions +++ /dev/null @@ -1,7 +0,0 @@ -# Seeds for failure cases proptest has generated in the past. It is -# automatically read and these particular cases re-run before any -# novel cases are generated. -# -# It is recommended to check this file in to source control so that -# everyone who runs the test benefits from these saved cases. -cc 29d468f98608e0be6ec875319f36575c261b651faa2b9605d9ec22bccce04744 # shrinks to kv_expiry = [], blob_rows = [], queue_expiry = [0] diff --git a/crates/crab-cell-runtime/tests/runtime/scheduler_properties.rs b/crates/crab-cell-runtime/tests/runtime/scheduler_properties.rs deleted file mode 100644 index c6f0e7a7b..000000000 --- a/crates/crab-cell-runtime/tests/runtime/scheduler_properties.rs +++ /dev/null @@ -1,314 +0,0 @@ -//! Randomized retention properties for maintenance Ticks. -//! -//! A Tick deletes durable rows. The invariant is that it never deletes a row -//! whose deadline has not passed and never deletes an upload that a live object -//! reference still names — the primitive analogue of "GC never deletes -//! referenced data or anything inside the grace period". The queue adds one -//! step: an expired ready message is dead-lettered by the expire class, and the -//! retention class removes it on a later Tick once its effect has settled. - -use crab_cell_runtime::cell::schema::install_runtime_schema; -use crab_cell_runtime::fleet::scheduler::scheduler_tick; -use crab_cell_runtime::identity::{ - ApplicationId, CellTarget, IncarnationId, NamespaceId, TenantId, -}; -use crab_cell_runtime::primitives::blob::install_blob_schema; -use crab_cell_runtime::primitives::kv::install_kv_schema; -use crab_cell_runtime::primitives::queue::install_queue_schema; -use crab_ltx::rusqlite::{Connection, params}; -use proptest::prelude::*; - -const NOW_MS: i64 = 1_000; - -fn target() -> CellTarget { - CellTarget::new( - TenantId::from_bytes([1; 16]), - ApplicationId::from_bytes([2; 16]), - NamespaceId::from_bytes([3; 16]), - b"scheduler-retention", - ) - .unwrap() -} - -fn connection() -> Connection { - let mut connection = Connection::open_in_memory().unwrap(); - install_runtime_schema( - &mut connection, - target().cell_id(), - IncarnationId::from_bytes([2; 16]), - 1, - ) - .unwrap(); - connection -} - -fn count(transaction: &crab_ltx::rusqlite::Transaction<'_>, table: &str) -> usize { - transaction - .query_row(&format!("SELECT count(*) FROM {table}"), [], |row| { - row.get::<_, i64>(0) - }) - .unwrap() - .try_into() - .unwrap() -} - -fn queue_state_count(transaction: &crab_ltx::rusqlite::Transaction<'_>, state: i64) -> usize { - transaction - .query_row( - "SELECT count(*) FROM queue_messages WHERE state = ?1", - [state], - |row| row.get::<_, i64>(0), - ) - .unwrap() - .try_into() - .unwrap() -} - -fn state_count( - transaction: &crab_ltx::rusqlite::Transaction<'_>, - table: &str, - state: i64, -) -> usize { - transaction - .query_row( - &format!("SELECT count(*) FROM {table} WHERE state = ?1"), - [state], - |row| row.get::<_, i64>(0), - ) - .unwrap() - .try_into() - .unwrap() -} - -/// Effect expiries are either already past (including the deadline itself) or -/// far enough ahead that the reclaim retry delay cannot consume them. -fn effect_expiry() -> impl Strategy { - prop_oneof![-64i32..=0, 60_000i32..=120_000] -} - -proptest! { - #![proptest_config(ProptestConfig { cases: 16, ..ProptestConfig::default() })] - - #[test] - fn tick_keeps_rows_inside_retention_and_removes_expired_garbage( - kv_expiry in prop::collection::vec(prop::option::of(-64i32..=64), 0..=6), - blob_rows in prop::collection::vec((any::(), -64i32..=64), 0..=6), - queue_expiry in prop::collection::vec(-64i32..=64, 0..=6), - ) { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - install_kv_schema(&transaction).unwrap(); - install_queue_schema(&transaction).unwrap(); - install_blob_schema(&transaction).unwrap(); - - let mut live_kv = 0; - for (index, offset) in kv_expiry.iter().enumerate() { - let expires = offset.map(|offset| NOW_MS + i64::from(offset)); - transaction - .execute( - "INSERT INTO kv_entries(scope, key, version, value, expires_at_ms) \ - VALUES (X'01', ?1, zeroblob(28), X'', ?2)", - params![vec![u8::try_from(index).unwrap() + 1], expires], - ) - .unwrap(); - if expires.is_none_or(|expires| expires > NOW_MS) { - live_kv += 1; - } - } - - let mut live_blobs = 0; - for (index, (referenced, offset)) in blob_rows.iter().enumerate() { - let upload_id = vec![u8::try_from(index).unwrap() + 1; 16]; - let object_key = vec![u8::try_from(index).unwrap() + 1]; - let expires = NOW_MS + i64::from(*offset); - transaction - .execute( - "INSERT INTO blob_uploads VALUES \ - (?1, ?2, zeroblob(32), 0, NULL, NULL, X'', 0, ?3, 0, NULL, 0, 0)", - params![upload_id, object_key, expires], - ) - .unwrap(); - if *referenced { - transaction - .execute( - "INSERT INTO blob_objects VALUES \ - (?1, ?2, zeroblob(32), 0, 1, NULL, X'', 0, 0)", - params![object_key, upload_id], - ) - .unwrap(); - } - // An expired upload survives while an object still names it. - if expires > NOW_MS || *referenced { - live_blobs += 1; - } - } - - let mut live_queue = 0; - for (index, offset) in queue_expiry.iter().enumerate() { - let expires = NOW_MS + i64::from(*offset); - transaction - .execute( - "INSERT INTO queue_messages(message_id, payload, state, attempt, \ - due_at_ms, expires_at_ms, token, lease_until_ms, result_code, \ - dead_letter_effect_id) VALUES (?1, X'', 0, 0, ?2, ?3, NULL, NULL, NULL, NULL)", - params![ - vec![u8::try_from(index).unwrap() + 1; 16], - NOW_MS + 500, - expires - ], - ) - .unwrap(); - if expires > NOW_MS { - live_queue += 1; - } - } - - scheduler_tick(&transaction, &target(), NOW_MS, &[], None, &[]).unwrap(); - - prop_assert_eq!( - count(&transaction, "kv_entries"), - live_kv, - "KV rows inside retention must survive the Tick" - ); - prop_assert_eq!( - count(&transaction, "blob_uploads"), - live_blobs, - "uploads inside retention or named by an object must survive the Tick" - ); - // Expired ready messages are dead-lettered first; the cleanup class runs - // before that transition, so their removal lands on the next Tick. - prop_assert_eq!( - queue_state_count(&transaction, 0), - live_queue, - "queue messages inside retention must stay ready" - ); - prop_assert_eq!( - queue_state_count(&transaction, 3), - queue_expiry.len() - live_queue, - "expired ready messages must be dead-lettered" - ); - - scheduler_tick(&transaction, &target(), NOW_MS, &[], None, &[]).unwrap(); - prop_assert_eq!( - count(&transaction, "kv_entries"), - live_kv, - "a settled Tick must not remove KV rows it already kept" - ); - prop_assert_eq!( - count(&transaction, "blob_uploads"), - live_blobs, - "a settled Tick must not remove uploads it already kept" - ); - prop_assert_eq!( - count(&transaction, "queue_messages"), - live_queue, - "the retention cleanup must remove dead-lettered messages" - ); - } - - #[test] - fn tick_settles_effect_expiry_and_reclaims_only_expired_leases( - ready_expiry in prop::collection::vec(effect_expiry(), 0..=4), - leases in prop::collection::vec((any::(), effect_expiry()), 0..=4), - terminal in prop::collection::vec((any::(), effect_expiry()), 0..=4), - ) { - let mut connection = connection(); - let transaction = connection.transaction().unwrap(); - - let mut next = 0_u8; - let mut insert = |state: i64, expiry_offset: i32, lease_expired: bool| { - next += 1; - let (token, lease_until_ms) = if state == 1 { - let lease = if lease_expired { -10 } else { 10 }; - (Some(vec![next; 16]), Some(NOW_MS + lease)) - } else { - (None, None) - }; - transaction - .execute( - "INSERT INTO sys_effects(effect_id, destination, operation, state, attempt, \ - due_at_ms, expires_at_ms, token, lease_until_ms, created_sequence, result) \ - VALUES (?1, zeroblob(32), X'', ?2, 0, ?3, ?4, ?5, ?6, 1, NULL)", - params![ - vec![next; 32], - state, - NOW_MS - 100, - NOW_MS + i64::from(expiry_offset), - token, - lease_until_ms - ], - ) - .unwrap(); - }; - - for offset in &ready_expiry { - insert(0, *offset, false); - } - for (lease_expired, offset) in &leases { - insert(1, *offset, *lease_expired); - } - for (failed, offset) in &terminal { - insert(if *failed { 3 } else { 2 }, *offset, false); - } - - let live = |offset: &i32| NOW_MS + i64::from(*offset) > NOW_MS; - let live_ready = ready_expiry.iter().filter(|offset| live(offset)).count(); - let expired_ready = ready_expiry.len() - live_ready; - // A live lease keeps its row regardless of the effect's own expiry: only - // the expired lease returns the row to ready (or fails an expired one). - let live_leases = leases - .iter() - .filter(|(lease_expired, _)| !*lease_expired) - .count(); - let reclaimed = - leases.iter().filter(|(lease_expired, offset)| *lease_expired && live(offset)).count(); - let failed_reclaims = - leases.iter().filter(|(lease_expired, offset)| *lease_expired && !live(offset)).count(); - let live_done = terminal.iter().filter(|(failed, offset)| !*failed && live(offset)).count(); - let live_failed = terminal.iter().filter(|(failed, offset)| *failed && live(offset)).count(); - let expired_terminal = - terminal.iter().filter(|(_, offset)| !live(offset)).count(); - let survives = ready_expiry.len() - + leases.len() - + terminal.len() - - expired_terminal; - - // First Tick: expired ready rows fail, a live lease is never reclaimed, - // and only terminal rows past retention are removed immediately. - scheduler_tick(&transaction, &target(), NOW_MS, &[], None, &[]).unwrap(); - prop_assert_eq!( - state_count(&transaction, "sys_effects", 0), - live_ready + reclaimed, - "ready rows inside retention stay ready and expired leases return to ready" - ); - prop_assert_eq!( - state_count(&transaction, "sys_effects", 1), - live_leases, - "a live lease must never be reclaimed" - ); - prop_assert_eq!( - state_count(&transaction, "sys_effects", 2), - live_done, - "a completed effect inside retention must survive" - ); - prop_assert_eq!( - state_count(&transaction, "sys_effects", 3), - live_failed + expired_ready + failed_reclaims, - "expired ready rows and exhausted leases fail before cleanup sees them" - ); - prop_assert_eq!( - count(&transaction, "sys_effects"), - survives, - "terminal rows past retention are removed by the first Tick" - ); - - // Second Tick: the failures from the first Tick are now terminal and past - // retention, so exactly the rows inside retention remain. - scheduler_tick(&transaction, &target(), NOW_MS, &[], None, &[]).unwrap(); - prop_assert_eq!( - count(&transaction, "sys_effects"), - live_ready + live_leases + reclaimed + live_done + live_failed, - "only rows inside retention survive the sweep" - ); - } -} diff --git a/crates/crab-cell-runtime/tests/runtime/workers.rs b/crates/crab-cell-runtime/tests/runtime/workers.rs deleted file mode 100644 index 3d8d4433f..000000000 --- a/crates/crab-cell-runtime/tests/runtime/workers.rs +++ /dev/null @@ -1,455 +0,0 @@ -use std::sync::{Arc, mpsc}; - -use crab_cell_runtime::Error; -use crab_cell_runtime::cell::executor::{CellExecutor, HandlerOutcome, StoredOutcome}; -use crab_cell_runtime::cell::schema::install_runtime_schema; -use crab_cell_runtime::cell::worker::SqlWorkerPool; -use crab_cell_runtime::cell::worker::WorkerExecution; -use crab_cell_runtime::identity::IncarnationId; -use crab_cell_runtime::identity::{CellId, Digest}; -use crab_ltx::CellStorageLayout; -use crab_ltx::{CellReplica, Db, Limits}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path}; - -use crate::support::fixtures::mutation_identity_window; - -const RESULT_LIMIT: usize = 1 << 20; - -#[tokio::test] -async fn primitive_admission_refuses_expired_deadlines_even_with_free_capacity() { - use crab_cell_runtime::{CellRuntime, SessionId}; - use std::time::{Duration, Instant}; - - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 1 << 20, - SessionId::from_bytes([91; 16]), - ) - .unwrap(); - let result = runtime - .reserve_worker_job(Instant::now() - Duration::from_millis(1)) - .await; - let expired = matches!(result, Err(Error::Deadline)); - drop(result); - runtime.shutdown().await.unwrap(); - assert!(expired, "free capacity must not revive an expired request"); -} - -#[tokio::test] -async fn primitive_admission_does_not_revive_an_expired_waiter_when_a_slot_opens() { - use crab_cell_runtime::{CellRuntime, SessionId}; - use std::time::{Duration, Instant}; - - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 1 << 20, - SessionId::from_bytes([93; 16]), - ) - .unwrap(); - let held = runtime.try_reserve_worker_job().unwrap().unwrap(); - let deadline = Instant::now() + Duration::from_millis(10); - let waiter = runtime.reserve_worker_job(deadline); - tokio::pin!(waiter); - assert!(futures_util::poll!(&mut waiter).is_pending()); - tokio::time::sleep_until((deadline + Duration::from_millis(1)).into()).await; - drop(held); - assert!(matches!(waiter.await, Err(Error::Deadline))); - assert_eq!(runtime.stats().primitive_jobs(), 0); - runtime.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn primitive_waiters_release_accounting_on_cancel_timeout_and_shutdown() { - use crab_cell_runtime::{CellRuntime, SessionId}; - use std::time::{Duration, Instant}; - - let runtime = CellRuntime::new( - SqlWorkerPool::new(1, 1).unwrap(), - 1 << 20, - SessionId::from_bytes([92; 16]), - ) - .unwrap(); - let held = runtime.try_reserve_worker_job().unwrap().unwrap(); - let deadline = Instant::now() + Duration::from_secs(5); - { - let waiter = runtime.reserve_worker_job(deadline); - tokio::pin!(waiter); - assert!(futures_util::poll!(&mut waiter).is_pending()); - assert_eq!(runtime.stats().primitive_jobs(), 1); - } - let timed_out = runtime - .reserve_worker_job(Instant::now() + Duration::from_millis(10)) - .await; - assert!(matches!(timed_out, Err(Error::Deadline))); - assert_eq!(runtime.stats().primitive_jobs(), 1); - let waiter = runtime.reserve_worker_job(deadline); - tokio::pin!(waiter); - assert!(futures_util::poll!(&mut waiter).is_pending()); - drop(held); - let admitted = waiter.await.unwrap(); - assert!(runtime.try_reserve_worker_job().unwrap().is_none()); - assert_eq!(runtime.stats().primitive_jobs(), 1); - let waiter = runtime.reserve_worker_job(deadline); - tokio::pin!(waiter); - assert!(futures_util::poll!(&mut waiter).is_pending()); - runtime.shutdown().await.unwrap(); - assert!(matches!(waiter.await, Err(Error::RuntimeClosed))); - drop(admitted); - assert_eq!(runtime.stats().primitive_jobs(), 0); -} - -struct Fixture { - _directory: tempfile::TempDir, - cell: CellId, - replica: CellReplica, - executor: CellExecutor, -} - -fn fixture(cell_byte: u8) -> Fixture { - let cell = CellId::from_bytes([cell_byte; 32]); - let incarnation = IncarnationId::from_bytes([2; 16]); - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store, Path::from("runtime"), [3; 16]); - let replica = CellReplica::new( - layout, - *cell.as_bytes(), - *incarnation.as_bytes(), - Limits::default(), - ) - .unwrap(); - let directory = tempfile::TempDir::new().unwrap(); - let database = directory.path().join("cell.sqlite"); - let mut connection = crab_ltx::rusqlite::Connection::open(&database).unwrap(); - install_runtime_schema(&mut connection, cell, incarnation, 1).unwrap(); - connection - .execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (0)", - ) - .unwrap(); - drop(connection); - let writer = Db::open(&database, Limits::default()).unwrap(); - Fixture { - _directory: directory, - cell, - replica, - executor: CellExecutor::new(writer, cell, incarnation, 1), - } -} - -#[tokio::test] -async fn native_memory_budget_preserves_writer_limit_and_active_reservations() { - let first = fixture(98); - let second = fixture(99); - let pool = SqlWorkerPool::new(1, 1) - .unwrap() - .with_native_memory_limit(32 << 20) - .unwrap(); - pool.activate(first.cell, first.executor).await.unwrap(); - assert!(matches!( - pool.activate(second.cell, second.executor).await, - Err(Error::Capacity(_)) - )); - for limit in [0, 1] { - assert!(matches!( - pool.clone().with_native_memory_limit(limit), - Err(Error::Capacity(_)) - )); - } - pool.deactivate(first.cell).await.unwrap(); - pool.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn fixed_workers_own_execute_prepare_confirm_and_dedup() { - let Fixture { - _directory, - cell, - replica, - executor, - } = fixture(1); - let pool = SqlWorkerPool::new(2, 10).unwrap(); - pool.activate(cell, executor).await.unwrap(); - let request = mutation_identity_window(4, 10, 10_000); - let digest = Digest::from_bytes([5; 32]); - let pending = match pool - .execute(cell, request, digest, 20, RESULT_LIMIT, |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"one".to_vec())) - }) - .await - .unwrap() - { - WorkerExecution::Pending(pending) => pending, - WorkerExecution::Recorded(_) => panic!("first execution cannot be recorded"), - }; - let prepared = replica - .prepare(None, pending.cuts(), pending.outcome().commit_sequence(), 1) - .await - .unwrap(); - pool.bind_prepared(cell, prepared.clone()).await.unwrap(); - assert!(matches!( - pool.confirm_published(cell, prepared.root()).await.unwrap(), - StoredOutcome::Success { ref result, commit_sequence: 1 } if result == b"one" - )); - assert!(matches!( - pool.execute(cell, request, digest, 21, RESULT_LIMIT, |_| { - Ok(HandlerOutcome::Success(b"wrong".to_vec())) - }) - .await - .unwrap(), - WorkerExecution::Recorded(StoredOutcome::Success { ref result, commit_sequence: 1 }) - if result == b"one" - )); - pool.deactivate(cell).await.unwrap(); -} - -#[tokio::test] -async fn cancelled_waiter_does_not_cancel_an_accepted_sql_command() { - let Fixture { - _directory, - cell, - replica, - executor, - } = fixture(6); - let pool = SqlWorkerPool::new(1, 1).unwrap(); - pool.activate(cell, executor).await.unwrap(); - let (started_tx, started_rx) = mpsc::channel(); - let (release_tx, release_rx) = mpsc::channel(); - let waiting = { - let pool = pool.clone(); - tokio::spawn(async move { - pool.execute( - cell, - mutation_identity_window(7, 10, 10_000), - Digest::from_bytes([8; 32]), - 20, - RESULT_LIMIT, - move |transaction| { - started_tx.send(()).unwrap(); - release_rx.recv().unwrap(); - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(HandlerOutcome::Success(b"survived".to_vec())) - }, - ) - .await - }) - }; - tokio::task::spawn_blocking(move || started_rx.recv().unwrap()) - .await - .unwrap(); - let queued = { - let pool = pool.clone(); - tokio::spawn(async move { - pool.execute( - cell, - mutation_identity_window(9, 10, 10_000), - Digest::from_bytes([10; 32]), - 20, - RESULT_LIMIT, - |_| Ok(HandlerOutcome::Success(Vec::new())), - ) - .await - }) - }; - tokio::task::yield_now().await; - assert!( - !queued.is_finished(), - "a worker-saturated request must wait for admission" - ); - queued.abort(); - assert!(matches!(queued.await, Err(error) if error.is_cancelled())); - waiting.abort(); - assert!(matches!(waiting.await, Err(error) if error.is_cancelled())); - release_tx.send(()).unwrap(); - - let pending = pool.pending(cell).await.unwrap().unwrap(); - let prepared = replica - .prepare(None, pending.cuts(), pending.outcome().commit_sequence(), 1) - .await - .unwrap(); - pool.bind_prepared(cell, prepared.clone()).await.unwrap(); - assert!(matches!( - pool.confirm_published(cell, prepared.root()).await.unwrap(), - StoredOutcome::Success { ref result, .. } if result == b"survived" - )); - pool.deactivate(cell).await.unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn a_busy_shard_does_not_reserve_an_idle_workers_capacity() { - let first = fixture(0); - let queued = fixture(2); - let resident = fixture(1); - let pool = SqlWorkerPool::new(2, 3).unwrap(); - pool.activate(first.cell, first.executor).await.unwrap(); - pool.activate(queued.cell, queued.executor).await.unwrap(); - pool.activate(resident.cell, resident.executor) - .await - .unwrap(); - - let (started, entered) = tokio::sync::oneshot::channel(); - let (release, blocked) = mpsc::channel(); - let running = { - let pool = pool.clone(); - tokio::spawn(async move { - pool.execute( - first.cell, - mutation_identity_window(30, 10, 10_000), - Digest::from_bytes([30; 32]), - 20, - RESULT_LIMIT, - move |_| { - started.send(()).unwrap(); - blocked.recv().unwrap(); - Ok(HandlerOutcome::Success(Vec::new())) - }, - ) - .await - }) - }; - entered.await.unwrap(); - - // Cells 0 and 2 share a worker. Poll until admission or its reply blocks, - // then keep that waiter alive while Cell 1 uses the other worker. - let mut waiting = Box::pin(pool.execute( - queued.cell, - mutation_identity_window(31, 10, 10_000), - Digest::from_bytes([31; 32]), - 20, - RESULT_LIMIT, - |_| Ok(HandlerOutcome::Success(Vec::new())), - )); - assert!(futures_util::poll!(&mut waiting).is_pending()); - let independent = tokio::time::timeout( - std::time::Duration::from_secs(1), - pool.execute( - resident.cell, - mutation_identity_window(32, 10, 10_000), - Digest::from_bytes([32; 32]), - 20, - RESULT_LIMIT, - |transaction| { - let value: i64 = - transaction.query_row("SELECT value FROM counter", [], |row| row.get(0))?; - Ok(HandlerOutcome::Success(value.to_le_bytes().to_vec())) - }, - ), - ) - .await; - drop(waiting); - release.send(()).unwrap(); - running.await.unwrap().unwrap(); - - // Release the blocked worker before asserting so a failed regression - // cannot deadlock the pool's thread join during unwinding. - assert!( - independent.is_ok(), - "a queued job held the idle worker's capacity" - ); - independent.unwrap().unwrap(); - assert!(pool.pending(queued.cell).await.unwrap().is_none()); - for (cell, replica) in [ - (first.cell, first.replica), - (queued.cell, queued.replica), - (resident.cell, resident.replica), - ] { - if let Some(pending) = pool.pending(cell).await.unwrap() { - let prepared = replica - .prepare(None, pending.cuts(), pending.outcome().commit_sequence(), 1) - .await - .unwrap(); - pool.bind_prepared(cell, prepared.clone()).await.unwrap(); - pool.confirm_published(cell, prepared.root()).await.unwrap(); - } - pool.deactivate(cell).await.unwrap(); - } - pool.shutdown().await.unwrap(); -} - -#[tokio::test] -async fn panicking_handler_fences_only_its_cell_and_worker_continues() { - let first = fixture(21); - let second = fixture(22); - let pool = SqlWorkerPool::new(1, 2).unwrap(); - pool.activate(first.cell, first.executor).await.unwrap(); - pool.activate(second.cell, second.executor).await.unwrap(); - - let panic = pool - .execute( - first.cell, - mutation_identity_window(21, 10, 10_000), - Digest::from_bytes([21; 32]), - 20, - RESULT_LIMIT, - |transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - panic!("native handler panic") - }, - ) - .await; - assert!(matches!(panic, Err(Error::NativePanic))); - assert!(matches!( - pool.execute( - first.cell, - mutation_identity_window(22, 10, 10_000), - Digest::from_bytes([22; 32]), - 21, - RESULT_LIMIT, - |_| Ok(HandlerOutcome::Success(Vec::new())), - ) - .await, - Err(Error::Fenced) - )); - - assert!(matches!( - pool.execute( - second.cell, - mutation_identity_window(23, 10, 10_000), - Digest::from_bytes([23; 32]), - 22, - RESULT_LIMIT, - |_| Ok(HandlerOutcome::Success(b"worker survived".to_vec())), - ) - .await - .unwrap(), - WorkerExecution::Pending(_) - )); -} - -#[tokio::test] -async fn active_cell_admission_is_global_and_released_after_drain() { - let first = fixture(1); - let second = fixture(2); - let pool = SqlWorkerPool::new(2, 1).unwrap(); - pool.activate(first.cell, first.executor).await.unwrap(); - assert!(matches!( - pool.activate(second.cell, second.executor).await, - Err(Error::Capacity(_)) - )); - pool.deactivate(first.cell).await.unwrap(); - - let replacement = fixture(2); - pool.activate(replacement.cell, replacement.executor) - .await - .unwrap(); - pool.deactivate(replacement.cell).await.unwrap(); -} - -#[tokio::test] -async fn worker_shutdown_requires_an_empty_pool_and_closes_every_clone() { - let fixture = fixture(3); - let cell = fixture.cell; - let pool = SqlWorkerPool::new(2, 10).unwrap(); - let clone = pool.clone(); - pool.activate(cell, fixture.executor).await.unwrap(); - assert!(matches!(pool.shutdown().await, Err(Error::Control(_)))); - - pool.deactivate(cell).await.unwrap(); - pool.shutdown().await.unwrap(); - assert!(matches!( - clone.pending(cell).await, - Err(Error::RuntimeClosed) - )); - assert!(matches!(clone.shutdown().await, Err(Error::RuntimeClosed))); -} diff --git a/crates/crab-cell-runtime/tests/support/fencing.rs b/crates/crab-cell-runtime/tests/support/fencing.rs deleted file mode 100644 index 362e38f31..000000000 --- a/crates/crab-cell-runtime/tests/support/fencing.rs +++ /dev/null @@ -1,82 +0,0 @@ -//! Canonical expired-session fixture shared by integration suites. - -use crab_cell_runtime::identity::NodeId; -use crab_cell_runtime::identity::{Digest, SessionId}; -use crab_cell_runtime::node::{ - FencedNodeSession, NodeAdvertisement, NodeCapacity, NodeDirectory, NodeFailureDomain, -}; -use crab_ltx::CellStorageLayout; - -pub async fn fence_session( - layout: &CellStorageLayout, - session: SessionId, - claimant: SessionId, -) -> FencedNodeSession { - let fleet = Digest::from_bytes([90; 32]); - let image = Digest::from_bytes([91; 32]); - let release = Digest::from_bytes([92; 32]); - let directory = NodeDirectory::new(layout.clone(), fleet, image, release); - let key = ed25519_dalek::SigningKey::from_bytes(&[93; 32]); - directory - .create( - NodeAdvertisement::sign( - NodeId::from_bytes(*session.as_bytes()), - session, - "https://expired.internal:8081".into(), - fleet, - Digest::from_bytes([94; 32]), - image, - release, - &key, - 1, - 1, - 10_001, - vec![Digest::from_bytes([95; 32])], - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 1, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .unwrap(), - 1, - ) - .await - .unwrap(); - directory - .create( - NodeAdvertisement::sign( - NodeId::from_bytes(*claimant.as_bytes()), - claimant, - "https://claimant.internal:8081".into(), - fleet, - Digest::from_bytes([94; 32]), - image, - release, - &key, - 1, - 10_000, - 20_000, - vec![Digest::from_bytes([95; 32])], - vec![1], - NodeFailureDomain::default(), - NodeCapacity { - free_memory_bytes: 1, - free_disk_bytes: 1, - job_credits: 1, - ..NodeCapacity::default() - }, - ) - .unwrap(), - 10_000, - ) - .await - .unwrap(); - directory - .claim_expired(session, claimant, 10_001) - .await - .unwrap() -} diff --git a/crates/crab-cell-runtime/tests/support/fixtures.rs b/crates/crab-cell-runtime/tests/support/fixtures.rs deleted file mode 100644 index f98c12394..000000000 --- a/crates/crab-cell-runtime/tests/support/fixtures.rs +++ /dev/null @@ -1,54 +0,0 @@ -//! Fixtures shared by more than one integration suite. -//! -//! Every suite compiles this module through `mod support;`, so a fixture that -//! one suite never calls is expected rather than dead code. -#![allow(dead_code)] - -use std::fmt; - -use crab_cell_runtime::cell::executor::MutationIdentity; -use crab_cell_runtime::codec::{BoundedDecoder, BoundedEncoder, WireValue}; -use crab_cell_runtime::identity::RequestId; - -/// Returns the current wall-clock time in milliseconds. -pub fn now_ms() -> i64 { - i64::try_from( - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_millis(), - ) - .unwrap() -} - -/// Returns one mutation identity that stays valid for a minute from now. -pub fn mutation_identity(byte: u8) -> MutationIdentity { - let issued_at_ms = now_ms(); - mutation_identity_window(byte, issued_at_ms, issued_at_ms + 60_000) -} - -/// Returns one mutation identity with an explicit validity window. -/// -/// Suites that drive a clock use this: a fixed window keeps a case reproducible -/// while every other case can start from [`now_ms`]. -pub fn mutation_identity_window( - byte: u8, - issued_at_ms: i64, - expires_at_ms: i64, -) -> MutationIdentity { - MutationIdentity { - request_id: RequestId::from_bytes([byte; 16]), - issued_at_ms, - expires_at_ms, - } -} - -/// Encodes and decodes one bounded wire value, asserting an exact round trip. -pub fn codec_roundtrip(value: T) { - let mut encoder = BoundedEncoder::new(1024 * 1024).unwrap(); - value.encode(&mut encoder).unwrap(); - let bytes = encoder.finish(); - let mut decoder = BoundedDecoder::new(&bytes, 1024 * 1024).unwrap(); - assert_eq!(T::decode(&mut decoder).unwrap(), value); - decoder.finish().unwrap(); -} diff --git a/crates/crab-cell-runtime/tests/support/mod.rs b/crates/crab-cell-runtime/tests/support/mod.rs deleted file mode 100644 index 4b57bd5d8..000000000 --- a/crates/crab-cell-runtime/tests/support/mod.rs +++ /dev/null @@ -1,8 +0,0 @@ -//! Shared integration-test harness. -//! -//! Suites declare `mod support;` and use `crate::support::`. Only -//! genuinely shared fixtures live here; a fixture used by a single suite stays -//! in that suite's module. - -pub mod fencing; -pub mod fixtures; diff --git a/crates/crab-http-server/README.md b/crates/crab-http-server/README.md index 365dd758e..3aeee4d04 100644 --- a/crates/crab-http-server/README.md +++ b/crates/crab-http-server/README.md @@ -211,7 +211,7 @@ Read these sources in order: | [app.rs](src/app.rs) | Admission timeout, repository access, input validation, and HTTP error mapping | | [cells/router.rs](src/cells/router.rs) | Published-root validation plus local-owner restore or authenticated peer dispatch | | [cells/repository.rs](src/cells/repository.rs) | SQLite transaction, submission reservation, number allocation, and typed outcome | -| [crab-ltx](../crab-ltx/README.md) | Verified immutable publication and exact source-loss restore | +| [Cellule LTX](https://github.com/crabbuild/cellule/tree/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx) | Verified immutable publication and exact source-loss restore | Example JSON body, subject to the server's authentication and mutation checks: diff --git a/crates/crab-http-server/deploy/cell-issue-fleet/README.md b/crates/crab-http-server/deploy/cell-issue-fleet/README.md index 23497122b..275f12a59 100644 --- a/crates/crab-http-server/deploy/cell-issue-fleet/README.md +++ b/crates/crab-http-server/deploy/cell-issue-fleet/README.md @@ -344,7 +344,7 @@ snapshots, and owner maps are retained under `overload.*`, `startup.json`, and `five-startup.json`. Healthy ingress does not imply even owner execution. See the -[LTX audit](../../../crab-cell-runtime/docs/ltx-performance-audit.md) for the +[LTX audit](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-runtime/docs/ltx-performance-audit.md) for the execution-distribution and shared-resource gates before scale comparisons. ## Measure public-host actions against RustFS @@ -375,7 +375,7 @@ still three processes on one machine; they use the Compose RustFS service but do not use the 20 Compose node processes. Keep the raw test output to compare RustFS measurements with the in-memory baseline. Serial actions do not establish saturation throughput or a production SLO. See the -[RustFS action measurements](../../../crab-cell-app/performance/2026-09-25-public-host-rustfs.md). +[RustFS action measurements](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-app/performance/2026-09-25-public-host-rustfs.md). To stop this **disposable** project while preserving its data, use `down` without `--volumes`. Removing its volumes deletes the RustFS data, peer diff --git a/crates/crab-http-server/next-architecture/README.md b/crates/crab-http-server/next-architecture/README.md index 2ddcdbbb4..dc02df2b0 100644 --- a/crates/crab-http-server/next-architecture/README.md +++ b/crates/crab-http-server/next-architecture/README.md @@ -1,5 +1,10 @@ # Next-generation crab-http-server: repository SQLite cells and LTX durability +> Historical design record. The server now embeds the pinned +> [Cellule framework](https://github.com/crabbuild/cellule); its current +> source, tests, and qualification contracts live there. Links to Cellule +> below point to the revision used by this server. + Status: target server architecture. Local and optional remote `crab-ltx` mechanics are implemented, and the server now has a statically registered repository issue/comment/label/status/check/settings/Pull/Release module proven @@ -51,7 +56,7 @@ limits and qualification status are recorded in [crab-ltx](crab-ltx.md). Audience: implementers of the HTTP application, storage and publication owners, operators, and reviewers of correctness and migration evidence. -The [embedded Rust Cell runtime specification](../../crab-cell-runtime/docs/README.md) +The [embedded Rust Cell runtime specification](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-runtime/docs/README.md) now owns the low-level shared runtime APIs, Cell formats, primitive contracts and compiled-release lifecycle. It narrows delivery to Rust handlers embedded in this server; no standalone multi-language platform is planned. For overlapping runtime @@ -65,7 +70,7 @@ modules independently. Product HTTP/Git routes remain the only public API, while Cell and primitive capabilities remain private Rust contracts. The exact source-change, registration, route-adapter, compatibility-test and whole-image rollout sequence is the -[native contributor procedure](../../crab-cell-runtime/docs/rust-api.md#add-a-native-feature). +[native contributor procedure](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-runtime/docs/rust-api.md#add-a-native-feature). The Git browse projection is now rebuilt asynchronously from the canonical object-store snapshot. Direct `crab`/Git pushes do not require this server: diff --git a/crates/crab-http-server/next-architecture/celld-and-rust.md b/crates/crab-http-server/next-architecture/celld-and-rust.md index a0fe175d5..4e96d1019 100644 --- a/crates/crab-http-server/next-architecture/celld-and-rust.md +++ b/crates/crab-http-server/next-architecture/celld-and-rust.md @@ -63,7 +63,7 @@ must hold independently. Follower selection is outside Crab's first release. The detailed target extension uses a node-session recovery interlock plus a control-pinned recovery overlay so follower durability remains compatible with Crab's exact roots; see -[follower durability and warm failover](../../crab-cell-runtime/docs/failover-and-followers.md). +[follower durability and warm failover](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-runtime/docs/failover-and-followers.md). Celld's pinned peer transport uses fleet HMAC authentication over private HTTP; network encryption is supplied externally. Crab proposes mutual TLS on its @@ -89,7 +89,7 @@ that a partial port inherits Celld's guarantees. The inspected Celld revision is `10cb1303dac710dcb3b557e318e08c855261f68b`. The implemented -[import inventory](../../crab-ltx/UPSTREAM.md) records this revision, original +[import inventory](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/UPSTREAM.md) records this revision, original hashes, licenses, omitted modules and intentional local changes. Do not depend on a floating `main` branch. diff --git a/crates/crab-http-server/next-architecture/crab-ltx.md b/crates/crab-http-server/next-architecture/crab-ltx.md index 458813676..d92f57820 100644 --- a/crates/crab-http-server/next-architecture/crab-ltx.md +++ b/crates/crab-http-server/next-architecture/crab-ltx.md @@ -1,8 +1,12 @@ # crab-ltx: reuse of Celld's SQLite replication engine +> Historical implementation design. The `crab-ltx` package has moved out of +> this workspace; the active package is +> [cellule-ltx](https://github.com/crabbuild/cellule/tree/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx). + [Design index](README.md) · Local/remote library and issue/comment/label/status/check/settings HTTP integration implemented. -[crab-ltx](../../crab-ltx/README.md) now implements the embedded local SQLite +[crab-ltx](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md) now implements the embedded local SQLite WAL-to-LTX mechanics and optional canonical Cell-root transport. The Cargo member contains a pinned, modified source integration of `celld-ltx`, not a Git dependency or separate daemon. The HTTP server consumes it through `crab-cell-runtime`: repository issue, @@ -27,9 +31,9 @@ performance and process/network fault qualification remain delivery work. | Host facilities | Injectable filesystem/base VFS/clock/executor; shared page-fault worker/cache and I/O/job/recovery/dirty concurrency budgets; one-MiB full-job scratch permits; temporary reservations follow cancelled jobs but are removed from returned long-lived handles | | Server wiring | All repository collaboration metadata, owner/control publication, scheduled owner-bound compaction, exact local-cut pruning, local/remote/idle/stale-owner routing, public HTTP response gating and source-loss restore are wired; immutable Release asset bodies remain object data by design | -Source and usage: [crate README](../../crab-ltx/README.md), -[public API](../../crab-ltx/src/lib.rs), -[import inventory/notices](../../crab-ltx/UPSTREAM.md). +Source and usage: [crate README](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md), +[public API](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/src/lib.rs), +[import inventory/notices](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/UPSTREAM.md). Local proof includes real SQLite, process kill followed by source-directory loss, independent CRC/format vectors and exact snapshot/compaction comparison. It does not qualify the multi-node server. A separate real RustFS Cell round trip @@ -60,9 +64,9 @@ Foreground faults and owner-driven hydration share write/truncate bookkeeping; capture/snapshot reads also use the VFS. Each frame is BLAKE3/CRC verified. The existing full-restore server activation protocol remains a valid initial policy; selecting sparse activation still requires HTTP admission/output-gate wiring. -See the [Cell API and limits](../../crab-ltx/README.md#core-api). -The [Litestream comparison](../../crab-ltx/README.md#litestream-comparison) and -[upstream record](../../crab-ltx/UPSTREAM.md) describe the capabilities and +See the [Cell API and limits](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md#core-api). +The [Litestream comparison](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md#litestream-comparison) and +[upstream record](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/UPSTREAM.md) describe the capabilities and intentional deviations from Litestream and the pinned Celld implementation. The server can now supply one `Host` throughout local resume, Cell recovery, @@ -78,8 +82,8 @@ timers, distributed fencing or a deterministic cluster simulator. The requested capacity is 1K–10K active databases per node, 100–5,000 MB each, with 1,000 TPS aggregate per node. This is a target, not current qualification. -The [resource limits](../../crab-ltx/README.md#resource-limits) and -[verification](../../crab-ltx/README.md#verification) sections record current +The [resource limits](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md#resource-limits) and +[verification](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md#verification) sections record current bounds and remaining qualification work. Raising `Limits` alone is insufficient. The [Celld comparison](celld-and-rust.md) explains the system-level differences. @@ -334,7 +338,7 @@ adaptation passes. Source reuse does not justify publishing an unknown checksum. ### Implemented library API The following signatures summarize callable crate APIs. Full types and a -compilable example live in the [crate README](../../crab-ltx/README.md). +compilable example live in the [crate README](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md). ```rust impl Db { @@ -545,8 +549,8 @@ The next server slice binds that root to owner/control authority and proves owner replacement with empty local disks through an HTTP mutation and reload. These remain [delivery gates](validation-and-delivery.md), not outcomes of the local crate tests. -Run `cargo test -p crab-ltx --locked` with the worktree\'s external -`CARGO_TARGET_DIR`. See the [crate verification scope](../../crab-ltx/README.md#verification) +Run `cargo test -p cellule-ltx --locked` from the Cellule checkout with its +own external `CARGO_TARGET_DIR`. See the [crate verification scope](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md#verification) for the remaining interoperability, fuzz, fault, memory and platform gates. Freeze the encoding capability, capture result shape and local retention rules diff --git a/crates/crab-http-server/next-architecture/current-implementation.md b/crates/crab-http-server/next-architecture/current-implementation.md index cb0fa12d9..49b4127c9 100644 --- a/crates/crab-http-server/next-architecture/current-implementation.md +++ b/crates/crab-http-server/next-architecture/current-implementation.md @@ -96,7 +96,7 @@ and in-progress epochs while deleting immutable rows in bounded batches. ### Replication crate now available -[crab-ltx](../../crab-ltx/README.md) is a workspace member based on pinned, +[crab-ltx](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md) is a workspace member based on pinned, modified Celld source. It supplies owned SQLite writer/capture lifecycle, checksum-bearing LTX, full snapshots, exact verified local restore and complete chain compaction. Empty default features keep the local library provider/runtime @@ -120,7 +120,7 @@ application sequence, schema, scheduler deadline and exact endpoint. Published bootstrap, command and migration batches are reverified and pruned from the local managed session before success escapes. It does not introduce a second SQLite library. See the -[crab-ltx safety model](../../crab-ltx/README.md#safety-model) for API and +[crab-ltx safety model](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md#safety-model) for API and qualification boundaries. Local tests cover commit/rollback, checkpoint/shrink/regrowth, source-directory diff --git a/crates/crab-http-server/next-architecture/validation-and-delivery.md b/crates/crab-http-server/next-architecture/validation-and-delivery.md index 0f48abd83..22cf0676c 100644 --- a/crates/crab-http-server/next-architecture/validation-and-delivery.md +++ b/crates/crab-http-server/next-architecture/validation-and-delivery.md @@ -157,7 +157,7 @@ losing accounting. Named base VFS and independent worker lifecycle are exercised three epochs additionally cover these storage boundaries. This does not simulate RustFS disk/power loss. -Run the [crate's scoped commands](../../crab-ltx/README.md#verification) for tests, +Run the [crate's scoped commands](https://github.com/crabbuild/cellule/blob/70c3c218fafc815b951a40d4df1b50bba15b7d4c/crates/cellule-ltx/README.md#verification) for tests, Clippy and format checks. Existing dependency versions remain unchanged and the workspace resolves a single SQLite linkage. The existing workspace CI will run the new member; broad CI and cross-platform results must be recorded separately. diff --git a/crates/crab-http-server/tests/qualify_compose_cluster.sh b/crates/crab-http-server/tests/qualify_compose_cluster.sh index 94706c7c9..d59a975b1 100755 --- a/crates/crab-http-server/tests/qualify_compose_cluster.sh +++ b/crates/crab-http-server/tests/qualify_compose_cluster.sh @@ -1757,8 +1757,7 @@ if [ "${CRAB_HTTP_CLUSTER_VALIDATE:-false}" = true ]; then if [ "$qualified_image_ref" != source-only ]; then receipt_mode=release fi - CARGO_TARGET_DIR="${CRAB_HTTP_CLUSTER_CARGO_TARGET_DIR:-${TMPDIR:-/tmp}/crab-http-cluster-target}" \ - cargo run --quiet --locked -p crab-cell-runtime --bin qualification_receipt -- \ + python3 "$repo_root/crab/scripts/cellule-qualification.py" run \ validate-cluster "$receipt_path" "$source_revision" "$qualified_image_digest" "$receipt_mode" fi cat "$receipt_path" diff --git a/crates/crab-ltx/AGENTS.md b/crates/crab-ltx/AGENTS.md deleted file mode 100644 index 25dec9a5e..000000000 --- a/crates/crab-ltx/AGENTS.md +++ /dev/null @@ -1,122 +0,0 @@ -# crab-ltx - -Root `AGENTS.md` and `crates/AGENTS.md` apply. Read `README.md` and -`UPSTREAM.md` before changing adapted code. - -## Purpose and ownership - -Managed SQLite WAL capture and exact, checksum-verified LTX recovery. The -optional `replica` feature adds Cell-root publication, bundles, compaction, and -sparse paged SQL. Cell authority, leases, retention policy, and HTTP stay -outside this crate. - -## Read first - -1. `src/lib.rs` — public surface and feature gating. -2. `src/db.rs` — managed connections, transactions, and capture boundary. -3. `src/capture.rs` with `src/capture/{wal,checkpoint,verify}.rs`, and - `src/ltx.rs` — WAL capture and LTX encoding. -4. `src/recovery.rs` — verified plans and exact restore. -5. `src/replica.rs` and `src/replica/` — Cell roots, directories, uploads, and - compaction (see the module map below). -6. `UPSTREAM.md` — Celld lineage, licenses, and review rules for imports. -7. `src/internal.rs` — the unstable inspection surface external fuzzers and - auditors use; `tests/vectors/README.md` records the external fixture - provenance. - -## Module map - -Subsystem roots keep the shared contract; the named child modules own one -concern each. Module files sit beside their root (`foo.rs` + `foo/`). - -- Capture: `capture.rs` + `capture/{wal,checkpoint,verify}.rs`. -- Replica: `replica.rs` + - `replica/{cache,compaction,directory,restore,root,upload,verify}.rs`, - `replica/compaction/scratch.rs`, `replica/directory/{initial,update,relocate}.rs`. -- Environment: `environment.rs` + - `environment/{directory_cache,executor,host,resources,telemetry,tests}.rs`. -- Storage and IO: `pages.rs`, `paged.rs`, `paged_io.rs`, `writable_vfs.rs`, - `writable_vfs/hydration.rs` (asynchronous fetch and owner-thread installation), - `wal.rs`, `bundle.rs`, `codec.rs`, `lz4_block.rs`, `node_frame.rs`, - `cell_layout.rs`. -- Top level: `host.rs`, `types.rs`, `error.rs`, `commit.rs`, `format_tests.rs`. - -## Common changes - -| Task | Start here | Also inspect | -| --- | --- | --- | -| Change WAL capture | `src/capture/wal.rs` | `src/db.rs`, `tests/ltx/crash.rs` | -| Change LTX encoding | `src/ltx.rs`, `src/codec.rs` | `src/format_tests.rs`, `tests/cell/` | -| Change restore or compaction | `src/recovery.rs`, `src/replica/compaction.rs` | `tests/cell/restore.rs`, `tests/ltx/properties.rs` | -| Change the paged VFS | `src/writable_vfs.rs`, `src/paged_io.rs` | `tests/cell/roots.rs` | -| Change host hooks | `src/environment/` | `tests/host/hooks.rs` | -| Change a decoder or the format | `src/codec.rs`, `src/ltx.rs`, `src/internal.rs` | `src/format_tests.rs`, `tests/ltx/vectors.rs`, `fuzz/` | -| Change a durability seam | `src/capture/`, `src/db.rs`, `src/recovery.rs` | `tests/host/hooks/matrix.rs` | - -## Layout and tests - -- `src/` is production code; the environment splits into `host`, `resources`, - `telemetry`, `directory_cache`, and `executor`, with replica-only modules - gated at the declaration. -- Integration suites: `tests/cell.rs`, `tests/ltx.rs`, `tests/host.rs`, each - with modules in the matching directory. No `#[path]` attributes. -- Unit tests for codec, page, VFS, and replica mechanics stay in `src/` and are - listed in `tests-allow-list.txt`; they assert crate-private state that the - public API intentionally does not expose. - -## Invariants - -- Checksum-bearing LTX only: readers reject checksum-disabled files and - zero-checksum continuation markers. -- Capture records the committed WAL boundary; a valid prefix cannot hide a - corrupt later committed frame. -- A commit is never stranded by size: an incremental cut that cannot fit - `max_capture_bytes` is captured as a full database image bounded by - `max_file_bytes`, and only a failure after the cut writer starts fences the - session. -- Restore and compaction install only fresh destinations and verify the exact - requested endpoint. -- Cancellation never pretends to roll back dispatched work: admission and - scratch stay owned until that work finishes. -- Callers branch on `CrabError::classify()`, never on error text. - -## Features and platform - -- `replica` (off by default) adds object-store transport, bundles, compaction, - and sparse paged SQL. `tests/cell.rs` and `tests/host.rs` are gated on it. -- Local capture needs no network; provider examples are documented in - `examples/README.md`. - -## Verification - -```sh -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/ \ - cargo test -p crab-ltx --features replica --locked -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/ \ - cargo test -p crab-ltx --locked -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/ \ - cargo clippy -p crab-ltx --all-targets --features replica --locked -- -D warnings -python3 crab/scripts/check-cell-ltx-layout.py -``` - -External vectors and decoder fuzzing: - -```sh -# Regenerate the celld-written fixtures (see tests/vectors/README.md). -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-ltx-vectors \ - cargo run --release --manifest-path \ - crates/crab-ltx/tests/vectors/generate/Cargo.toml -- crates/crab-ltx/tests/vectors - -# Deep search needs nightly; tests/ltx/vectors.rs replays the same entry points -# on the stable toolchain during `cargo test -p crab-ltx --features replica`. -CARGO_TARGET_DIR=$HOME/Workspace/crabbuild-target/crab-ltx-fuzz \ - cargo +nightly fuzz run ltx -- -max_total_time=600 -``` - -`.github/workflows/crab-ltx-fuzz.yml` runs the same targets per pull request and -on a nightly schedule, seeded from `tests/vectors/`. - -## Related documentation - -`README.md`, `UPSTREAM.md`, `examples/README.md`, `perf/README.md`, and -`../crab-cell-runtime/docs/canonical-ltx-scaling.md`. diff --git a/crates/crab-ltx/CLAUDE.md b/crates/crab-ltx/CLAUDE.md deleted file mode 120000 index 47dc3e3d8..000000000 --- a/crates/crab-ltx/CLAUDE.md +++ /dev/null @@ -1 +0,0 @@ -AGENTS.md \ No newline at end of file diff --git a/crates/crab-ltx/Cargo.toml b/crates/crab-ltx/Cargo.toml deleted file mode 100644 index 74fe16dfa..000000000 --- a/crates/crab-ltx/Cargo.toml +++ /dev/null @@ -1,43 +0,0 @@ -[package] -name = "crab-ltx" -version = "0.1.0" -edition.workspace = true -license.workspace = true -repository.workspace = true -rust-version.workspace = true -publish = false -description = "SQLite LTX capture, object-store replication, and verified recovery for Crab" - -[features] -default = [] -replica = ["dep:crab-storage", "dep:object_store", "dep:bytes", "dep:serde", "dep:serde_json", "dep:tokio", "dep:async-trait", "dep:tokio-util", "dep:futures-util"] - -[dependencies] -rusqlite.workspace = true -thiserror.workspace = true -blake3.workspace = true -tempfile.workspace = true -crc-fast = { version = "=1.10.0", default-features = false, features = ["std"] } -lz4_flex = "=0.11.6" -crab-storage = { workspace = true, optional = true } -object_store = { workspace = true, optional = true } -bytes = { workspace = true, optional = true } -serde = { workspace = true, optional = true } -serde_json = { workspace = true, optional = true } -tokio = { workspace = true, optional = true, features = ["rt-multi-thread", "sync", "time", "net", "macros"] } -tokio-util = { workspace = true, optional = true } -async-trait = { workspace = true, optional = true } -futures-util = { workspace = true, optional = true } - -[dev-dependencies] -object_store = { workspace = true, features = ["fs"] } -proptest = "1" -tokio = { workspace = true, features = ["macros", "rt-multi-thread", "test-util"] } - -[[example]] -name = "rustfs_cell_replica_scale_load" -required-features = ["replica"] - -[[example]] -name = "power_cut_probe" -required-features = ["replica"] diff --git a/crates/crab-ltx/LICENSE b/crates/crab-ltx/LICENSE deleted file mode 100644 index d64569567..000000000 --- a/crates/crab-ltx/LICENSE +++ /dev/null @@ -1,202 +0,0 @@ - - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - - Copyright [yyyy] [name of copyright owner] - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. diff --git a/crates/crab-ltx/LICENSE.pierrec-lz4 b/crates/crab-ltx/LICENSE.pierrec-lz4 deleted file mode 100644 index 438ff7f7d..000000000 --- a/crates/crab-ltx/LICENSE.pierrec-lz4 +++ /dev/null @@ -1,27 +0,0 @@ -Copyright (c) 2015, Pierre Curto -All rights reserved. - -Redistribution and use in source and binary forms, with or without -modification, are permitted provided that the following conditions are met: - -* Redistributions of source code must retain the above copyright notice, this - list of conditions and the following disclaimer. - -* Redistributions in binary form must reproduce the above copyright notice, - this list of conditions and the following disclaimer in the documentation - and/or other materials provided with the distribution. - -* Neither the name of xxHash nor the names of its contributors may be used to - endorse or promote products derived from this software without specific - prior written permission. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" -AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE -IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE -DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE -FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL -DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR -SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER -CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, -OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE -OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. diff --git a/crates/crab-ltx/README.md b/crates/crab-ltx/README.md deleted file mode 100644 index bc8371250..000000000 --- a/crates/crab-ltx/README.md +++ /dev/null @@ -1,883 +0,0 @@ -# `crab-ltx` - -`crab-ltx` is an embeddable Rust library for capturing SQLite WAL commits as -checksum-bearing LTX files and recovering an exact, verified database state. -With the optional `replica` feature, it can prepare immutable Crab Cell roots -in object storage and open them for exact restore or sparse SQL. - -It is **not Litestream packaged as a Rust crate**. Litestream is a standalone -sidecar that monitors a database and operates its replication lifecycle. -`crab-ltx` runs inside the application and leaves scheduling, storage -configuration, authority, retention, and request acknowledgement to its host. - -![Litestream sidecar and crab-ltx embedded architecture](diagram/litestream-vs-crab-ltx.svg) - -## Choose the right tool - -Use Litestream when you want an operational SQLite backup tool with a CLI, -configuration file, background synchronization, provider integrations, -compaction, snapshots, and retention. - -Use `crab-ltx` when a Rust service must: - -- execute SQLite writes and capture their exact WAL commit boundary in one - owned session; -- validate every LTX segment before selecting it for recovery; -- publish immutable objects behind an application-specific authority CAS; -- restore a caller-selected state without listing a bucket for “latest”; or -- lazily open an authenticated Cell root and hydrate it through SQLite. - -Do not use `crab-ltx` as a drop-in Litestream client or point the two systems at -the same replica prefix. They share LTX concepts, not a publication protocol. - -## Litestream comparison - -This comparison was checked against the official -[Litestream v0.5.17 release](https://github.com/benbjohnson/litestream/releases/tag/v0.5.17), -its [`DB`](https://github.com/benbjohnson/litestream/blob/v0.5.17/db.go), -[`Replica`](https://github.com/benbjohnson/litestream/blob/v0.5.17/replica.go), -and [`Store`](https://github.com/benbjohnson/litestream/blob/v0.5.17/store.go) -implementations, and the official -[How it works](https://litestream.io/how-it-works/) guide. - -| Concern | Litestream v0.5.17 | `crab-ltx` | -| --- | --- | --- | -| Deployment | Standalone process next to the application | Library linked into a Rust process | -| Write ownership | Observes an application-owned SQLite database through SQLite and WAL files | All supported SQL writes pass through an exclusive `Db` session | -| Progress | Background monitor loops sync WAL, upload LTX, compact levels, create snapshots, and enforce retention | The host explicitly calls `capture`, `checkpoint`, `prepare`, compaction, and pruning | -| Remote state | `ReplicaClient` lists LTX levels and selects ranges for restore | `CellReplica` opens an exact `RootRef`; it never discovers truth by listing objects | -| Durability boundary | A successful replica sync advances Litestream's replica position | Uploaded immutable objects are only a proposal; the host must publish the root with Cell authority before acknowledging | -| Storage providers | Litestream owns its CLI/config provider integrations | The host supplies a `crab-storage` `Store`; `crab-ltx` binds it to Cell paths with `CellStorageLayout` | -| Restore selection | Latest, TXID, or timestamp is resolved from replica LTX files | The caller supplies a verified plan or an authority-pinned Cell root | -| Retention | Built-in snapshot and LTX retention monitors | Remote pinning, retention, and garbage collection are host policy | -| Format | Uses `superfly/ltx` v0.5.2 | Writes checksum-bearing LTX v3 files using the v0.5.2 sized-block layout and reads that layout plus older checksummed LZ4-frame files | - -The matching LTX dependency means current Litestream can parse the sized-block -encoding used here. It does **not** prove end-to-end interoperability: Crab adds -its own BLAKE3-bound segment metadata, authenticated Cell directory, immutable -root schema, and authority protocol. External Litestream/Celld golden-vector -qualification remains a release gate. See [UPSTREAM.md](UPSTREAM.md) for source -lineage and the exact compatibility boundary. - -For reproducible local capture, compaction, and restore measurements against -the pinned Celld implementation, see the -[performance harness](perf/README.md). - -## Lifecycle - -The embedding runtime owns the steps around the library calls. In particular, -`prepare` does not publish a mutable head and `capture` does not mean remote -durability. - -![Capture, publish, and recover sequence](diagram/capture-publish-recover.svg) - -The safe write path is: - -1. Run a transaction through `Db`. -2. Call `capture()` to produce one or more ordered local LTX segments. -3. Upload and verify them with `CellReplica::prepare()`. -4. Publish the returned root through the embedding runtime's owner/head CAS. -5. Only after that CAS is durable, acknowledge the mutation and prune the exact - captured batch. - -Recovery reverses the boundary: load the authority-pinned `RootRef`, verify its -complete immutable object graph, then restore it or activate sparse SQL. - -`VerifiedRoot::open_read_only` opens a fresh immutable SQLite view over the -exact root's authenticated pages. Call this synchronous opener on a SQLite -worker, separately from the LTX blocking-I/O pool used by page fetches. -The private local file is an empty placeholder; page bodies use the existing -8 MiB shared bounded cache and the managed 64 KiB SQLite cache. No capture -session, WAL, checksum sidecar, or writable database handle is created. -SQLite's immutable VFS flag and read-only handle enforce the boundary even -if a caller disables `query_only`. The owned view closes SQLite before -removing its placeholder. A dispatched opener whose waiter is cancelled must -retain ownership until completion, then drop the unclaimed view. -Use `with_paged_io_deadline` around blocking SQL and `take_io_error` to retain -the provider/checksum cause behind SQLite's I/O error. Cell authority and -freshness checks remain the embedding runtime's responsibility. - -### What Cell authority does - -Cell authority is the publication boundary implemented by `crab-cell-runtime`, -not by `crab-ltx`. It stores one strict, versioned control record for a Cell: -the current incarnation, owner, lifecycle state, revision, and published -`RootRef`. - -For each update, the runtime reads that exact record with its object-store ETag, -builds a named and fully validated transition, and conditionally writes the -complete successor using the observed ETag. There is no blind overwrite or -“latest root” discovery by listing objects. If another owner wins first, the -conditional write conflicts; the runtime reloads the record and rejects or -fences the stale writer. - -This separates two guarantees: - -- `CellReplica` verifies and uploads immutable objects, then returns a root - proposal. -- Cell authority atomically chooses which proposal is the published root for - the current owner and incarnation. - -Only a successful authority CAS makes the root durable truth. The host may then -acknowledge the mutation and prune the exact captured batch. An uploaded root -whose CAS did not succeed remains an unreferenced proposal, never an -acknowledged database state. - -## Features - -| Feature | Default | Adds | -| --- | --- | --- | -| none | yes | Local capture, checkpointing, snapshots, exact verification, restore, and compaction | -| `replica` | no | `crab-storage` transport, Cell roots, bundles, remote compaction, exact-root restore, sparse SQL, and hydration | - -The crate is currently an unpublished workspace library. - -## Local capture and exact restore - -The following example is compiled as a Rust doc test. Both destination -directories exist, and the restored database path does not. - -```rust,no_run -use crab_ltx::{Limits, Db, VerifiedPlan, restore_exact}; - -fn main() -> crab_ltx::Result<()> { - let source = tempfile::tempdir()?; - let restored = tempfile::tempdir()?; - let limits = Limits::default(); - let database_path = source.path().join("repository.sqlite"); - - let mut database = Db::open(&database_path, limits)?; - database.transaction(|transaction| { - transaction.execute( - "CREATE TABLE issues (number INTEGER PRIMARY KEY, title TEXT NOT NULL)", - [], - )?; - transaction.execute( - "INSERT INTO issues VALUES (1, 'Recover this issue from LTX')", - [], - )?; - Ok(()) - })?; - - let captured = database.capture()?; - let plan = VerifiedPlan::new( - &captured.segments, - captured.position, - limits, - )?; - let restored_path = restored.path().join("repository.sqlite"); - let restored_position = restore_exact(&plan, &restored_path)?; - - assert_eq!(restored_position, captured.position); - database.close()?; - Ok(()) -} -``` - -`LocalSegment::new` only describes a selected file and its expected metadata. -It is not trusted until `VerifiedPlan::new` has read the bytes, checked the -BLAKE3 digest and LTX structure, verified the complete checksum-linked chain, -and reconstructed the requested endpoint. The plan owns that verified database -image, so source files may be removed or replaced afterward without changing -what restore, resume, or compaction consumes. - -Run the complete local demonstration from the repository root: - -```sh -export CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-your-worktree" -cargo run -p crab-ltx --example local_roundtrip --locked -``` - -It writes to real SQLite, copies the resulting LTX artifacts across the -transport boundary, deletes the source directory, restores the database, and -queries the recovered row. - -## Preserve application errors - -Use `transaction_with` when the callback can reject a mutation for an -application reason. `TransactionError::Operation` means the callback failed and -the SQLite transaction was rolled back; commit ambiguity or capture failures -use different variants and may fence the session. SQLite can automatically roll -back the entire transaction on `SQLITE_FULL`, interruption, or a ROLLBACK -constraint. The managed writer recognizes that state only when autocommit is -restored and its WAL observer recorded no commit. It preserves the original -operation error, prior committed capture boundary, and disk admission; a -redundant ROLLBACK must not turn a capacity refusal into a fenced session. -See SQLite's [automatic rollback contract](https://www.sqlite.org/c3ref/get_autocommit.html). - -```rust,no_run -use std::io; - -use crab_ltx::{Db, TransactionError}; - -fn rename_issue( - database: &mut Db, - number: i64, - title: &str, -) -> Result<(), TransactionError> { - database.transaction_with(|transaction| { - if title.trim().is_empty() { - return Err(io::Error::new( - io::ErrorKind::InvalidInput, - "an issue title cannot be empty", - )); - } - - transaction - .execute( - "UPDATE issues SET title = ?1 WHERE number = ?2", - (title, number), - ) - .map_err(io::Error::other)?; - Ok(()) - }) -} -``` - -A successful transaction is still only a local SQLite commit. Capture and -publication remain separate durability steps. - -### Grouping capture durability barriers - -`capture()` is the synchronous convenience path: it syncs the LTX file and -its published name before returning. A host that already has a higher-level -acknowledgement barrier can capture several batches with -`capture_deferred()`, then make all of their LTX files durable with one -`durability_barrier()` call: - -```rust,no_run -use crab_ltx::Db; - -fn capture_group(database: &mut Db) -> crab_ltx::Result<()> { - let first = database.capture_deferred()?; - let second = database.capture_deferred()?; - - // Do not acknowledge local durability or close the session until this - // flushes every completed file and seals their directory entries. - database.durability_barrier()?; - - assert!(second.position.txid >= first.position.txid); - Ok(()) -} -``` - -The default `capture()` contract is unchanged. A failed barrier fences the -session, so the host must not acknowledge either batch. Checkpoint and snapshot -operations flush pending deferred files before changing the WAL lifecycle. - -An embedding protocol with a stronger external proof can instead publish the -exact deferred bytes to that boundary and call `prune_captured()` only after -publication succeeds. This is how `crab-cell-runtime` avoids duplicating a -follower fsync or authoritative object-root CAS with a soon-to-be-deleted local -file barrier. Failure before the external proof remains an unknown outcome; -the runtime never acknowledges the local cut alone. Published-cut cleanup -reverifies the local file through a buffered stream and unlinks it without -making that deletion an acknowledgement barrier. A session always uses a fresh -metadata directory, so crash-resurrected cleanup residue remains quarantined. -Streaming avoids retaining the complete compressed file; verification still -decodes all pages and retains the observed page index on the caller's worker. -Verification streams the footer through a bounded buffer and builds replica -lookup entries only when the caller requests them; memory still grows with the -number of observed pages. - -## Checkpoint without losing capture boundaries - -Call `checkpoint` instead of issuing SQLite checkpoint pragmas directly. The -returned batch includes the pending write cut and any additional cut created by -checkpoint maintenance; publish the whole batch before acknowledging it. - -```rust,no_run -use crab_ltx::{CaptureBatch, CheckpointMode, Db}; - -fn capture_and_truncate_wal(database: &mut Db) -> crab_ltx::Result { - database.checkpoint(CheckpointMode::Truncate) -} -``` - -`CheckpointMode::Passive` avoids waiting for other readers. `Truncate` is the -stronger maintenance operation and should run only when the host has budgeted -for it. - -## Resume a verified lineage - -`Db::resume` installs a verified plan into a fresh path and seeds the -next capture with the plan's exact TXID and rolling checksum. It never infers -acknowledged state from an abandoned database directory. - -```rust,no_run -use crab_ltx::{Limits, Db, VerifiedPlan}; - -fn main() -> crab_ltx::Result<()> { - let source = tempfile::tempdir()?; - let destination = tempfile::tempdir()?; - let limits = Limits::default(); - - let mut original = Db::open(&source.path().join("state.sqlite"), limits)?; - original.transaction(|transaction| { - transaction.execute("CREATE TABLE events (value TEXT NOT NULL)", [])?; - transaction.execute("INSERT INTO events VALUES ('first')", [])?; - Ok(()) - })?; - let captured = original.capture()?; - let plan = VerifiedPlan::new(&captured.segments, captured.position, limits)?; - original.close()?; - - let mut resumed = Db::resume( - &plan, - &destination.path().join("state.sqlite"), - limits, - )?; - resumed.transaction(|transaction| { - transaction.execute("INSERT INTO events VALUES ('second')", [])?; - Ok(()) - })?; - let continuation = resumed.capture()?; - - assert!(continuation.position.txid > plan.position().txid); - resumed.close()?; - Ok(()) -} -``` - -## Resume a local image without an origin read - -When a caller owns a local database it already proved against one published -position, `Db::persist_continuation` records what a later session needs to -continue it: the position, page size, page count, and the dense page checksums, -written as two sidecars next to the database. `CellReplica::open_resumed` moves -those files onto a fresh path and seeds the new capture session from them, so no -origin object is read. - -The record is structural, never authoritative. It refuses a database whose WAL -is not checkpointed (the file may sit behind the continuation) and a sparse -activation that is not fully materialized (an unfaulted page is a hole, not -data), and the writing side proves the dense copy still folds to the aggregate -it seeds. The resumed open also checks every checksum-bearing database page -against the recorded dense checksums before SQLite can reuse the image. -Ownership and root identity stay with the caller: only open a resumed database -that a resume record has already matched against the authoritative -control, and discard it (`CellReplica::discard_resumed`) on any mismatch. - -Opening a captured or restored database does not force a synthetic sequence -write. If no application write occurs, a fully hydrated database can close and -resume at the same exact position. The first capture creates a WAL frame only -when needed, then binds the inherited checksum state to that WAL's complete -committed prefix. Later captures retain the usual salt and committed-boundary -checks; SQLite `synchronous=FULL` and resume verification are unchanged. - -## Local durability boundaries - -| Operation | Local barrier | What it proves | May release a Cell response? | -| --- | --- | --- | --- | -| SQLite commit | SQLite WAL sync under `synchronous=FULL` | The local commit reached SQLite's WAL boundary | No | -| `capture()` | LTX file sync, rename, parent sync, then first-cut directory-chain sync | The returned standalone LTX cuts have durable bytes and names | No | -| `capture_deferred()` | No LTX file or name barrier; a sparse writer updates its mutable checksum sidecar without syncing it | The cut is readable for publication, but its LTX durability is pending | No | -| `durability_barrier()` | Pending LTX files, their parent directories, and the first-cut directory chain | Those deferred local cuts are durable | No | -| `persist_continuation()` | New dense checksum file and continuation, each with parent sync | A clean, drained local image can be considered for warm reuse | No | -| Cell root publication or selected follower proof | Runtime owned provider or fleet proof | The exact authoritative root or durable follower tail covers the command | Yes, when runtime checks the matching position | - -The mutable checksum sidecar is local capture state, not a root selector. A -missing or invalid sidecar discards warm reuse; the runtime restores its -authority-pinned root. Process-kill tests do not prove physical power-loss -durability for SQLite, the LTX file, or the filesystem's sync implementation. - -## Preparing a Cell root - -Enable `replica`, construct a `CellStorageLayout` from the application's -existing `crab_storage::Store`, then bind `CellReplica` to exactly one Cell and -incarnation. Provider credentials, leases, and authority stay outside this -crate. - -```rust,ignore -use crab_ltx::{CellReplica, CellStorageLayout, Limits}; -use crab_storage::Store; -use object_store::path::Path; - -fn bind_replica(store: Store) -> crab_ltx::Result { - let layout = CellStorageLayout::new( - store, - Path::from("tenant-a"), - [0x11; 16], // application ID - ); - CellReplica::new( - layout, - [0x22; 32], // Cell ID - [0x33; 16], // incarnation ID - Limits::default(), - ) -} -``` - -```rust,ignore -use crab_ltx::{CaptureBatch, CellReplica, Db, PreparedRoot, RootRef}; - -async fn prepare_root( - database: &mut Db, - replica: &CellReplica, - previous: Option<&RootRef>, - commit_sequence: u64, - schema: u32, -) -> crab_ltx::Result<(PreparedRoot, CaptureBatch)> { - let captured = database.capture()?; - let prepared = replica - .prepare(previous, &captured, commit_sequence, schema) - .await?; - - // `prepared.root()` is a proposal. The embedding runtime must publish it - // with its owner/head CAS before acknowledging the mutation or pruning - // `captured`. - Ok((prepared, captured)) -} -``` - -`CellReplica::prepare` writes only immutable, content-addressed objects. The -embedding runtime publishes `PreparedRoot::root()` through `crab-cell-runtime` -authority. After durable publication it may call -`Db::prune_captured(&captured)` for that exact acknowledged batch. - -Preparation opens each selected capture once and keeps that exact file handle -through verification and every provider retry. Replacing its path therefore -cannot redirect the proposal. The LTX inspection verifies the declared size, -metadata, and digest; multipart upload hashes the complete source again before -publishing the immutable object and rechecks its length afterward. An in-place -mutation fails one of those gates. This path needs no upload scratch or local -write: immutable upload plus the embedding runtime's authority CAS remains the -durability boundary. Up to four capture handles are opened and inspected in -order-preserving parallel waves; predecessor verification progresses alongside -that local work, and final chain validation still waits for both exact inputs. - -`Db` also carries the page index produced while it encodes each fresh capture. -`CellReplica::prepare` can therefore publish that exact capture without decoding -the complete LTX file a second time; the multipart whole-object hash still -proves that the pinned bytes match the encoder's digest. Segments created with -the public `LocalSegment::new` constructor carry no trusted encoder state and -continue through full structural inspection before upload. -The retained indexes share storage across `CaptureBatch` clones and are capped -at 1 MiB per captured batch; descriptor construction, directory updates, and -immutable upload reuse those same bytes without another full index copy. Larger -batches use the inspection fallback. - -Immutable preparation overlaps independent uploads without weakening the root -gate: each LTX body uploads alongside its index, changed directory nodes upload -concurrently, initial directory construction streams nodes in eight-object -waves, and the root document uploads alongside its segment pages. Up to four -captured segments and eight small metadata objects progress concurrently; the -shared host I/O permits remain the process-wide request ceiling. A root proposal -is returned only after every dependency succeeds, so failed work can leave -unreachable content-addressed objects but cannot publish an incomplete root. - -The live RustFS example exercises Cell publication, sparse activation, -compaction, source deletion, and exact recovery. See the -[examples guide](examples/README.md) before running it against a disposable -bucket. - -## Reopen and restore an exact root - -The caller obtains `RootRef` from authenticated authority state. `open_root` -does not list storage or choose “latest”; it verifies the named root and its -complete metadata graph. `restore` then authenticates every page while writing -a fresh destination. - -```rust,ignore -use std::path::Path; - -use crab_ltx::{CellReplica, RootRef}; - -async fn restore_published_root( - replica: &CellReplica, - published: &RootRef, - destination: &Path, -) -> crab_ltx::Result<()> { - let verified = replica.open_root(published).await?; - let restored = verified.restore(destination).await?; - assert_eq!(restored, published.position); - Ok(()) -} -``` - -For backup pinning or garbage-collection marking, traverse the same verified -graph instead of reconstructing object names. The result uses `RootObjectRef` -because each entry is authenticated as a dependency of that exact root. - -```rust,ignore -use crab_ltx::{CellReplica, RootObjectRef, RootRef}; - -async fn objects_to_pin( - replica: &CellReplica, - published: &RootRef, -) -> crab_ltx::Result> { - replica.reachable_objects(published).await -} -``` - -## Activate sparse writable SQL - -A verified root can become writable without first downloading every page. -`prepare_writable` fetches the authenticated checksum directory asynchronously -and dispatches local checksum creation, buffered writes, metadata checks, and -failure cleanup through the bounded host executor. Dispatched jobs retain their -admission until completion even if the caller cancels. -Checksum bases and sparse/immutable placeholders are derived session files, so -activation does not sync them or their names. SQLite retains its normal WAL -durability policy; warm reuse separately writes and syncs a verified continuation. -`open_writable` must then run on the Cell's dedicated SQLite worker. Page faults -fetch and verify missing pages. Direct synchronous embedders can use -`hydrate_step` on their database worker. The Cell runtime instead uses -`prepare_hydration` to select at most 64 missing pages, fetches through -`db::HydrationRead::fetch` outside its SQL worker, then returns the resulting -`db::HydrationBatch` to `install_hydration` on the owning activation. - -The caller admits retained page bytes before fetching and retains that -reservation through installation, including canceled installation waiters. -Demand-prefetched pages are reused. Installation verifies activation identity -and skips pages superseded by checkpointed writes or truncation. Abandoning a -fetch does not advance progress; an installation error fences the database. -SQLite demand faults remain synchronous, and local page installation can -still delay the worker on a slow disk. - -```rust,ignore -use std::path::Path; - -use crab_ltx::{CellReplica, Hydration, Db, RootRef}; - -async fn activate_sparse( - replica: &CellReplica, - published: &RootRef, - destination: &Path, -) -> crab_ltx::Result { - let verified = replica.open_root(published).await?; - let writable = verified.paged().prepare_writable(destination).await?; - let mut database = writable.open_writable(destination)?; - - let Hydration { resolved, total, .. } = database.hydrate_step(128)?; - assert!(resolved <= total); - Ok(database) -} -``` - -The sparse database remains pinned to the selected root. New writes still use -`Db::transaction`, `capture`, immutable preparation, and authority CAS -in that order. - -Sparse opening claims its canonical path under a short registry lock. File -creation, syncs, bridge startup and allocations happen outside that lock, so -one slow activation does not hold up another Cell's registration or teardown. -Failed setup releases its registry claim and leaves created local files -quarantined; it never deletes or adopts an interrupted sparse database. - -## Core API - -### Local capture and recovery - -| API | Contract | -| --- | --- | -| `Db::open` | Claims a fresh exclusive session and owns the writer, control, and read-lock SQLite connections | -| `Db::transaction` | Commits one local SQL transaction; does not claim remote durability | -| `Db::capture` | Returns every new ordered cut plus its exact TXID/checksum endpoint | -| `Db::capture_deferred` | Returns complete, readable LTX files whose durability remains pending | -| `Db::durability_barrier` | Flushes deferred files concurrently, then syncs their parent and the new directory chain once; failure fences the session | -| `Db::checkpoint` | Captures a barrier, runs the selected SQLite checkpoint, and returns every generated cut | -| `Db::snapshot` | Returns an independent full snapshot plus any pending captured cuts | -| `VerifiedPlan::new` | Verifies the complete selected snapshot-plus-delta chain and owns its exact reconstructed image | -| `restore_exact` | Installs a fresh database at exactly the verified endpoint; never overwrites | -| `compact_exact` | Produces a verified full snapshot without deleting its inputs | -| `Db::resume` | Restores a verified plan into a fresh session and continues its TXID/checksum lineage | -| `Db::persist_continuation` | Records the local continuation and dense page checksums a later resumed open seeds from | -| `Db::open_resumed` | Opens a cleanly checkpointed, fully materialized local image and continues its lineage without reading an origin object | - -### Cell replication (`replica`) - -| API | Contract | -| --- | --- | -| `CellReplica::prepare` | Verifies captured cuts and prepares an immutable successor root | -| `CellReplica::prepare_bundle` | Selects and verifies this Cell's rows from a shared bundle | -| `CellReplica::prepare_compaction` | Rewrites an exact range into a representation-only prepared root | -| `CellReplica::open_root` | Reopens one exact root and verifies its scope, chain, metadata, and directory | -| `VerifiedRoot::restore` | Streams an exact verified database into a fresh destination | -| `VerifiedRoot::paged` | Opens authenticated page and page-run reads | -| `CellPagedDatabase::prepare_writable` | Seeds a fresh sparse writable activation at the root's exact position | -| `Db::hydrate_step` | Resolves a bounded number of missing sparse pages on the owner-controlled database worker | -| `Db::prepare_hydration` / `db::HydrationRead::fetch` / `Db::install_hydration` | Splits bounded page selection and owner installation from asynchronous authenticated fetch; caller owns memory admission and scheduling | -| `CellReplica::reachable_objects` | Returns `RootObjectRef` values for the verified immutable dependency set | -| `CellReplica::open_resumed` | Moves a resumable local image onto a fresh path and continues its capture session | -| `CellReplica::discard_resumed` | Removes a local image and its resume sidecars that the caller refused | - -## Safety model - -`crab-ltx` fails closed around state selection and reconstruction: - -- The first plan segment must be a full snapshot. Later segments must be - contiguous, checksum-linked, ordered, and consistent in page size. -- A commit whose delta cannot fit `Limits::max_capture_bytes` is captured as a - full database image bounded by `Limits::max_file_bytes`, not refused after the - commit: the image keeps the commit's TXID, pre-apply checksum, and chain - position, so a large write can never leave a local commit the session cannot - capture. -- Every segment's declared size, BLAKE3 digest, LTX checksum, page ordering, - page coverage, and pre/post database checksum is verified. -- Restore and compaction create a new destination and never replace an existing - database or consult sidecars, local listings, or object listings for truth. -- A capture/checkpoint error fences the managed handle when the commit boundary - can no longer be proven. -- Cell objects are scoped to one Cell incarnation. A `PreparedRoot` is not a - lease, authority update, or durable response gate. -- Cancellation does not roll back work already dispatched to blocking or - object-store workers. The host must await or reconcile the exact root before - retrying. - -### Failure classes - -`CrabError::classify()` returns the contract a caller branches on: - -| Class | Caller action | -| --- | --- | -| `Retryable { after }` | Retry within the caller's own attempt budget, honoring `after` when the provider named one | -| `Capacity` | The request was refused before an acknowledged side effect; free the resource, raise the bound, or split the request | -| `Permanent` | The request cannot succeed with the same inputs or selected state | -| `Ambiguous` | Work may have taken effect; reconcile before retrying | -| `Fenced` | Close the handle and restore authoritative state | - -Callers must not dispatch on error messages, and a declared `Limit` failure is -never a fence. A capture failure raised before the cut writer starts leaves the -session usable with `Db::has_pending_capture()` set; the host must not -acknowledge or serve that commit until a capture succeeds or the session is -discarded. - -LTX CRC64 protects file structure and rolling database state. It is not a -cryptographic authenticator. Crab manifests and Cell objects add BLAKE3 digests; -the embedding service remains responsible for authenticating the manifest or -authority record that selects them. - -## Host responsibilities - -The application must provide the policy a sidecar would normally own: - -- serialize SQL and publication for each database; -- keep all mutations inside `Db` and avoid direct checkpoints, `ATTACH`, - pager-changing pragmas, and edits to `_litestream_seq` or `_litestream_lock`; -- publish roots through owner/incarnation/sequence authority before responding; -- schedule capture, checkpoint, hydration, compaction, and retries; -- configure provider access through `crab-storage`; -- budget local disk, scratch space, blocking work, remote I/O, and active SQLite - connections; and -- pin live roots and own remote retention and garbage collection. - -Local calls are synchronous and should run on a dedicated database thread or a -bounded blocking executor. `&mut Db` serializes access within one handle; -it does not create a distributed lock. - -The database path, SQLite sidecars, and private -`.-crab-ltx` directory must have one owner. Parent directories must -already exist. Reopening a prior session directory is intentionally refused; -recover the authoritative plan or Cell root into a fresh directory instead. - -## Resource limits - -`Limits::default()` admits a 512 MiB database, 64 MiB per capture, 512 MiB per -input/output file, 1 GiB across a plan or retained captures, and 1,024 segments. -These are per-operation correctness bounds, not an RSS quota. - -`max_capture_bytes` bounds one incremental cut; a commit that cannot fit it is -captured as a full database image bounded by `max_file_bytes`, and the -publication path admits each captured segment up to `max_file_bytes` only when -its index proves full-page coverage. - -One command that publishes a fresh root uploads a bounded set of immutable -objects: the segment body, its index, the changed directory node, the root -document, and any segment page. `CellReplica::publication_cost` and -`take_publication_cost` report the exact object count and bytes per root so a -host can budget object-store cost per command instead of inferring it from the -database size. The local measurement frozen in -`tests/cell/roots/lifecycle.rs` is five objects per small append (about 7 KiB -for a 4 KiB payload); provider-scale cost distributions remain outstanding. - -Each live `VerifiedPlan` retains one reconstructed database image, bounded by -`max_database_bytes`, plus its checksum state and segment metadata. Drop plans -after restore, resume, or compaction; services that build several plans at once -must admit their combined decoded size rather than only their compressed LTX -input size. - -Checksum candidates retain an isolated overlay until their LTX cut is sealed. -The successful merge retires the predecessor and reuses an exclusively owned -memory array for fixed-size updates. Growth and retained shared snapshots may -still allocate; large truncations release excess capacity. Restored capture -reads overwritten checksums through one 4 KiB local window per cut, while -truncation and clean-handoff scans retain their 64 KiB sequential buffers. -A sealed merge consumes the changed-page overlay, releasing its hash-table -allocation so later small cuts do not clone historical capacity. Unmerged -recovery overlays retain their normal clone semantics. -A failed sidecar merge fences the session. This changes local bookkeeping, -not the LTX format or the authenticated metadata walk required for activation. - -Each open `Db` retains three SQLite connections with a 64 KiB page-cache -target per connection. `Host` can share disk, I/O, blocking-job, recovery, -dirty-job, scratch, and telemetry admission across many databases. Sparse page -read-ahead is capped at 64 pages or 1 MiB per request, and the shared decoded -page cache is capped at 8 MiB. - -Demand reads and asynchronous hydration stop a missing prefix before pages -already cached in the same view. This avoids transferring and decoding an -overlapping prefetched suffix when SQLite visits fragmented page ranges. -Eviction can cause later misses; cache bytes never replace exact-root page -authentication. New immutable views still have separate demand-cache identities. - -`Host::with_directory_cache(root).await?` opens persistent cache membership on -an admitted blocking job. Restart reads at most 16 MiB of index input and -retains at most 16,384 entries within the shared disk budget. The runtime -creates one cache owner beside the activation's database and shares its host -through recovery, SQLite and publication. Cancellation keeps dispatched work -and its reservations alive until completion. Local capture APIs remain -synchronous and retain their caller-owned worker contract. - -Verified directory reads release object-store admission before persisting a -cache fill. Fills use immediate blocking-job admission and skip persistence -when that pool is busy, keeping pending node buffers bounded without a second -queue. Verified bytes return after dispatch, without waiting for local syncs. -Concurrent fills of one key are deduplicated, and cache lookups skip busy jobs -or fill locks. Accepted fills retain their buffers, job and disk accounting -until completion, including canceled readers, dispatch rejection and panics. -Reopening a cache excludes outstanding fills from temporary-file cleanup. - -Call `Host::drain_cache_fills().await` after stopping replica work and before -reusing its local directories or stopping its executor. `CellRuntime::shutdown` -does this after draining Cells and workers. A canceled drain can be retried; -the host can subsequently serve a new runtime. Cache persistence is best -effort, so skipped entries may require another verified origin read. Fills -still share the blocking pool with required work; this does not establish -foreground latency isolation or reduce membership-index rewrite cost. - -Cell compaction buffers sequential index reads within a combined 960 KiB -budget and dispatches bounded merge batches through `Host` jobs. Scratch files -and admission remain owned through canceled jobs and cleanup. Range compaction -spools only selected indexes/bodies, then streams new locators through the -authenticated directory. It retains newer page versions, including disjoint -segments in one bundle, and reuses unchanged branches. Traversal retains -bounded state per tree level and at most eight pending node uploads. Scratch -covers the selected inputs, worst-case encoded output and both index copies; -it does not reserve two complete database images. Full-range compaction still -visits the full directory. These bounds do not establish foreground latency -isolation or sustained publisher capacity. - -Large-database and multi-tenant capacity still require workload-specific -measurement. The existing tests prove bounded correctness behavior; they do not -establish a 10,000-database or 1,000-TPS production capacity claim. - -## Verification - -From the repository root, choose a target directory unique to this checkout: - -```sh -export CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-your-worktree" - -cargo test -p crab-ltx --locked -cargo test -p crab-ltx --features replica --locked -cargo test -p crab-ltx --doc --features replica --locked -cargo clippy -p crab-ltx --all-targets --locked -- -D warnings -cargo clippy -p crab-ltx --features replica --all-targets --locked -- -D warnings -cargo fmt -p crab-ltx -- --check -``` - -The suite covers real SQLite commits and rollbacks, checkpoints, database -growth and truncation, cold restore, process death followed by source loss, -snapshot/compaction byte identity, both supported page encodings, malformed -chains, checksum failures, exact Cell roots, bundles, sparse activation, -hydration, remote compaction, and provider/cache lifecycle boundaries. - -After configuring the existing [RustFS test environment](examples/README.md), -run the cache-admission fault test against its real object store: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-your-worktree" \ - cargo test -p crab-ltx --features replica --locked --test host \ - rustfs_directory_cache_fill -- --ignored --nocapture -``` - -It pauses cache fsync with one origin permit and one/two blocking slots, -verifies the original and concurrent reads return, and restores exact SQLite -state. A canceled drain retains admission; cache reopen waits for persistence. -Each run uses a unique `crab-ltx-tests/cache-admission/` prefix and retains its objects. -The pause/deadline assertions are concurrency proof, not latency percentiles. - -The sparse-activation isolation test uses the same environment: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-your-worktree" \ - cargo test -p crab-ltx --features replica --locked --test host \ - rustfs_sparse_activation_io -- --ignored --nocapture -``` - -It pauses file sync, parent sync and bridge startup in turn while a different -Cell opens and another reads and closes. Each Cell must return its own value -from the exact selected root. Each run retains objects under a unique -`crab-ltx-tests/activation-registry/` prefix. The regular activation suite also -covers conflicting path claims, setup failures and refusal of leftover files. - -Range compaction can also be checked against that real object store: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-your-worktree" \ - cargo test -p crab-ltx --features replica --locked --test cell \ - rustfs_range_compaction_preserves_exact_native_and_bundled_roots -- --ignored --nocapture -``` - -It compacts two updates on larger bases with a cold metadata cache and one -MiB of scratch admission. Native and shared-bundle inputs, 512/4096-byte pages, -multiple directory levels, and a newer overwrite must restore byte-identically. -It bounds origin reads and uploaded objects, retaining its unique -`crab-ltx-tests/range-compaction/` prefix. These are work and correctness checks, -not service latency percentiles. - -The demand-read regression uses the same RustFS environment: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-your-worktree" \ - cargo test -p crab-ltx --features replica --locked --test cell \ - rustfs_fragmented_snapshot_demand_reads_do_not_refetch_cached_frames -- --ignored --nocapture -``` - -It changes a separate counter, opens the resulting immutable view, and reads -the unchanged payload at 512/4096-byte page sizes. Payload hashes must match, -origin byte ranges must not overlap while the working set fits the isolated -cache, and a repeated scan must issue no range reads. Objects remain under a -unique `crab-ltx-tests/read-ahead/` prefix. This proves bounded transfer work, -not a service latency percentile or cache reuse across different views. - -The suite also ships the independent half of the format proof: - -- `tests/vectors/` holds snapshot files written by the pinned upstream Celld - encoder and by the `superfly/ltx` v0.5.2 Go reference writer Litestream uses; - `src/format_tests.rs` decodes them, restores their exact image, and requires - the sized-block files to be byte-identical to this crate's writer. -- `tests/ltx/vectors.rs` replays every truncation and deterministic mutation of - those vectors through the same decoders `fuzz/fuzz_targets/` drives, so a - decoder panic fails the stable-toolchain test run. -- `tests/host/hooks/matrix.rs` injects ordered failures at the capture, barrier, - checkpoint, restore, compaction, and publication seams and asserts the error - class plus the durable outcome. - -The nightly [`crab-ltx fuzz`](../../.github/workflows/crab-ltx-fuzz.yml) workflow -runs the same targets on a schedule and per pull request, seeded from these -vectors, and the stable replay stays in the normal test run. - -Still required before a broad production-readiness claim: exhaustive filesystem -and power-loss faults beyond the injected matrix, broader platform/provider CI, -and measured latency, memory, scratch, and concurrency qualification at fleet -scale. - -## Provenance and compatibility - -This is Crab-owned, modified source derived from the Apache-2.0 Celld LTX -implementation—not an unmodified vendor directory or a floating dependency. -The readable source inventory, upstream revisions, deliberate adaptations, -license obligations, and future-import checklist live in -[UPSTREAM.md](UPSTREAM.md). - -Compatibility summary: - -- readers accept checksum-bearing sized-block LTX and the older checksummed - LZ4-frame representation; -- writers emit only the v0.5.2 sized-block representation; -- checksum-disabled LTX is rejected; -- retired standalone Crab epoch-head and page-map layouts are not read; and -- Cell root JSON, authenticated directories, bundle envelopes, and authority - records are Crab-specific contracts. diff --git a/crates/crab-ltx/UPSTREAM.md b/crates/crab-ltx/UPSTREAM.md deleted file mode 100644 index 0e0a7cc4b..000000000 --- a/crates/crab-ltx/UPSTREAM.md +++ /dev/null @@ -1,191 +0,0 @@ -# Upstream sources and compatibility - -`crab-ltx` contains adapted upstream code. This page explains where it came -from, what Crab changed, which licenses must travel with it, and how to review a -future upstream import. - -For usage and the Litestream sidecar comparison, start with -[README.md](README.md). - -## Source lineage - -The direct source snapshot is the unpublished `crates/ltx` package from -[`denoland/celld`](https://github.com/denoland/celld): - -| Field | Value | -| --- | --- | -| Pinned revision | [`10cb1303dac710dcb3b557e318e08c855261f68b`](https://github.com/denoland/celld/tree/10cb1303dac710dcb3b557e318e08c855261f68b/crates/ltx) | -| Imported | 2026-09-13 | -| Original package | `celld-ltx` `0.0.0`, unpublished | -| Crab package | `crab-ltx` `0.1.0`, unpublished | -| Ownership now | Modified, Crab-owned source; not a vendor mirror or floating dependency | - -Celld's Rust implementation was informed by -[`rustyriver`](https://github.com/mikenomitch/rustyriver), a from-scratch Rust -implementation of Litestream v0.5 and LTX. The wire-format reference is -[`superfly/ltx` v0.5.2](https://github.com/superfly/ltx/tree/v0.5.2). - -Crab's current behavioral comparison was checked against -[Litestream v0.5.17](https://github.com/benbjohnson/litestream/releases/tag/v0.5.17). -That release also uses `superfly/ltx` v0.5.2. Sharing that dependency version -means the current Litestream decoder understands the sized-block representation -written by `crab-ltx`; it does not make the two replica layouts or publication -protocols compatible. - -```text -Litestream v0.5 ──────┐ - ├── rustyriver ── Celld crates/ltx ── crab-ltx -superfly/ltx v0.5.2 ──┘ │ - └─ Crab Cell roots, - authority and storage -``` - -## Licenses and attribution - -Distributions containing `crab-ltx` must retain these attributions: - -| Work | Attribution | License / version | -| --- | --- | --- | -| Celld | Celld contributors | Apache License 2.0; pinned revision above | -| rustyriver | Copyright 2026 The rustyriver authors | Apache License 2.0; Celld snapshot dated 2026-08-03 | -| Litestream | Copyright Ben Johnson and the Litestream authors | Apache License 2.0; v0.5 lineage | -| LTX reference implementation | Copyright Superfly, Inc. | Apache License 2.0; v0.5.2 | -| LZ4 block implementation | Copyright 2015 Pierre Curto | BSD 3-Clause; `pierrec/lz4` v4.1.23 lineage | - -The complete texts are [LICENSE](LICENSE) and -[LICENSE.pierrec-lz4](LICENSE.pierrec-lz4). Keep both with source and binary -distributions that include this crate. The pinned Celld subtree has no `NOTICE` -file. Celld's root `LICENSE.tokio` applies to source outside this import; no -Tokio runtime source was copied into `crab-ltx`. - -## What was adapted - -The imported files were not kept as a parallel source tree. Their responsibilities -were moved behind Crab-owned APIs: - -| Celld source | Crab disposition | -| --- | --- | -| `lib.rs`, `db.rs`, `wal.rs` | WAL validation and capture split across `lib.rs`, `capture.rs`, `capture/`, `wal.rs`, `db.rs`, and `types.rs` | -| `ltx.rs`, `codec.rs`, `lz4_block.rs` | Strict LTX parsing, dual decoding, sized-block encoding, and checked LZ4 helpers | -| `compactor.rs` | Exact-input local and Cell compaction with endpoint verification | -| `host.rs` | Injectable filesystem, clock, SQLite VFS, disk admission, telemetry, executor, and worker contracts in `environment.rs` | -| `paged.rs`, `paged_vfs.rs` | Private authenticated page access plus writable sparse and immutable read-only Cell VFS modes | -| `bundle.rs`, `client/bundle.rs` | Checked CRB1 bundles and exact Cell-scoped recovery overlays | -| `replica.rs`, `replica_compactor.rs` | Design reference only; the standalone epoch-head API was removed | -| `client/epochs.rs`, `client/mod.rs`, `client/object_store.rs` | Replaced by exact Cell roots and existing `crab-storage` transport | -| `compaction_level.rs` | Scheduling remains an embedding-runtime responsibility | - -Source headers identify adapted files. The original-byte SHA-256 inventory used -for the import review remains recoverable from repository history at the import -commit; it is not a runtime or compatibility contract. - -## Deliberate Crab changes - -### Embedded ownership - -- `Db` owns the SQLite writer, control connection, read lock, WAL commit - observation, and fresh local session claim. -- No background daemon, provider URL parser, credential loader, HTTP service, - retention loop, or scheduler is included. -- Local APIs are synchronous. The embedding service supplies its database thread - or bounded blocking executor. - -### Exact state selection - -- Recovery accepts an explicit verified plan or authority-pinned `RootRef`. -- Bucket listing, “latest” discovery, local leftovers, and mutable epoch heads - never select authoritative state. -- Restore and compaction install only fresh destinations and verify the exact - requested endpoint. - -### Stronger verification - -- Writers emit checksum-bearing LTX v3 files using the v0.5.2 sized-block page - representation. Readers also accept the older checksummed LZ4-frame encoding. -- Checksum-disabled LTX and zero-checksum continuation markers are rejected. -- Capture records the application's committed WAL boundary so a valid prefix - cannot hide a corrupt later committed frame. -- Verification checks BLAKE3 metadata, the complete LTX structure, page order - and coverage, every pre/post rolling database checksum, and the final image. - -### Capture representation and failure contract - -- A commit whose delta cannot fit `Limits::max_capture_bytes` is captured as a - full database image bounded by `Limits::max_file_bytes` instead of failing - after the commit. The image keeps the delivered TXID, pre-apply checksum, and - chain position, so it stays a valid successor cut and no oversized write can - strand a session. -- The publication path admits a segment above the incremental bound only when - its index proves full-page coverage, in the native and bundle paths alike. -- `CrabError::classify()` publishes the retry, capacity, permanent, ambiguous, - and fenced contract. Callers branch on the class, never on error text. - -### Cell replication - -- `CellReplica` writes immutable content-addressed LTX, index, directory, bundle, - and root objects scoped to one Cell incarnation. -- Cell authority, ownership, leases, command acknowledgement, pinning, retention, - and deletion stay in `crab-cell-runtime` and the server composition layer. -- `PreparedRoot` is only a proposal for authority CAS. It is never a mutable head - and never authorizes an HTTP response by itself. -- Sparse reads authenticate directory paths, compressed frames, page numbers, - and rolling checksums. Writable activation continues from the pinned root's - exact TXID/checksum. - -### Bounded execution - -- File and WAL reads are bounded by `Limits`; large Cell work uses file-backed - scratch, range reads, bounded frame batches, and replayable upload sources. -- `Host` exposes shared I/O, blocking-job, recovery, dirty-job, scratch, local - disk, and runtime-ledger admission. -- Cancellation does not pretend to roll back dispatched work. Admission and - scratch stay owned until that work actually finishes. -- Each managed SQLite connection uses a 64 KiB page-cache target; one - `Db` retains three connections. - -## Compatibility boundary - -The following are compatible at the file-decoder level, subject to independent -fixture qualification: - -- checksum-bearing LTX v3 headers and trailers; -- the v0.5.2 sized-block page representation; and -- the older checksummed LZ4-frame representation accepted by Crab's reader. - -The following are Crab-specific and must not be inferred from Litestream or -Celld compatibility: - -- `SegmentInfo` BLAKE3 expectations and verified plans; -- Cell object paths, root JSON, descriptor pages, and authenticated radix - directories; -- CRB1 bundle routing and recovery overlays; -- owner/incarnation/sequence authority and response durability; and -- remote pinning, retention, and collection policy. - -Older Litestream releases before the `superfly/ltx` v0.5.2 update cannot decode -the sized-block representation despite the unchanged LTX file-version number. -Use Litestream v0.5.16 or newer for format experiments, and do not treat that as -a supported shared-replica configuration. - -The removed standalone Crab epoch-head, public page-map, read-only VFS, and -scheduler layouts remain outside the Cell graph. `crab-ltx` intentionally has -no compatibility reader or alias for them. The shipped-contract decision and -historical object-layout evidence are recorded in -[`standalone-replication-audit.md`](../crab-cell-runtime/docs/standalone-replication-audit.md). - -## Reviewing a future import - -Do not replace the crate wholesale. For each upstream change: - -1. Compare the changed function together with its WAL, checkpoint, restore, and - compaction callers. -2. Recheck all licenses, notices, copied headers, and dependency versions. -3. Preserve mandatory checksums, the committed-WAL boundary, exact-plan - validation, source errors, cancellation ownership, and drop ordering. -4. Keep provider construction, authority, retention, and scheduling outside this - library unless Crab deliberately changes that architecture. -5. Run real-SQLite capture/checkpoint/recovery tests, malformed-input tests, - process-kill recovery, Cell-root and sparse-VFS suites, and external format - vectors before claiming compatibility. -6. Record the new revision and explain every retained, rejected, or modified - upstream behavior here. diff --git a/crates/crab-ltx/api-prelude.txt b/crates/crab-ltx/api-prelude.txt deleted file mode 100644 index aecfe50fa..000000000 --- a/crates/crab-ltx/api-prelude.txt +++ /dev/null @@ -1,51 +0,0 @@ -CaptureBatch -CaptureTiming -CellObjectKind -CellPagedDatabase -CellReplica -CellStorageLayout -CellWritableDatabase -CheckpointMode -CrabError -Db -DirectoryCacheStats -DiskBudget -DiskBudgetAdmission -DiskReservation -FailureClass -Host -HostResourceAdmission -HostResourceKind -HostResourcePermit -Hydration -LimitKind -Limits -LocalSegment -LtxPhase -LtxReadOrigin -LtxRequestOutcome -LtxTelemetry -MANAGED_CONNECTION_PAGE_CACHE_BYTES -MANAGED_SQLITE_CONNECTIONS -NodeFrameScope -Position -PreparedRoot -PublicationCost -QueryError -ReadOnlyRoot -RecoveryOverlay -Result -RootObjectRef -RootRef -ScratchMonitor -SegmentInfo -TransactionError -VerifiedNodeFrame -VerifiedPlan -VerifiedRoot -compact_exact -encode_node_frame -inspect_node_frame -restore_exact -rusqlite -with_paged_io_deadline diff --git a/crates/crab-ltx/diagram/capture-publish-recover.svg b/crates/crab-ltx/diagram/capture-publish-recover.svg deleted file mode 100644 index dd30b2303..000000000 --- a/crates/crab-ltx/diagram/capture-publish-recover.svg +++ /dev/null @@ -1,73 +0,0 @@ - - Crab LTX capture, publication, and recovery sequence - The application writes through Db, prepares immutable objects through CellReplica, publishes the root through Cell authority, and later recovers only that authority-pinned root. - - - - - - - - - - - - Capture → prepare → publish → recover - Immutable upload is not acknowledgement. Authority publication is the durability boundary. - - - - - - - - - transaction(...) - - commit SQLite WAL - - capture() - - CaptureBatch + exact Position - - prepare(previous, batch, sequence, schema) - - write verified immutable objects - - content digests - - PreparedRoot (proposal only) - - owner/head compare-and-swap - - durably published exact RootRef - - prune_captured(exact batch) - - open_root(authority-pinned RootRef) - - verify graph, restore or sparse-read - - - - Host runtime - - - Db - - - CellReplica - - - Cell authority - - - Object store - immutable only - - - - DO NOT ACKNOWLEDGE AFTER CAPTURE OR UPLOAD - Only a successful authority CAS selects the durable root. - On timeout, reconcile the exact proposed root before retrying. - diff --git a/crates/crab-ltx/diagram/capture-publish-recover@2x.png b/crates/crab-ltx/diagram/capture-publish-recover@2x.png deleted file mode 100644 index 40f450797..000000000 Binary files a/crates/crab-ltx/diagram/capture-publish-recover@2x.png and /dev/null differ diff --git a/crates/crab-ltx/diagram/litestream-vs-crab-ltx.svg b/crates/crab-ltx/diagram/litestream-vs-crab-ltx.svg deleted file mode 100644 index a8f25fcd7..000000000 --- a/crates/crab-ltx/diagram/litestream-vs-crab-ltx.svg +++ /dev/null @@ -1,119 +0,0 @@ - - Litestream sidecar and crab-ltx embedded architecture - Litestream operates background replication beside an application. Crab-ltx is embedded in the host runtime and prepares immutable data. Cell authority validates the observed owner and version, atomically publishes the selected root, and rejects stale writers. - - - - - - - - - - - - - - Two LTX architectures, two ownership models - The important difference is who drives progress and who selects durable truth. - - - LITESTREAM · SIDECAR - Background process owns replication scheduling and maintenance - - - - - - observe WAL - sync loop - upload / list - - - - Application - owns SQL connections - - - - SQLite database - main file + WAL - - - - Litestream - monitor + checkpoints - compaction + retention - - - - Local LTX levels - replica position - - - - - - - - - - - Replica store - - - - CRAB-LTX · EMBEDDED - Host schedules work; Cell authority serializes ownership and root publication - - - - - - - - capture - immutable upload - root proposal - publish RootRef with observed owner + ETag - CAS success gates acknowledgement - - - - Host runtime - application + policy - - - - Db - SQLite + WAL ownership - verified local cuts - - - - CellReplica - verify + prepare - never publishes authority - - - - - - - - - - - Immutable objects - - - - - Cell authority - versioned owner + root record - valid transition + exact ETag CAS - selects the published RootRef - conflict rejects a stale writer - - Shared concept: checksum-linked LTX ranges. Different contract: lifecycle, layout, authority, and restore selection. - diff --git a/crates/crab-ltx/diagram/litestream-vs-crab-ltx@2x.png b/crates/crab-ltx/diagram/litestream-vs-crab-ltx@2x.png deleted file mode 100644 index 6699a90e3..000000000 Binary files a/crates/crab-ltx/diagram/litestream-vs-crab-ltx@2x.png and /dev/null differ diff --git a/crates/crab-ltx/examples/README.md b/crates/crab-ltx/examples/README.md deleted file mode 100644 index c989f681f..000000000 --- a/crates/crab-ltx/examples/README.md +++ /dev/null @@ -1,216 +0,0 @@ -# `crab-ltx` examples - -These programs demonstrate the canonical Rust Cell persistence boundary. Local -capture remains usable without the `replica` feature; object-store examples use -`CellReplica` and never publish a standalone epoch head. - -| Example | Demonstrates | -| --- | --- | -| `local_roundtrip` | Local WAL capture, verified plan construction, exact restore, and SQL verification | -| `rustfs_cell_replica_scale_load` | Immutable publication, sparse activation phase measurements, source deletion, exact restore, and full-range compaction with checksum verification | -| `power_cut_probe` | Exact capture and clean-continuation checkpoints for an external power-cut controller | - -Run the local example from the repository root: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-main" \ - cargo run -p crab-ltx --example local_roundtrip --locked -``` - -## RustFS scale workload - -Provision a disposable, pre-created RustFS bucket and export its endpoint and -credentials outside tracked files: - -```sh -export CRAB_LTX_TEST_BUCKET=crab-ltx-examples -export CRAB_LTX_TEST_ENDPOINT=http://127.0.0.1:9000 -export AWS_ACCESS_KEY_ID="" -export AWS_SECRET_ACCESS_KEY="" -export CRAB_LTX_WORKLOAD_ROOT="$HOME/Workspace/crab-ltx-workloads" -``` - -The default workload grows a 5 GiB incompressible SQLite database. Keep source, -restore, and compaction scratch files on the external workspace volume: - -```sh -CRAB_CELL_LTX_TARGET_BYTES=$((5 * 1024 * 1024 * 1024)) \ -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-main" \ - cargo run --release -p crab-ltx --features replica \ - --example rustfs_cell_replica_scale_load --locked -``` - -The example uses a unique Cell storage prefix, prepares each capture through -`CellReplica`, prunes only the exact batch after the prepared root is verified, -deletes the source database, restores the published root, compacts its complete -range, and compares source and restored BLAKE3/length. It never lists or deletes -remote objects. Run it only against a disposable bucket and let the bucket -owner clean up remote data. - -After deleting the source, the example emits 18 JSON lines with -`measurement: "sparse_activation"`: three rounds at one, four, and eight shared -I/O slots, each with a fresh metadata cache and a second activation reusing it. -Slot order reverses in the middle round. Each line separates exact-root open, -checksum preparation, writable open (including blocking-worker dispatch), and -the first row-length query. Each phase reports elapsed microseconds, storage -read calls, and returned bytes. Hydrated-page and page-fault counts show how -much of the database the initial SQL access actually touched. - -Fresh `Store` identities exclude previously cached immutable metadata; provider -connections and RustFS caches remain warm. The reused pass retains bounded -metadata caches and uses a fresh sparse destination, so a database larger than -the metadata cache may still require origin reads. There is no persistent -directory cache in this probe. Read calls include the Store read API's GET, -range, and HEAD observations, not provider-internal retries. Three samples per -slot/cache pair are diagnostics, not p95/p99 or a supported latency limit. - -For a small real-provider smoke, use `CRAB_CELL_LTX_TARGET_BYTES=33554432` -with the same command. The final load, exact restore, compaction, and compacted -restore timers cover separate phases; digest comparison is outside restore -timing. Retain stdout with the source revision, image/provider version, -architecture, filesystem, and resource limits when comparing runs. - -### Concurrent activation and first-write recovery - -Pass `--activation-cells 4` (range 1–16) to prepare that many distinct Cell -graphs from the same source capture stream. The default run keeps its existing -single-Cell workload. Each additional Cell uploads its own scoped objects, so -choose the target size with total provider storage in mind: - -```sh -CRAB_CELL_LTX_TARGET_BYTES=33554432 \ -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-main" \ - cargo run --release -p crab-ltx --features replica \ - --example rustfs_cell_replica_scale_load --locked -- --activation-cells 4 -``` - -After the original eighteen activation samples, four rounds run with concurrent -activation limits `1, 4, 4, 1`. Every round uses the same four authenticated -roots and fresh sparse files. New per-Cell Store identities discard prior -metadata caches; the provider connection pool, provider caches and default -Host admission remain shared. Each Cell's first application SQL replaces the -first 1 MiB payload with a known compressible Cell-specific value, captures the update -through `capture_deferred`, and prepares the next immutable root. A distinct -mutation ID in each replacement prevents later rounds from measuring a repeated -immutable upload. No payload query or hydration precedes the write; its timer -includes any demand page faults. SQLite's own writable-open reads are recorded -separately. The original eighteen activation samples still measure first reads. - -`activation_burst` marks `access_order: "write_first"` and records each -Cell/root identity, dispatch delay, time to the first completed transaction and -prepared root, plus separate root-open, checksum, writable-open, mutation, -capture and root-prepare read counters/timers. Older reports without -`access_order` measured a payload read before the mutation; keep them separate. -These read counters observe Store GET/range/HEAD calls and bytes; they exclude -provider-internal retries. `prepared_objects` and `prepared_bytes` report the -replica publication ledger separately. The mutation timer is a -local SQLite transaction; the prepared root has not passed a runtime authority -CAS and is not an application acknowledgement. - -Verification starts only after every Cell in the round reaches its prepared -cut, so a full verification restore cannot contaminate another Cell's measured -activation. It removes the local database and captured cut, independently -reopens the new root, restores it, and checks the digest of **every payload** -against the source plus the intended replacement. `activation_burst_complete` -reports time until all roots are prepared separately from total time including -verification. A failed activation or verification is reported and returns a -nonzero exit; the round drains its in-flight tasks before returning. - -The probe uses Tokio blocking jobs for synchronous SQLite work and the LTX -Host's shared admission for internal I/O. It does not use runtime SQL-worker -sharding, Cell authority, `CellNode` or HTTP. Record actual cgroup limits and -memory/CPU counters when running under the one-vCPU/one-GiB profile; the default -Host is not itself an RSS limit. Two serial and two burst rounds are diagnostic, -not service latency percentiles or independent-host recovery qualification. - -To run with enforced container limits, follow the source-directory and RustFS -setup in the [Compose worker profile](../../crab-cell-runtime/qualification/worker-profile.md). -Build this example instead of the runtime test binary, then override the worker -entrypoint. The worker keeps its one-vCPU/one-GiB/no-swap limits and local scratch -volume. Use fresh container names for another size and retain the logs and -container inspection separately: - -```sh -worker_compose run --no-deps --name "$worker_project-build" build \ - cargo build --release --locked -p crab-ltx --features replica \ - --example rustfs_cell_replica_scale_load -worker_compose run --no-deps --name "$worker_project-burst" \ - --entrypoint /target/release/examples/rustfs_cell_replica_scale_load \ - -e CRAB_LTX_WORKLOAD_ROOT=/scratch -e CRAB_CELL_LTX_TARGET_BYTES=33554432 \ - worker --activation-cells 4 > "$CRAB_WORKER_STATE/evidence/burst.log" 2>&1 -docker --context "$CRAB_WORKER_CONTEXT" inspect "$worker_project-burst" \ - > "$CRAB_WORKER_STATE/evidence/burst-container.json" -``` - -## Public API exercised - -With `--features replica`, the canonical surface is: - -| Type | Main API | Purpose | -| --- | --- | --- | -| `CellReplica` | `new`, `prepare`, `prepare_bundle`, `prepare_compaction`, `open_root` | Prepare, reopen, and compact immutable Cell roots | -| `PreparedRoot` | `root`, `predecessor`, `verified` | Carry an exact proposal into Cell authority CAS | -| `CellPagedDatabase` | `read_page`, `read_run`, `prepare_writable` | Authenticate sparse reads and seed a writable activation | -| `CellWritableDatabase` | `open_writable` | Open a fresh sparse SQLite writer at one exact root | -| `Db` | `hydration`, `hydrate_step`, `take_io_error`, `prune_captured` | Drive bounded hydration and release exact acknowledged captures | -| `Hydration` | `resolved`, `total`, `faults`, `complete` | Report sparse activation progress | -| `bundle` | `Bundle::encode`, `decode`; `BundleEntry::for_cell` | Carry verified Cell-scoped LTX ranges into recovery overlays | - -Mutable owner/epoch/root publication remains a `crab-cell-runtime` authority -operation. `crab-ltx` prepares immutable bytes and never acknowledges an HTTP -request or changes mutable Cell control. - -## Dedicated-host power-cut probe - -`power_cut_probe` has a writer stage and a verifier stage for each local -durability contract. Run it on a dedicated fault host with the probe directory -on a disposable test filesystem. Keep the controller and its captured stdout -on a separate, unaffected device. Before starting each writer, create its fresh -test directory and sync that directory's parent; otherwise losing the test -directory's own unsynced name can masquerade as an LTX failure. The writer -emits `READY_CAPTURE` immediately -after a successful standalone `capture()` and parks with SQLite still open; -`READY_RESUME` follows a successful `persist_continuation()` and `close()`. -The controller must cut host power or inject the planned block-device fault -when it observes the relevant marker. A process signal alone is only a smoke -test, not power-loss evidence. For a block-device fault, verify only after a -fresh mount without the writer's warm page cache; otherwise cached bytes can -hide lost device writes. - -Build once on that host with its own mounted target directory: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-" \ - cargo build --release -p crab-ltx --features replica \ - --example power_cut_probe --locked -``` - -For the capture case, use a fresh directory on the disposable device. Record -the `SEGMENT`, `TXID`, `CHECKSUM`, and `DIGEST` lines off-device before cutting -at `READY_CAPTURE`. After reboot, use those exact values: - -```sh -power_cut_probe capture-write -power_cut_probe capture-verify \ - -``` - -The verifier checks the recorded LTX digest, constructs an exact verified -plan, restores it into a fresh file, and reads the expected SQL value. For -clean continuation, use a different fresh directory, record its `TXID`, -`CHECKSUM`, and `DIGEST`, and cut at `READY_RESUME`: - -```sh -power_cut_probe resume-write -power_cut_probe resume-verify \ - -``` - -The resume verifier first checks the complete database BLAKE3 digest, then -moves the continuation to a fresh path, verifies its exact recorded position -and SQL value, and captures the next transaction. Each verifier is one-shot: -start from a new directory for every cut. Save host/kernel, filesystem and -mount options, device cache mode, fault mechanism, cut marker and timing, -writer stdout, verifier stdout/stderr/exit status, and the expected and -observed endpoint with the off-device evidence. Neither this probe nor a -local process-kill smoke replaces the dedicated-host run in Plan 035. diff --git a/crates/crab-ltx/examples/local_roundtrip.rs b/crates/crab-ltx/examples/local_roundtrip.rs deleted file mode 100644 index c39e80619..000000000 --- a/crates/crab-ltx/examples/local_roundtrip.rs +++ /dev/null @@ -1,47 +0,0 @@ -use crab_ltx::{Db, Limits, LocalSegment, VerifiedPlan, restore_exact}; - -fn main() -> crab_ltx::Result<()> { - let source = tempfile::tempdir()?; - let replica = tempfile::tempdir()?; - let restored = tempfile::tempdir()?; - let limits = Limits::default(); - let mut db = Db::open(&source.path().join("repository.sqlite"), limits)?; - db.transaction(|tx| { - tx.execute( - "CREATE TABLE issues (number INTEGER PRIMARY KEY, title TEXT NOT NULL)", - [], - )?; - tx.execute( - "INSERT INTO issues VALUES (1, 'Recover this issue from LTX')", - [], - )?; - Ok(()) - })?; - let batch = db.capture()?; - let mut files = Vec::new(); - for segment in batch.segments { - let path = replica.path().join(format!( - "{}-{}.ltx", - segment.info().min_txid, - segment.info().max_txid - )); - // A local copy demonstrates the transport boundary, not cloud durability. - std::fs::copy(segment.path(), &path)?; - files.push(LocalSegment::new(path, segment.info().clone())); - } - db.close()?; - source.close()?; - let plan = VerifiedPlan::new(&files, batch.position, limits)?; - let path = restored.path().join("repository.sqlite"); - restore_exact(&plan, &path)?; - let conn = crab_ltx::rusqlite::Connection::open(path)?; - let title: String = conn.query_row("SELECT title FROM issues WHERE number = 1", [], |row| { - row.get(0) - })?; - println!("Restored issue #1: {title}"); - println!( - "Verified LTX position: {} / {:016x}", - batch.position.txid, batch.position.checksum - ); - Ok(()) -} diff --git a/crates/crab-ltx/examples/power_cut_probe.rs b/crates/crab-ltx/examples/power_cut_probe.rs deleted file mode 100644 index 4e80602cc..000000000 --- a/crates/crab-ltx/examples/power_cut_probe.rs +++ /dev/null @@ -1,191 +0,0 @@ -//! Standalone capture and clean-resume checkpoints for a dedicated fault host. -//! The controller records stdout outside the test device and cuts power at READY. - -use std::{io::Write, path::Path, sync::Arc}; - -use crab_ltx::{ - CellReplica, CellStorageLayout, CrabError, Db, Limits, LocalSegment, Position, SegmentInfo, - VerifiedPlan, internal::inspect_ltx, restore_exact, -}; -use crab_storage::Store; -use object_store::{memory::InMemory, path::Path as ObjectPath}; - -fn main() -> crab_ltx::Result<()> { - let mut args = std::env::args().skip(1); - let command = arg(&mut args)?; - let directory = arg(&mut args)?; - let directory = Path::new(&directory); - match command.as_str() { - "capture-write" => capture_write(directory), - "capture-verify" => { - let segment = arg(&mut args)?; - let txid = number(&arg(&mut args)?)?; - let checksum = number(&arg(&mut args)?)?; - let digest = digest(&arg(&mut args)?)?; - capture_verify( - directory, - Path::new(&segment), - Position { txid, checksum }, - digest, - ) - } - "resume-write" => resume_write(directory), - "resume-verify" => { - let txid = number(&arg(&mut args)?)?; - let checksum = number(&arg(&mut args)?)?; - let digest = digest(&arg(&mut args)?)?; - resume_verify(directory, Position { txid, checksum }, digest) - } - _ => Err(CrabError::InvalidState("unknown power-cut probe command")), - } -} - -fn arg(args: &mut impl Iterator) -> crab_ltx::Result { - args.next() - .ok_or(CrabError::InvalidState("missing power-cut probe argument")) -} - -fn number(value: &str) -> crab_ltx::Result { - value - .parse() - .map_err(|_| CrabError::InvalidState("invalid power-cut probe number")) -} - -fn digest(value: &str) -> crab_ltx::Result<[u8; 32]> { - if value.len() != 64 { - return Err(CrabError::InvalidState("invalid power-cut probe digest")); - } - let mut bytes = [0; 32]; - for (index, byte) in bytes.iter_mut().enumerate() { - let pair = value - .get(index * 2..index * 2 + 2) - .ok_or(CrabError::InvalidState("invalid power-cut probe digest"))?; - *byte = u8::from_str_radix(pair, 16) - .map_err(|_| CrabError::InvalidState("invalid power-cut probe digest"))?; - } - Ok(bytes) -} - -fn hex(bytes: &[u8; 32]) -> String { - bytes.iter().map(|byte| format!("{byte:02x}")).collect() -} - -fn capture_write(directory: &Path) -> crab_ltx::Result<()> { - let mut db = Db::open(&directory.join("capture.sqlite"), Limits::default())?; - db.transaction(|tx| { - tx.execute_batch("CREATE TABLE witness(v); INSERT INTO witness VALUES(7)") - })?; - let cut = db.capture()?; - let [segment] = cut.segments.as_slice() else { - return Err(CrabError::InvalidState("probe expected one LTX cut")); - }; - println!("SEGMENT={}", segment.path().display()); - println!("TXID={}", cut.position.txid); - println!("CHECKSUM={}", cut.position.checksum); - println!("DIGEST={}", hex(&segment.info().blake3)); - println!("READY_CAPTURE"); - std::io::stdout().flush()?; - park_forever() -} - -fn capture_verify( - directory: &Path, - segment: &Path, - position: Position, - digest: [u8; 32], -) -> crab_ltx::Result<()> { - let bytes = std::fs::read(segment)?; - let inspected = inspect_ltx(&bytes)?; - if inspected.blake3 != digest { - return Err(CrabError::ChecksumMismatch); - } - let info = SegmentInfo { - min_txid: inspected.min_txid, - max_txid: inspected.max_txid, - page_size: inspected.page_size, - database_pages: inspected.commit, - pre_checksum: inspected.pre_apply_checksum, - post_checksum: inspected.post_apply_checksum, - size_bytes: inspected.size_bytes, - blake3: inspected.blake3, - }; - let plan = VerifiedPlan::new( - &[LocalSegment::new(segment.to_owned(), info)], - position, - Limits::default(), - )?; - let restored = directory.join("capture-restored.sqlite"); - restore_exact(&plan, &restored)?; - let connection = crab_ltx::rusqlite::Connection::open(restored)?; - let value: i64 = connection.query_row("SELECT v FROM witness", [], |row| row.get(0))?; - if value != 7 { - return Err(CrabError::ChecksumMismatch); - } - println!("VERIFIED_CAPTURE {} {}", position.txid, position.checksum); - Ok(()) -} - -fn resume_write(directory: &Path) -> crab_ltx::Result<()> { - let mut db = Db::open(&directory.join("resume.sqlite"), Limits::default())?; - db.transaction(|tx| { - tx.execute_batch("CREATE TABLE witness(v); INSERT INTO witness VALUES(7)") - })?; - let cut = db.capture()?; - db.persist_continuation()?; - db.close()?; - let digest = blake3::hash(&std::fs::read(directory.join("resume.sqlite"))?); - println!("TXID={}", cut.position.txid); - println!("CHECKSUM={}", cut.position.checksum); - println!("DIGEST={}", hex(digest.as_bytes())); - println!("READY_RESUME"); - std::io::stdout().flush()?; - park_forever() -} - -fn resume_verify(directory: &Path, position: Position, digest: [u8; 32]) -> crab_ltx::Result<()> { - let source = directory.join("resume.sqlite"); - if blake3::hash(&std::fs::read(&source)?).as_bytes() != &digest { - return Err(CrabError::ChecksumMismatch); - } - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("power-cut-probe"), - [1; 16], - ), - [2; 32], - [3; 16], - Limits::default(), - )?; - let mut db = replica.open_resumed(&source, &directory.join("resume-verified.sqlite"))?; - if db.position() != position { - return Err(CrabError::ChecksumMismatch); - } - let value: i64 = db - .query_with(|connection| { - connection.query_row("SELECT v FROM witness", [], |row| row.get(0)) - }) - .map_err(|error| CrabError::Other(Box::new(error)))?; - if value != 7 { - return Err(CrabError::ChecksumMismatch); - } - db.transaction(|tx| tx.execute_batch("INSERT INTO witness VALUES(8)"))?; - let next = db.capture()?; - if Some(next.position.txid) != position.txid.checked_add(1) - || next.segments.first().map(|cut| cut.info().pre_checksum) != Some(position.checksum) - { - return Err(CrabError::ChecksumMismatch); - } - db.close()?; - println!( - "VERIFIED_RESUME {} {}", - next.position.txid, next.position.checksum - ); - Ok(()) -} - -fn park_forever() -> ! { - loop { - std::thread::park(); - } -} diff --git a/crates/crab-ltx/examples/rustfs_cell_replica_scale_load.rs b/crates/crab-ltx/examples/rustfs_cell_replica_scale_load.rs deleted file mode 100644 index e08ef9905..000000000 --- a/crates/crab-ltx/examples/rustfs_cell_replica_scale_load.rs +++ /dev/null @@ -1,669 +0,0 @@ -mod support; - -use crab_ltx::{CellReplica, CellStorageLayout, CrabError, Db, Host, Limits, RootRef}; -use crab_storage::Store; -use futures_util::StreamExt; -use object_store::path::Path as ObjectPath; -use std::{ - fs::{self, File}, - io::{self, Read as _}, - path::{Path, PathBuf}, - sync::{ - Arc, - atomic::{AtomicU64, Ordering}, - }, - time::Instant, -}; -use support::rustfs_target; - -const MIB: u64 = 1 << 20; -const GIB: u64 = 1 << 30; -const ROW_BYTES: i64 = 1 << 20; -const CUT_BYTES: u64 = 32 * MIB; -// Repeated roots must still produce distinct writes, otherwise later rounds -// can measure immutable-upload deduplication instead of first-write preparation. -static ACTIVATION_MUTATION: AtomicU64 = AtomicU64::new(1); - -#[tokio::main] -async fn main() -> crab_ltx::Result<()> { - let activation_cells = activation_cells()?; - let target_bytes = target_bytes()?; - if target_bytes < CUT_BYTES { - return Err(CrabError::InvalidState( - "CRAB_CELL_LTX_TARGET_BYTES must be at least 32 MiB", - )); - } - let workload_root = workload_root()?; - let source_directory = temporary_directory(&workload_root, "source")?; - let recovery_directory = temporary_directory(&workload_root, "recovery")?; - let scratch_directory = temporary_directory(&workload_root, "compaction")?; - let limits = limits_for(target_bytes); - let database = source_directory.path().join("cell.sqlite"); - let mut writer = Db::open(&database, limits)?; - writer.transaction(|transaction| { - transaction.execute_batch("CREATE TABLE payload(value BLOB NOT NULL)") - })?; - - let target = rustfs_target("cell-replica-native-scale")?; - let layout = CellStorageLayout::new( - target.store.clone(), - ObjectPath::from(target.repository_prefix.as_str()), - [7; 16], - ); - let replica = CellReplica::new(layout.clone(), [8; 32], [9; 16], limits)?; - let mut siblings = (1..activation_cells.unwrap_or(1)) - .map(|index| { - CellReplica::new(layout.clone(), [8 + index as u8; 32], [9; 16], limits) - .map(|replica| (replica, None)) - }) - .collect::>>()?; - let mut root: Option = None; - let mut prepared: Option = None; - let mut sequence = 1_u64; - let mut written = 0_u64; - let started = Instant::now(); - while written < target_bytes { - let batch_bytes = (target_bytes - written).min(CUT_BYTES); - let rows = i64::try_from(batch_bytes / ROW_BYTES as u64) - .map_err(|_| CrabError::InvalidState("Cell scale row count"))?; - let (next_writer, capture) = tokio::task::spawn_blocking(move || { - let mut writer = writer; - writer.transaction(|transaction| { - for _ in 0..rows { - transaction.execute( - "INSERT INTO payload(value) VALUES(randomblob(?1))", - [ROW_BYTES], - )?; - } - Ok(()) - })?; - let capture = writer.capture()?; - Ok::<_, crab_ltx::CrabError>((writer, capture)) - }) - .await - .map_err(|_| CrabError::InvalidState("Cell scale writer stopped"))??; - writer = next_writer; - - let next = match replica.prepare(root.as_ref(), &capture, sequence, 1).await { - Ok(next) => next, - Err(error) => { - eprintln!( - "CellReplica prepare failed after {written} bytes: {error:?}; capture segments={} position={:?}", - capture.segments.len(), - capture.position - ); - if let Some(previous) = &prepared { - eprintln!( - " predecessor root txid={} pages={} directory_height={} segments={}", - previous.root().position.txid, - previous.verified().database_pages(), - previous.verified().directory_height(), - previous.verified().segment_count() - ); - } - for segment in &capture.segments { - eprintln!( - " segment txid={}..{} bytes={} pages={} pre={} post={}", - segment.info().min_txid, - segment.info().max_txid, - segment.info().size_bytes, - segment.info().database_pages, - segment.info().pre_checksum, - segment.info().post_checksum - ); - if let Ok(bytes) = fs::read(segment.path()) - && bytes.len() >= 106 - { - let page = u32::from_be_bytes(bytes[100..104].try_into().unwrap_or([0; 4])); - let flags = - u16::from_be_bytes(bytes[104..106].try_into().unwrap_or([0; 2])); - eprintln!( - " first frame page={} flags={} length={}", - page, - flags, - bytes.len() - ); - } - } - return Err(error); - } - }; - root = Some(next.root()); - prepared = Some(next); - // Each probe Cell has its own authenticated graph. Reusing the same - // capture gives every Cell identical input bytes without sharing identity. - for (replica, previous) in &mut siblings { - *previous = Some( - replica - .prepare(previous.as_ref(), &capture, sequence, 1) - .await? - .root(), - ); - } - let (next_writer, _) = tokio::task::spawn_blocking(move || { - let mut writer = writer; - let removed = writer.prune_captured(&capture)?; - Ok::<_, crab_ltx::CrabError>((writer, removed)) - }) - .await - .map_err(|_| CrabError::InvalidState("Cell scale prune stopped"))??; - writer = next_writer; - written = written - .checked_add(batch_bytes) - .ok_or(CrabError::Limit(crab_ltx::LimitKind::CellScaleBytes))?; - sequence = sequence - .checked_add(1) - .ok_or(CrabError::Limit(crab_ltx::LimitKind::CellScaleSequence))?; - if written == target_bytes || (written / CUT_BYTES).is_multiple_of(8) { - println!("prepared {written} / {target_bytes} bytes"); - } - } - - tokio::task::spawn_blocking(move || writer.close()) - .await - .map_err(|_| CrabError::InvalidState("Cell scale writer stopped"))??; - let load_elapsed = started.elapsed(); - let source_digest = stream_digest(&database)?; - let payloads = if activation_cells.is_some() { - let path = database.clone(); - Some( - tokio::task::spawn_blocking(move || payload_digests(&path)) - .await - .map_err(|error| CrabError::Other(Box::new(error)))??, - ) - } else { - None - }; - remove_sqlite_artifacts(&database)?; - - let prepared = prepared.ok_or(CrabError::InvalidState("Cell scale produced no root"))?; - let root = root.ok_or(CrabError::InvalidState("Cell scale produced no root"))?; - measure_activation(&target, &root, limits, &workload_root).await?; - if let Some(payloads) = payloads { - let mut roots = vec![root]; - for (_, root) in siblings { - roots.push(root.ok_or(CrabError::InvalidState("activation Cell has no root"))?); - } - measure_activation_burst(&target, &roots, limits, &workload_root, Arc::new(payloads)) - .await?; - } - let restored = recovery_directory.path().join("restored.sqlite"); - let restore_started = Instant::now(); - prepared.verified().restore(&restored).await?; - let restore_elapsed = restore_started.elapsed(); - let restored_digest = stream_digest(&restored)?; - if restored_digest != source_digest { - return Err(CrabError::ChecksumMismatch); - } - - let compaction_started = Instant::now(); - let compacted = replica - .prepare_compaction( - &root, - 0..prepared.verified().segment_count(), - 9, - scratch_directory.path(), - ) - .await?; - let compaction_elapsed = compaction_started.elapsed(); - let compacted_path = recovery_directory.path().join("compacted.sqlite"); - let compacted_restore_started = Instant::now(); - compacted.verified().restore(&compacted_path).await?; - let compacted_restore_elapsed = compacted_restore_started.elapsed(); - if stream_digest(&compacted_path)? != source_digest { - return Err(CrabError::ChecksumMismatch); - } - - println!("\nRustFS CellReplica native scale load complete"); - println!("target bytes: {target_bytes}"); - println!( - "segments: {}", - prepared.verified().segment_count() - ); - println!("source checksum: {}", hex_digest(source_digest)); - println!( - "root digest: {}", - blake3::Hash::from_bytes(root.digest).to_hex() - ); - println!("load wall time: {load_elapsed:.3?}"); - println!("restore wall time: {restore_elapsed:.3?}"); - println!("compaction wall time: {compaction_elapsed:.3?}"); - println!("compacted restore time: {compacted_restore_elapsed:.3?}"); - println!("source deleted: true"); - println!("compaction checksum: exact"); - Ok(()) -} - -#[derive(Default)] -struct Reads { - requests: AtomicU64, - bytes: AtomicU64, -} - -impl Reads { - fn finish(&self, started: Instant) -> serde_json::Value { - serde_json::json!({ - "elapsed_us": started.elapsed().as_micros(), - "requests": self.requests.swap(0, Ordering::Relaxed), - "bytes": self.bytes.swap(0, Ordering::Relaxed), - }) - } -} - -async fn measure_activation( - target: &support::RustfsTarget, - root: &RootRef, - limits: Limits, - workload_root: &Path, -) -> crab_ltx::Result<()> { - for round in 0..3 { - let order = if round % 2 == 0 { [1, 4, 8] } else { [8, 4, 1] }; - for slots in order { - let reads = Arc::new(Reads::default()); - let requests = reads.clone(); - let bytes = reads.clone(); - // A new Store identity excludes the publication process's immutable - // caches. Reuse the provider connection pool to isolate metadata work. - let store = Store::new(target.store.inner().clone()) - .with_read_request_observer(Arc::new(move |_| { - requests.requests.fetch_add(1, Ordering::Relaxed); - })) - .with_read_byte_observer(Arc::new(move |count| { - bytes.bytes.fetch_add(count, Ordering::Relaxed); - })); - let layout = CellStorageLayout::new( - store, - ObjectPath::from(target.repository_prefix.as_str()), - [7; 16], - ); - let host = Host::default().with_io_slots(Arc::new(tokio::sync::Semaphore::new(slots))); - let replica = CellReplica::new(layout, [8; 32], [9; 16], limits)?.with_host(host); - for cache in ["cold_metadata", "reused_metadata"] { - let directory = temporary_directory(workload_root, "activation")?; - let destination = directory.path().join("active.sqlite"); - let started = Instant::now(); - let verified = replica.open_root(root).await?; - let root_open = reads.finish(started); - let pages = verified.database_pages(); - let started = Instant::now(); - let prepared = verified.paged().prepare_writable(&destination).await?; - let checksums = reads.finish(started); - let worker_reads = reads.clone(); - let started = Instant::now(); - let (open, query, hydration) = tokio::task::spawn_blocking(move || { - let mut database = prepared.open_writable(&destination)?; - let open = worker_reads.finish(started); - let started = Instant::now(); - let length: i64 = database - .query_with(|db| { - db.query_row( - "SELECT length(value) FROM payload WHERE rowid = 1", - [], - |row| row.get(0), - ) - }) - .map_err(|error| CrabError::Other(Box::new(error)))?; - let query = worker_reads.finish(started); - if length != ROW_BYTES { - return Err(CrabError::ChecksumMismatch); - } - let hydration = database.hydration()?.ok_or(CrabError::InvalidState( - "Cell scale activation is not sparse", - ))?; - database.close()?; - Ok::<_, CrabError>((open, query, hydration)) - }) - .await - .map_err(|_| CrabError::InvalidState("Cell scale activation worker stopped"))??; - // Exclude close from the next root-open sample. - let _ = reads.finish(Instant::now()); - println!( - "{}", - serde_json::json!({ - "measurement": "sparse_activation", - "round": round, - "io_slots": slots, - "cache": cache, - "database_pages": pages, - "sqlite_version": rusqlite::version(), - "root_open": root_open, - "checksums": checksums, - "writable_open": open, - "first_query": query, - "hydrated_pages": hydration.resolved, - "page_faults": hydration.faults, - }) - ); - } - } - } - Ok(()) -} - -async fn measure_activation_burst( - target: &support::RustfsTarget, - roots: &[RootRef], - limits: Limits, - workload_root: &Path, - payloads: Arc>, -) -> crab_ltx::Result<()> { - let host = Host::default(); - for (round, concurrency) in [1, roots.len(), roots.len(), 1].into_iter().enumerate() { - let started = Instant::now(); - let mut tasks = futures_util::stream::iter(roots) - .map(|root| { - activate_and_mutate( - target, - *root, - limits, - workload_root, - payloads.clone(), - host.clone(), - started, - ) - }) - .buffer_unordered(concurrency); - let mut failure = None; - let mut samples = Vec::new(); - while let Some(result) = tasks.next().await { - match result { - Ok(sample) => samples.push(sample), - Err(error) => { - eprintln!("activation burst round {round}: {error:?}"); - failure.get_or_insert(error); - } - } - } - let ready_us = started.elapsed().as_micros(); - // Full restores must not compete with another Cell's measured activation. - // Hold the prepared writers until the entire burst has reached its cut. - for sample in samples { - match verify_activation(sample).await { - Ok(mut report) => { - report["round"] = round.into(); - report["concurrency"] = concurrency.into(); - println!("{report}"); - } - Err(error) => { - eprintln!("activation burst verification round {round}: {error:?}"); - failure.get_or_insert(error); - } - } - } - println!( - "{}", - serde_json::json!({ - "measurement": "activation_burst_complete", "round": round, - "concurrency": concurrency, "cells": roots.len(), - "ready_us": ready_us, "including_verification_us": started.elapsed().as_micros(), - "verified": failure.is_none(), - "io_slots": host.io_capacity(), "blocking_slots": host.job_capacity(), - "dirty_slots": host.dirty_capacity(), "recovery_slots": host.recovery_capacity(), - }) - ); - if let Some(error) = failure { - return Err(error); - } - } - Ok(()) -} - -struct ActivationSample { - report: serde_json::Value, - database: Db, - capture: crab_ltx::CaptureBatch, - replica: CellReplica, - root: RootRef, - directory: tempfile::TempDir, - expected: Vec<[u8; 32]>, -} - -async fn activate_and_mutate( - target: &support::RustfsTarget, - root: RootRef, - limits: Limits, - workload_root: &Path, - payloads: Arc>, - host: Host, - burst_started: Instant, -) -> crab_ltx::Result { - let started = Instant::now(); - let dispatch_delay = started.duration_since(burst_started); - let reads = Arc::new(Reads::default()); - let requests = reads.clone(); - let bytes = reads.clone(); - // Per-Cell Store identities exclude metadata left by bootstrap and prior - // rounds. Host admission and the provider connection pool remain shared. - let store = Store::new(target.store.inner().clone()) - .with_read_request_observer(Arc::new(move |_| { - requests.requests.fetch_add(1, Ordering::Relaxed); - })) - .with_read_byte_observer(Arc::new(move |count| { - bytes.bytes.fetch_add(count, Ordering::Relaxed); - })); - let layout = CellStorageLayout::new( - store, - ObjectPath::from(target.repository_prefix.as_str()), - [7; 16], - ); - let replica = CellReplica::new(layout, root.cell, root.incarnation, limits)?.with_host(host); - let directory = temporary_directory(workload_root, "activation-burst")?; - let destination = directory.path().join("active.sqlite"); - let phase = Instant::now(); - let verified = replica.open_root(&root).await?; - let root_open = reads.finish(phase); - let phase = Instant::now(); - let prepared = verified.paged().prepare_writable(&destination).await?; - let checksums = reads.finish(phase); - let worker_reads = reads.clone(); - let expected = payloads.clone(); - let active_path = destination.clone(); - let mutation_id = ACTIVATION_MUTATION.fetch_add(1, Ordering::Relaxed); - let phase = Instant::now(); - let (database, capture, open, mutation, capture_phase, first_mutation_us, expected) = - tokio::task::spawn_blocking(move || { - let mut database = prepared.open_writable(&active_path)?; - let open = worker_reads.finish(phase); - let mut replacement = vec![root.cell[0]; ROW_BYTES as usize]; - replacement[..8].copy_from_slice(&mutation_id.to_le_bytes()); - let mut expected = expected.as_ref().clone(); - *expected - .first_mut() - .ok_or(CrabError::InvalidState("activation source is empty"))? = - *blake3::hash(&replacement).as_bytes(); - let phase = Instant::now(); - // Make the first application SQL a write so its page faults are - // measured here. Full restored-payload verification follows the - // measured burst instead of warming this row with a prior query. - database.transaction(|transaction| { - transaction.execute( - "UPDATE payload SET value = ?1 WHERE rowid = 1", - [&replacement], - )?; - Ok(()) - })?; - let mutation = worker_reads.finish(phase); - let first_mutation_us = started.elapsed().as_micros(); - let phase = Instant::now(); - let capture = database.capture_deferred()?; - let capture_phase = worker_reads.finish(phase); - Ok::<_, CrabError>(( - database, - capture, - open, - mutation, - capture_phase, - first_mutation_us, - expected, - )) - }) - .await - .map_err(|error| CrabError::Other(Box::new(error)))??; - let phase = Instant::now(); - let sequence = root - .commit_sequence - .checked_add(1) - .ok_or(CrabError::Limit(crab_ltx::LimitKind::CellScaleSequence))?; - let next = replica.prepare(Some(&root), &capture, sequence, 1).await?; - let root_prepare = reads.finish(phase); - let publication = replica.take_publication_cost(); - let prepared_us = started.elapsed().as_micros(); - let next_root = next.root(); - Ok(ActivationSample { - report: serde_json::json!({ - "measurement": "activation_burst", "cell": root.cell, "root": root.digest, - "root_txid": root.position.txid, "prepared_root": next_root.digest, - "mutation_id": mutation_id, "access_order": "write_first", - "dispatch_delay_us": dispatch_delay.as_micros(), - "first_mutation_us": first_mutation_us, - "prepared_us": prepared_us, "root_open": root_open, "checksums": checksums, - "writable_open": open, "first_mutation": mutation, - "capture": capture_phase, "root_prepare": root_prepare, - "prepared_objects": publication.objects, "prepared_bytes": publication.bytes, - }), - database, - capture, - replica, - root: next_root, - directory, - expected, - }) -} - -async fn verify_activation(sample: ActivationSample) -> crab_ltx::Result { - let ActivationSample { - mut report, - mut database, - capture, - replica, - root, - directory, - expected, - } = sample; - let destination = database.path().to_owned(); - // Remove local write/capture state before independently reopening the root. - // Verification reads every payload, so a successful first-row check cannot - // conceal damage to untouched rows during capture or recovery. - tokio::task::spawn_blocking(move || { - database.prune_captured(&capture)?; - database.close()?; - remove_sqlite_artifacts(&destination) - }) - .await - .map_err(|error| CrabError::Other(Box::new(error)))??; - let restored = directory.path().join("restored.sqlite"); - replica.open_root(&root).await?.restore(&restored).await?; - let recovered = tokio::task::spawn_blocking(move || payload_digests(&restored)) - .await - .map_err(|error| CrabError::Other(Box::new(error)))??; - if recovered != expected { - return Err(CrabError::ChecksumMismatch); - } - report["verified_rows"] = recovered.len().into(); - report["source_deleted"] = true.into(); - Ok(report) -} - -fn payload_digests(path: &Path) -> crab_ltx::Result> { - let database = - rusqlite::Connection::open_with_flags(path, rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY)?; - let mut statement = database.prepare("SELECT value FROM payload ORDER BY rowid")?; - let rows = statement.query_map([], |row| row.get::<_, Vec>(0))?; - rows.map(|row| { - row.map(|value| *blake3::hash(&value).as_bytes()) - .map_err(Into::into) - }) - .collect() -} - -fn activation_cells() -> crab_ltx::Result> { - let mut args = std::env::args().skip(1); - let Some(flag) = args.next() else { - return Ok(None); - }; - let count = args.next().and_then(|value| value.parse::().ok()); - if flag != "--activation-cells" - || args.next().is_some() - || !count.is_some_and(|count| (1..=16).contains(&count)) - { - return Err(CrabError::InvalidState( - "usage: rustfs_cell_replica_scale_load [--activation-cells 1..16]", - )); - } - Ok(count) -} - -fn target_bytes() -> crab_ltx::Result { - std::env::var("CRAB_CELL_LTX_TARGET_BYTES") - .map(|value| { - value - .parse() - .map_err(|_| CrabError::InvalidState("invalid CRAB_CELL_LTX_TARGET_BYTES")) - }) - .unwrap_or(Ok(5 * GIB)) -} - -fn workload_root() -> crab_ltx::Result { - let root = std::env::var_os("CRAB_LTX_WORKLOAD_ROOT").ok_or(CrabError::InvalidState( - "CRAB_LTX_WORKLOAD_ROOT is required", - ))?; - let root = PathBuf::from(root); - fs::create_dir_all(&root)?; - Ok(root) -} - -fn temporary_directory(root: &Path, label: &str) -> crab_ltx::Result { - tempfile::Builder::new() - .prefix(&format!("crab-cell-ltx-{label}-")) - .tempdir_in(root) - .map_err(Into::into) -} - -fn limits_for(target_bytes: u64) -> Limits { - Limits { - max_database_bytes: target_bytes.saturating_add(512 * MIB), - max_capture_bytes: 64 * MIB, - max_file_bytes: target_bytes.saturating_add(512 * MIB), - max_plan_bytes: target_bytes.saturating_mul(2).saturating_add(512 * MIB), - max_segments: 1024, - } -} - -fn stream_digest(path: &Path) -> crab_ltx::Result<(u64, [u8; 32])> { - let mut file = File::open(path)?; - let mut hasher = blake3::Hasher::new(); - let mut buffer = [0_u8; 1 << 20]; - let mut length = 0_u64; - loop { - let read = file.read(&mut buffer)?; - if read == 0 { - break; - } - hasher.update(&buffer[..read]); - length = length.checked_add(read as u64).ok_or(CrabError::Limit( - crab_ltx::LimitKind::CellScaleChecksumLength, - ))?; - } - Ok((length, *hasher.finalize().as_bytes())) -} - -fn remove_sqlite_artifacts(database: &Path) -> crab_ltx::Result<()> { - for suffix in ["", "-wal", "-shm"] { - let mut path = database.as_os_str().to_owned(); - path.push(suffix); - match fs::remove_file(PathBuf::from(path)) { - Ok(()) => {} - Err(error) if error.kind() == io::ErrorKind::NotFound => {} - Err(error) => return Err(error.into()), - } - } - Ok(()) -} - -fn hex_digest(digest: (u64, [u8; 32])) -> String { - format!( - "{}:{}", - digest.0, - blake3::Hash::from_bytes(digest.1).to_hex() - ) -} diff --git a/crates/crab-ltx/examples/support/mod.rs b/crates/crab-ltx/examples/support/mod.rs deleted file mode 100644 index e61005e99..000000000 --- a/crates/crab-ltx/examples/support/mod.rs +++ /dev/null @@ -1,47 +0,0 @@ -use crab_ltx::CrabError; -use crab_storage::{ObjectStoreCredentials, Store}; -use std::time::{SystemTime, UNIX_EPOCH}; - -pub struct RustfsTarget { - pub store: Store, - pub repository_prefix: String, -} - -pub fn rustfs_target(workload: &str) -> crab_ltx::Result { - let bucket = required_environment("CRAB_LTX_TEST_BUCKET")?; - let endpoint = required_environment("CRAB_LTX_TEST_ENDPOINT")?; - let access_key_id = required_environment("AWS_ACCESS_KEY_ID")?; - let secret_access_key = required_environment("AWS_SECRET_ACCESS_KEY")?; - let allow_http = match endpoint.split_once("://").map(|(scheme, _)| scheme) { - Some("http") => true, - Some("https") => false, - _ => { - return Err(CrabError::InvalidState( - "RustFS endpoint must use http or https", - )); - } - }; - let store = crab_storage::build_explicit_store( - &bucket, - ObjectStoreCredentials::Aws { - access_key_id, - secret_access_key, - session_token: None, - region: "us-east-1".into(), - }, - Some(&endpoint), - allow_http, - )?; - let run = SystemTime::now() - .duration_since(UNIX_EPOCH) - .map_err(|_| CrabError::InvalidState("system clock is before the Unix epoch"))? - .as_millis(); - Ok(RustfsTarget { - store, - repository_prefix: format!("crab-ltx-examples/{workload}/{run}-{}", std::process::id()), - }) -} - -fn required_environment(name: &str) -> crab_ltx::Result { - std::env::var(name).map_err(|_| CrabError::InvalidState("missing RustFS test environment")) -} diff --git a/crates/crab-ltx/fuzz/Cargo.lock b/crates/crab-ltx/fuzz/Cargo.lock deleted file mode 100644 index 687a1b693..000000000 --- a/crates/crab-ltx/fuzz/Cargo.lock +++ /dev/null @@ -1,2409 +0,0 @@ -# This file is automatically @generated by Cargo. -# It is not intended for manual editing. -version = 4 - -[[package]] -name = "android_system_properties" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" -dependencies = [ - "libc", -] - -[[package]] -name = "arbitrary" -version = "1.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c3d036a3c4ab069c7b410a2ce876bd74808d2d0888a82667669f8e783a898bf1" - -[[package]] -name = "arrayvec" -version = "0.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" - -[[package]] -name = "async-trait" -version = "0.1.92" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "atomic-waker" -version = "1.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" - -[[package]] -name = "autocfg" -version = "1.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" - -[[package]] -name = "aws-lc-rs" -version = "1.18.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b281d307588d634de920874890732659e2e7672f72b5e10e81badc1a8a83621e" -dependencies = [ - "aws-lc-sys", - "zeroize", -] - -[[package]] -name = "aws-lc-sys" -version = "0.45.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9bff6c3b54fad79a2e60b8102caf565819711497c1f5f092f49508e2f5c31b27" -dependencies = [ - "cc", - "cmake", - "dunce", - "fs_extra", - "pkg-config", -] - -[[package]] -name = "base64" -version = "0.22.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" - -[[package]] -name = "base64" -version = "0.23.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" - -[[package]] -name = "bitflags" -version = "2.13.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" - -[[package]] -name = "blake3" -version = "1.8.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d9e454fc11f76977dc803893aff6304ed33d6a26efae8696573bea74baa27ae" -dependencies = [ - "arrayvec", - "cc", - "cfg-if", - "constant_time_eq", - "cpufeatures", -] - -[[package]] -name = "block-buffer" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2f6c7dbe95a6ed67ad9f18e57daf93a2f034c524b99fd2b76d18fdfeb6660aa" -dependencies = [ - "hybrid-array", -] - -[[package]] -name = "bumpalo" -version = "3.20.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" - -[[package]] -name = "bytes" -version = "1.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" - -[[package]] -name = "cc" -version = "1.4.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "54413ede23c2daf518f35156dfde027feb2374004d63bd497f983c8db9c0e313" -dependencies = [ - "find-msvc-tools", - "jobserver", - "libc", - "shlex", -] - -[[package]] -name = "cfg-if" -version = "1.0.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" - -[[package]] -name = "cfg_aliases" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" - -[[package]] -name = "chacha20" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "65c35e4b699c7e15ccbe7ee35c005e4fc0a278d22238a2857e6ce2dadeda1b06" -dependencies = [ - "cfg-if", - "cpufeatures", - "rand_core 0.10.1", -] - -[[package]] -name = "chrono" -version = "0.4.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" -dependencies = [ - "iana-time-zone", - "num-traits", - "serde", - "windows-link", -] - -[[package]] -name = "cmake" -version = "0.1.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c0f78a02292a74a88ac736019ab962ece0bc380e3f977bf72e376c5d78ff0678" -dependencies = [ - "cc", -] - -[[package]] -name = "combine" -version = "4.6.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cfc320937d09e6de266b31b9afb480f197d7a861be86be7cb2ea7e5d1bfffc5e" -dependencies = [ - "bytes", - "memchr", -] - -[[package]] -name = "constant_time_eq" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b" - -[[package]] -name = "core-foundation" -version = "0.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b2a6cd9ae233e7f62ba4e9353e81a88df7fc8a5987b8d445b4d90c879bd156f6" -dependencies = [ - "core-foundation-sys", - "libc", -] - -[[package]] -name = "core-foundation-sys" -version = "0.8.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" - -[[package]] -name = "cpufeatures" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ca28b0ae3115b884660db4118d803791fd6756b6e88f39c0f3f7859060d7566" -dependencies = [ - "libc", -] - -[[package]] -name = "crab-ltx" -version = "0.1.0" -dependencies = [ - "async-trait", - "blake3", - "bytes", - "crab-storage", - "crc-fast", - "futures-util", - "lz4_flex", - "object_store", - "rusqlite", - "serde", - "serde_json", - "tempfile", - "thiserror", - "tokio", - "tokio-util", -] - -[[package]] -name = "crab-ltx-fuzz" -version = "0.0.0" -dependencies = [ - "crab-ltx", - "libfuzzer-sys", -] - -[[package]] -name = "crab-storage" -version = "0.1.0" -dependencies = [ - "async-trait", - "blake3", - "bytes", - "crab-types", - "futures-util", - "hyper", - "object_store", - "rand 0.9.5", - "reqwest 0.12.28", - "serde", - "serde_json", - "thiserror", - "tokio", - "tokio-util", - "tracing", - "url", -] - -[[package]] -name = "crab-types" -version = "0.1.0" -dependencies = [ - "schemars", - "serde", -] - -[[package]] -name = "crc-fast" -version = "1.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5" -dependencies = [ - "digest 0.10.7", - "spin", -] - -[[package]] -name = "crypto-common" -version = "0.1.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" -dependencies = [ - "generic-array", - "typenum", -] - -[[package]] -name = "crypto-common" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce6e4c961d6cd6c9a86db418387425e8bdeaf05b3c8bc1411e6dca4c252f1453" -dependencies = [ - "hybrid-array", -] - -[[package]] -name = "digest" -version = "0.10.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" -dependencies = [ - "crypto-common 0.1.7", -] - -[[package]] -name = "digest" -version = "0.11.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2" -dependencies = [ - "block-buffer", - "crypto-common 0.2.2", -] - -[[package]] -name = "displaydoc" -version = "0.2.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "dunce" -version = "1.0.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813" - -[[package]] -name = "dyn-clone" -version = "1.0.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555" - -[[package]] -name = "either" -version = "1.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" - -[[package]] -name = "equivalent" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" - -[[package]] -name = "errno" -version = "0.3.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" -dependencies = [ - "libc", - "windows-sys 0.61.2", -] - -[[package]] -name = "fallible-iterator" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" - -[[package]] -name = "fallible-streaming-iterator" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" - -[[package]] -name = "fastrand" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" - -[[package]] -name = "find-msvc-tools" -version = "0.1.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef25905e51abafe4dcea6c15fec58c57b601cdbd0ee53d22ea1d3016c587d39b" - -[[package]] -name = "fnv" -version = "1.0.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" - -[[package]] -name = "foldhash" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" - -[[package]] -name = "form_urlencoded" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" -dependencies = [ - "percent-encoding", -] - -[[package]] -name = "fs_extra" -version = "1.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" - -[[package]] -name = "futures-channel" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" -dependencies = [ - "futures-core", - "futures-sink", -] - -[[package]] -name = "futures-core" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" - -[[package]] -name = "futures-io" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" - -[[package]] -name = "futures-macro" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "futures-sink" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" - -[[package]] -name = "futures-task" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" - -[[package]] -name = "futures-util" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" -dependencies = [ - "futures-core", - "futures-io", - "futures-macro", - "futures-sink", - "futures-task", - "memchr", - "pin-project-lite", - "slab", -] - -[[package]] -name = "generic-array" -version = "0.14.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" -dependencies = [ - "typenum", - "version_check", -] - -[[package]] -name = "getrandom" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" -dependencies = [ - "cfg-if", - "js-sys", - "libc", - "wasi", - "wasm-bindgen", -] - -[[package]] -name = "getrandom" -version = "0.3.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" -dependencies = [ - "cfg-if", - "libc", - "r-efi 5.3.0", - "wasip2", -] - -[[package]] -name = "getrandom" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" -dependencies = [ - "cfg-if", - "js-sys", - "libc", - "r-efi 6.0.0", - "rand_core 0.10.1", - "wasm-bindgen", -] - -[[package]] -name = "h2" -version = "0.4.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef8e5e5a340588f4452631496976cf8636d4a7ecf600239fdc27615d2530bc16" -dependencies = [ - "atomic-waker", - "bytes", - "fnv", - "futures-core", - "futures-sink", - "http", - "indexmap", - "slab", - "tokio", - "tokio-util", - "tracing", -] - -[[package]] -name = "hashbrown" -version = "0.15.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" -dependencies = [ - "foldhash", -] - -[[package]] -name = "hashbrown" -version = "0.17.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" - -[[package]] -name = "hashlink" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7382cf6263419f2d8df38c55d7da83da5c18aef87fc7a7fc1fb1e344edfe14c1" -dependencies = [ - "hashbrown 0.15.5", -] - -[[package]] -name = "http" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" -dependencies = [ - "bytes", - "itoa", -] - -[[package]] -name = "http-body" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" -dependencies = [ - "bytes", - "http", -] - -[[package]] -name = "http-body-util" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" -dependencies = [ - "bytes", - "futures-core", - "http", - "http-body", - "pin-project-lite", -] - -[[package]] -name = "httparse" -version = "1.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" - -[[package]] -name = "humantime" -version = "2.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" - -[[package]] -name = "hybrid-array" -version = "0.4.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "27f864f10dfb56725ce5ce5472bc52252c8f93a4ab86327122cebf62c5f59a17" -dependencies = [ - "typenum", -] - -[[package]] -name = "hyper" -version = "1.11.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "27b501faa50e7a26c3d3560ca625132f4078a17771f4810baf70475ae48cbe43" -dependencies = [ - "atomic-waker", - "bytes", - "futures-channel", - "futures-core", - "h2", - "http", - "http-body", - "httparse", - "itoa", - "pin-project-lite", - "smallvec", - "tokio", - "want", -] - -[[package]] -name = "hyper-rustls" -version = "0.27.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dfa8e654703247911e29c23fbeaa261834bd9bb74efba2f9acddc37bfb127f53" -dependencies = [ - "http", - "hyper", - "hyper-util", - "rustls", - "tokio", - "tokio-rustls", - "tower-service", -] - -[[package]] -name = "hyper-util" -version = "0.1.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddc03d96684f9226b8a787cdb71488417b53ab5ea8fdb1dac946cb9431cc8bff" -dependencies = [ - "base64 0.23.1", - "bytes", - "futures-channel", - "futures-util", - "http", - "http-body", - "httparse", - "hyper", - "ipnet", - "libc", - "percent-encoding", - "pin-project-lite", - "socket2", - "tokio", - "tower-service", - "tracing", -] - -[[package]] -name = "iana-time-zone" -version = "0.1.65" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" -dependencies = [ - "android_system_properties", - "core-foundation-sys", - "iana-time-zone-haiku", - "js-sys", - "log", - "wasm-bindgen", - "windows-core", -] - -[[package]] -name = "iana-time-zone-haiku" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" -dependencies = [ - "cc", -] - -[[package]] -name = "icu_collections" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" -dependencies = [ - "displaydoc", - "potential_utf", - "utf8_iter", - "yoke", - "zerofrom", - "zerovec", -] - -[[package]] -name = "icu_locale_core" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" -dependencies = [ - "displaydoc", - "litemap", - "tinystr", - "writeable", - "zerovec", -] - -[[package]] -name = "icu_normalizer" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" -dependencies = [ - "icu_collections", - "icu_normalizer_data", - "icu_properties", - "icu_provider", - "smallvec", - "zerovec", -] - -[[package]] -name = "icu_normalizer_data" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" - -[[package]] -name = "icu_properties" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" -dependencies = [ - "displaydoc", - "icu_collections", - "icu_locale_core", - "icu_properties_data", - "icu_provider", - "zerotrie", - "zerovec", -] - -[[package]] -name = "icu_properties_data" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" - -[[package]] -name = "icu_provider" -version = "2.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" -dependencies = [ - "displaydoc", - "icu_locale_core", - "writeable", - "yoke", - "zerofrom", - "zerotrie", - "zerovec", -] - -[[package]] -name = "idna" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3b0875f23caa03898994f6ddc501886a45c7d3d62d04d2d90788d47be1b1e4de" -dependencies = [ - "idna_adapter", - "smallvec", - "utf8_iter", -] - -[[package]] -name = "idna_adapter" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb68373c0d6620ef8105e855e7745e18b0d00d3bdb07fb532e434244cdb9a714" -dependencies = [ - "icu_normalizer", - "icu_properties", -] - -[[package]] -name = "indexmap" -version = "2.14.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" -dependencies = [ - "equivalent", - "hashbrown 0.17.1", -] - -[[package]] -name = "ipnet" -version = "2.12.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "791930b43c0d5973160d90a8f3894509f2b273430f5c5c73b668636d0287c5c0" - -[[package]] -name = "itertools" -version = "0.15.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b4baf93f58d4425749ca49a51c50ebab072c5df6994d08fed93541c331481dc" -dependencies = [ - "either", -] - -[[package]] -name = "itoa" -version = "1.0.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" - -[[package]] -name = "jni" -version = "0.22.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5efd9a482cf3a427f00d6b35f14332adc7902ce91efb778580e180ff90fa3498" -dependencies = [ - "cfg-if", - "combine", - "jni-macros", - "jni-sys", - "log", - "simd_cesu8", - "thiserror", - "walkdir", - "windows-link", -] - -[[package]] -name = "jni-macros" -version = "0.22.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a00109accc170f0bdb141fed3e393c565b6f5e072365c3bd58f5b062591560a3" -dependencies = [ - "proc-macro2", - "quote", - "rustc_version", - "simd_cesu8", - "syn 2.0.119", -] - -[[package]] -name = "jni-sys" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c6377a88cb3910bee9b0fa88d4f42e1d2da8e79915598f65fb0c7ee14c878af2" -dependencies = [ - "jni-sys-macros", -] - -[[package]] -name = "jni-sys-macros" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "38c0b942f458fe50cdac086d2f946512305e5631e720728f2a61aabcd47a6264" -dependencies = [ - "quote", - "syn 2.0.119", -] - -[[package]] -name = "jobserver" -version = "0.1.35" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3" -dependencies = [ - "getrandom 0.4.3", - "libc", -] - -[[package]] -name = "js-sys" -version = "0.3.105" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce57d20d1ea864ce2ac172ab472d409214f4fd359f0b2a2775abdf522e2af99e" -dependencies = [ - "cfg-if", - "futures-util", - "wasm-bindgen", -] - -[[package]] -name = "libc" -version = "0.2.189" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" - -[[package]] -name = "libfuzzer-sys" -version = "0.4.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9fd2f41a1cba099f79a0b6b6c35656cf7c03351a7bae8ff0f28f25270f929d2" -dependencies = [ - "arbitrary", - "cc", -] - -[[package]] -name = "libsqlite3-sys" -version = "0.32.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fbb8270bb4060bd76c6e96f20c52d80620f1d82a3470885694e41e0f81ef6fe7" -dependencies = [ - "cc", - "pkg-config", - "vcpkg", -] - -[[package]] -name = "linux-raw-sys" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" - -[[package]] -name = "litemap" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" - -[[package]] -name = "lock_api" -version = "0.4.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "224399e74b87b5f3557511d98dff8b14089b3dadafcab6bb93eab67d3aace965" -dependencies = [ - "scopeguard", -] - -[[package]] -name = "log" -version = "0.4.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" - -[[package]] -name = "lru-slab" -version = "0.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4050469837a6ff301cd14c1f8f24f88549e6d548f24f64e2148eb0f72cebc51f" - -[[package]] -name = "lz4_flex" -version = "0.11.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a" -dependencies = [ - "twox-hash", -] - -[[package]] -name = "md-5" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "69b6441f590336821bb897fb28fc622898ccceb1d6cea3fde5ea86b090c4de98" -dependencies = [ - "cfg-if", - "digest 0.11.3", -] - -[[package]] -name = "memchr" -version = "2.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" - -[[package]] -name = "mio" -version = "1.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b18443e9c262bfe8fa82f51666e2642c53393f7e5c27b3e1aeab922cff5b9d8" -dependencies = [ - "libc", - "wasi", - "windows-sys 0.61.2", -] - -[[package]] -name = "nix" -version = "0.31.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d" -dependencies = [ - "bitflags", - "cfg-if", - "cfg_aliases", - "libc", -] - -[[package]] -name = "num-traits" -version = "0.2.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" -dependencies = [ - "autocfg", -] - -[[package]] -name = "object_store" -version = "0.14.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1796bc93603f78c5760a69f2d58badc9618d22adade0a95385bb2adbae4eb94" -dependencies = [ - "async-trait", - "aws-lc-rs", - "base64 0.23.1", - "bytes", - "chrono", - "crc-fast", - "form_urlencoded", - "futures-channel", - "futures-core", - "futures-util", - "http", - "http-body-util", - "httparse", - "humantime", - "hyper", - "itertools", - "md-5", - "nix", - "parking_lot", - "percent-encoding", - "quick-xml", - "rand 0.10.3", - "reqwest 0.13.5", - "rustls-pki-types", - "serde", - "serde_json", - "serde_urlencoded", - "thiserror", - "tokio", - "tracing", - "url", - "walkdir", - "wasm-bindgen-futures", - "web-time", - "windows-sys 0.61.2", -] - -[[package]] -name = "once_cell" -version = "1.21.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" - -[[package]] -name = "openssl-probe" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7c87def4c32ab89d880effc9e097653c8da5d6ef28e6b539d313baaacfbafcbe" - -[[package]] -name = "parking_lot" -version = "0.12.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "93857453250e3077bd71ff98b6a65ea6621a19bb0f559a85248955ac12c45a1a" -dependencies = [ - "lock_api", - "parking_lot_core", -] - -[[package]] -name = "parking_lot_core" -version = "0.9.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2621685985a2ebf1c516881c026032ac7deafcda1a2c9b7850dc81e3dfcb64c1" -dependencies = [ - "cfg-if", - "libc", - "redox_syscall", - "smallvec", - "windows-link", -] - -[[package]] -name = "percent-encoding" -version = "2.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" - -[[package]] -name = "pin-project-lite" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" - -[[package]] -name = "pkg-config" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" - -[[package]] -name = "potential_utf" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" -dependencies = [ - "zerovec", -] - -[[package]] -name = "ppv-lite86" -version = "0.2.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" -dependencies = [ - "zerocopy", -] - -[[package]] -name = "proc-macro2" -version = "1.0.107" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "quick-xml" -version = "0.41.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1" -dependencies = [ - "memchr", - "serde", -] - -[[package]] -name = "quinn" -version = "0.11.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4051e23e9185c255a7e33ef59cdbca87a22d359052eecd22fc6b901fb37d9d11" -dependencies = [ - "bytes", - "cfg_aliases", - "pin-project-lite", - "quinn-proto", - "quinn-udp", - "rustc-hash", - "rustls", - "socket2", - "thiserror", - "tokio", - "tracing", - "web-time", -] - -[[package]] -name = "quinn-proto" -version = "0.11.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9746dbde176634f4f2f1faf2404e30a31b2bc1e9cafb5329c95d8177a18c9fc" -dependencies = [ - "aws-lc-rs", - "bytes", - "getrandom 0.4.3", - "lru-slab", - "rand 0.10.3", - "rand_pcg", - "ring", - "rustc-hash", - "rustls", - "rustls-pki-types", - "slab", - "thiserror", - "tinyvec", - "tracing", - "web-time", -] - -[[package]] -name = "quinn-udp" -version = "0.5.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694" -dependencies = [ - "cfg_aliases", - "libc", - "once_cell", - "socket2", - "tracing", - "windows-sys 0.61.2", -] - -[[package]] -name = "quote" -version = "1.0.47" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" -dependencies = [ - "proc-macro2", -] - -[[package]] -name = "r-efi" -version = "5.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" - -[[package]] -name = "r-efi" -version = "6.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" - -[[package]] -name = "rand" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" -dependencies = [ - "rand_chacha", - "rand_core 0.9.5", -] - -[[package]] -name = "rand" -version = "0.10.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "65c9fb96cbc91e3478eaae79a69fcd3f1ae4ad052e471fe6732fff548984b4af" -dependencies = [ - "chacha20", - "getrandom 0.4.3", - "rand_core 0.10.1", -] - -[[package]] -name = "rand_chacha" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" -dependencies = [ - "ppv-lite86", - "rand_core 0.9.5", -] - -[[package]] -name = "rand_core" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" -dependencies = [ - "getrandom 0.3.4", -] - -[[package]] -name = "rand_core" -version = "0.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "63b8176103e19a2643978565ca18b50549f6101881c443590420e4dc998a3c69" - -[[package]] -name = "rand_pcg" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "caa0f4137e1c0a72f4c651489402276c8e8e1cf081f3b0ba156d2cbeef09e86a" -dependencies = [ - "rand_core 0.10.1", -] - -[[package]] -name = "redox_syscall" -version = "0.5.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" -dependencies = [ - "bitflags", -] - -[[package]] -name = "reqwest" -version = "0.12.28" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147" -dependencies = [ - "base64 0.22.1", - "bytes", - "futures-core", - "http", - "http-body", - "http-body-util", - "hyper", - "hyper-util", - "js-sys", - "log", - "percent-encoding", - "pin-project-lite", - "serde", - "serde_json", - "serde_urlencoded", - "sync_wrapper", - "tokio", - "tower", - "tower-http", - "tower-service", - "url", - "wasm-bindgen", - "wasm-bindgen-futures", - "web-sys", -] - -[[package]] -name = "reqwest" -version = "0.13.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "16a1cfa75cc186dd73d5818e510e042e40927bccc9c236b061cea97e1eb08029" -dependencies = [ - "base64 0.23.1", - "bytes", - "futures-core", - "futures-util", - "h2", - "http", - "http-body", - "http-body-util", - "hyper", - "hyper-rustls", - "hyper-util", - "js-sys", - "log", - "percent-encoding", - "pin-project-lite", - "quinn", - "rustls", - "rustls-pki-types", - "rustls-platform-verifier", - "sync_wrapper", - "tokio", - "tokio-rustls", - "tokio-util", - "tower", - "tower-http", - "tower-service", - "url", - "wasm-bindgen", - "wasm-bindgen-futures", - "wasm-streams", - "web-sys", -] - -[[package]] -name = "ring" -version = "0.17.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7" -dependencies = [ - "cc", - "cfg-if", - "getrandom 0.2.17", - "libc", - "untrusted", - "windows-sys 0.52.0", -] - -[[package]] -name = "rusqlite" -version = "0.34.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "37e34486da88d8e051c7c0e23c3f15fd806ea8546260aa2fec247e97242ec143" -dependencies = [ - "bitflags", - "fallible-iterator", - "fallible-streaming-iterator", - "hashlink", - "libsqlite3-sys", - "smallvec", -] - -[[package]] -name = "rustc-hash" -version = "2.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d" - -[[package]] -name = "rustc_version" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cfcb3a22ef46e85b45de6ee7e79d063319ebb6594faafcf1c225ea92ab6e9b92" -dependencies = [ - "semver", -] - -[[package]] -name = "rustix" -version = "1.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "891efababe418670775f199f0d233d84843c227a0949a883ce15b37c78d6629d" -dependencies = [ - "bitflags", - "errno", - "libc", - "linux-raw-sys", - "windows-sys 0.61.2", -] - -[[package]] -name = "rustls" -version = "0.23.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d41d731c7d2f962d1ccc364cec258de3c0e93b38c2fb3ba97ac74513048d634" -dependencies = [ - "aws-lc-rs", - "once_cell", - "rustls-pki-types", - "rustls-webpki", - "subtle", - "zeroize", -] - -[[package]] -name = "rustls-native-certs" -version = "0.8.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dab5152771c58876a2146916e53e35057e1a4dfa2b9df0f0305b07f611fdea4d" -dependencies = [ - "openssl-probe", - "rustls-pki-types", - "schannel", - "security-framework", -] - -[[package]] -name = "rustls-pki-types" -version = "1.15.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" -dependencies = [ - "web-time", - "zeroize", -] - -[[package]] -name = "rustls-platform-verifier" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1167586491e2b18b8bfbb293e8180ec17c201c4f076d7cb3070ca964e7598f98" -dependencies = [ - "core-foundation", - "core-foundation-sys", - "jni", - "log", - "once_cell", - "rustls", - "rustls-native-certs", - "rustls-platform-verifier-android", - "rustls-webpki", - "security-framework", - "security-framework-sys", - "webpki-root-certs", - "windows-sys 0.61.2", -] - -[[package]] -name = "rustls-platform-verifier-android" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eec689c0bc40ff2458a5977b6619cb718087084a18e02a131c599b62d05e1a5f" - -[[package]] -name = "rustls-webpki" -version = "0.103.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" -dependencies = [ - "aws-lc-rs", - "ring", - "rustls-pki-types", - "untrusted", -] - -[[package]] -name = "rustversion" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" - -[[package]] -name = "ryu" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" - -[[package]] -name = "same-file" -version = "1.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502" -dependencies = [ - "winapi-util", -] - -[[package]] -name = "schannel" -version = "0.1.29" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "91c1b7e4904c873ef0710c1f407dde2e6287de2bebc1bbbf7d430bb7cbffd939" -dependencies = [ - "windows-sys 0.61.2", -] - -[[package]] -name = "schemars" -version = "0.8.22" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3fbf2ae1b8bc8e02df939598064d22402220cd5bbcca1c76f7d6a310974d5615" -dependencies = [ - "dyn-clone", - "schemars_derive", - "serde", - "serde_json", -] - -[[package]] -name = "schemars_derive" -version = "0.8.22" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32e265784ad618884abaea0600a9adf15393368d840e0222d101a072f3f7534d" -dependencies = [ - "proc-macro2", - "quote", - "serde_derive_internals", - "syn 2.0.119", -] - -[[package]] -name = "scopeguard" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" - -[[package]] -name = "security-framework" -version = "3.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" -dependencies = [ - "bitflags", - "core-foundation", - "core-foundation-sys", - "libc", - "security-framework-sys", -] - -[[package]] -name = "security-framework-sys" -version = "2.17.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ce2691df843ecc5d231c0b14ece2acc3efb62c0a398c7e1d875f3983ce020e3" -dependencies = [ - "core-foundation-sys", - "libc", -] - -[[package]] -name = "semver" -version = "1.0.28" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" - -[[package]] -name = "serde" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" -dependencies = [ - "serde_core", - "serde_derive", -] - -[[package]] -name = "serde_core" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" -dependencies = [ - "serde_derive", -] - -[[package]] -name = "serde_derive" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "serde_derive_internals" -version = "0.29.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "18d26a20a969b9e3fdf2fc2d9f21eda6c40e2de84c9408bb5d3b05d499aae711" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "serde_json" -version = "1.0.151" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" -dependencies = [ - "itoa", - "memchr", - "serde", - "serde_core", - "zmij", -] - -[[package]] -name = "serde_urlencoded" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3491c14715ca2294c4d6a88f15e84739788c1d030eed8c110436aafdaa2f3fd" -dependencies = [ - "form_urlencoded", - "itoa", - "ryu", - "serde", -] - -[[package]] -name = "shlex" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" - -[[package]] -name = "simd_cesu8" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "11031e251abf8611c80f460e19dbdeb54a66db918e49c65a7065b46ac7aec520" -dependencies = [ - "rustc_version", - "simdutf8", -] - -[[package]] -name = "simdutf8" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" - -[[package]] -name = "slab" -version = "0.4.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" - -[[package]] -name = "smallvec" -version = "1.16.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba467056f1b547ed52077911161fc86985becbc60e8e1857c8a144dab0def891" - -[[package]] -name = "socket2" -version = "0.6.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" -dependencies = [ - "libc", - "windows-sys 0.61.2", -] - -[[package]] -name = "spin" -version = "0.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" - -[[package]] -name = "stable_deref_trait" -version = "1.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" - -[[package]] -name = "subtle" -version = "2.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" - -[[package]] -name = "syn" -version = "2.0.119" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "syn" -version = "3.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "sync_wrapper" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0bf256ce5efdfa370213c1dabab5935a12e49f2c58d15e9eac2870d3b4f27263" -dependencies = [ - "futures-core", -] - -[[package]] -name = "synstructure" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "901704edd0dfe137f1987838ee4f259e4e063c31371bdb423f7ae38ec6f77f02" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tempfile" -version = "3.27.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" -dependencies = [ - "fastrand", - "getrandom 0.4.3", - "once_cell", - "rustix", - "windows-sys 0.61.2", -] - -[[package]] -name = "thiserror" -version = "2.0.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09e52cb86a36cede5cb101bf8908837b3e4c6e5e59fe7fd85c23fb56200d189e" -dependencies = [ - "thiserror-impl", -] - -[[package]] -name = "thiserror-impl" -version = "2.0.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fe5197923287db20a58125f0bc85c062f7f2c892de97b18c356f9efb14b28524" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tinystr" -version = "0.8.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" -dependencies = [ - "displaydoc", - "zerovec", -] - -[[package]] -name = "tinyvec" -version = "1.13.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fd3ca314f692efd6c868f8408f53fe444634a845f96c028b97d35f6a1f79f0ee" - -[[package]] -name = "tokio" -version = "1.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" -dependencies = [ - "bytes", - "libc", - "mio", - "pin-project-lite", - "socket2", - "tokio-macros", - "windows-sys 0.61.2", -] - -[[package]] -name = "tokio-macros" -version = "2.7.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tokio-rustls" -version = "0.26.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b0c85f2c3ef0b1cd58b36682f4b17aaa995f0e5db534d85692b4903abce21f67" -dependencies = [ - "rustls", - "tokio", -] - -[[package]] -name = "tokio-util" -version = "0.7.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" -dependencies = [ - "bytes", - "futures-core", - "futures-sink", - "futures-util", - "libc", - "pin-project-lite", - "tokio", -] - -[[package]] -name = "tower" -version = "0.5.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4" -dependencies = [ - "futures-core", - "futures-util", - "pin-project-lite", - "sync_wrapper", - "tokio", - "tower-layer", - "tower-service", -] - -[[package]] -name = "tower-http" -version = "0.6.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840" -dependencies = [ - "bitflags", - "bytes", - "futures-util", - "http", - "http-body", - "pin-project-lite", - "tower", - "tower-layer", - "tower-service", - "url", -] - -[[package]] -name = "tower-layer" -version = "0.3.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "121c2a6cda46980bb0fcd1647ffaf6cd3fc79a013de288782836f6df9c48780e" - -[[package]] -name = "tower-service" -version = "0.3.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3" - -[[package]] -name = "tracing" -version = "0.1.44" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" -dependencies = [ - "pin-project-lite", - "tracing-attributes", - "tracing-core", -] - -[[package]] -name = "tracing-attributes" -version = "0.1.31" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "tracing-core" -version = "0.1.36" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" -dependencies = [ - "once_cell", -] - -[[package]] -name = "try-lock" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" - -[[package]] -name = "twox-hash" -version = "2.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" - -[[package]] -name = "typenum" -version = "1.20.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" - -[[package]] -name = "unicode-ident" -version = "1.0.26" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" - -[[package]] -name = "untrusted" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" - -[[package]] -name = "url" -version = "2.5.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff67a8a4397373c3ef660812acab3268222035010ab8680ec4215f38ba3d0eed" -dependencies = [ - "form_urlencoded", - "idna", - "percent-encoding", - "serde", -] - -[[package]] -name = "utf8_iter" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" - -[[package]] -name = "vcpkg" -version = "0.2.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" - -[[package]] -name = "version_check" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" - -[[package]] -name = "walkdir" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29790946404f91d9c5d06f9874efddea1dc06c5efe94541a7d6863108e3a5e4b" -dependencies = [ - "same-file", - "winapi-util", -] - -[[package]] -name = "want" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bfa7760aed19e106de2c7c0b581b509f2f25d3dacaf737cb82ac61bc6d760b0e" -dependencies = [ - "try-lock", -] - -[[package]] -name = "wasi" -version = "0.11.1+wasi-snapshot-preview1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" - -[[package]] -name = "wasip2" -version = "1.0.4+wasi-0.2.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" -dependencies = [ - "wit-bindgen", -] - -[[package]] -name = "wasm-bindgen" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "aecb87a33d3b0c5e3b7aa46336eaf486cffafbd281b195e4c8b80d50df2351bf" -dependencies = [ - "cfg-if", - "once_cell", - "rustversion", - "wasm-bindgen-macro", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-futures" -version = "0.4.78" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ef4c5d3d2cdf5c54f4231181768f5510842e350db025faf1f7163b1030ed928" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "wasm-bindgen-macro" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a690d511e3c1a8b3a55e33511e3c2c00c78415cd23650f32b808627f5696b9ed" -dependencies = [ - "quote", - "wasm-bindgen-macro-support", -] - -[[package]] -name = "wasm-bindgen-macro-support" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "411e4887f0071ef2d2164a9d5fdf2d20efbef78fccd3a78b0c10a1dc5295e48a" -dependencies = [ - "bumpalo", - "proc-macro2", - "quote", - "syn 3.0.6", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-shared" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "81941cd78d0c92026c33e5e01312845a4cb1e9af3407f9134b100dd03144103e" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "wasm-streams" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d1ec4f6517c9e11ae630e200b2b65d193279042e28edd4a2cda233e46670bbb" -dependencies = [ - "futures-util", - "js-sys", - "wasm-bindgen", - "wasm-bindgen-futures", - "web-sys", -] - -[[package]] -name = "web-sys" -version = "0.3.105" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9fbddc4a036f00ec4f18c83445bd3115cb306a91da554919a099d9222fe4a7f8" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "web-time" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a6580f308b1fad9207618087a65c04e7a10bc77e02c8e84e9b00dd4b12fa0bb" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "webpki-root-certs" -version = "1.0.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b96554aa2acc8ccdb7e1c9a58a7a68dd5d13bccc69cd124cb09406db612a1c9b" -dependencies = [ - "rustls-pki-types", -] - -[[package]] -name = "winapi-util" -version = "0.1.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22" -dependencies = [ - "windows-sys 0.61.2", -] - -[[package]] -name = "windows-core" -version = "0.62.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" -dependencies = [ - "windows-implement", - "windows-interface", - "windows-link", - "windows-result", - "windows-strings", -] - -[[package]] -name = "windows-implement" -version = "0.60.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "windows-interface" -version = "0.59.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "windows-link" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" - -[[package]] -name = "windows-result" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-strings" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-sys" -version = "0.52.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" -dependencies = [ - "windows-targets", -] - -[[package]] -name = "windows-sys" -version = "0.61.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-targets" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" -dependencies = [ - "windows_aarch64_gnullvm", - "windows_aarch64_msvc", - "windows_i686_gnu", - "windows_i686_gnullvm", - "windows_i686_msvc", - "windows_x86_64_gnu", - "windows_x86_64_gnullvm", - "windows_x86_64_msvc", -] - -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" - -[[package]] -name = "windows_aarch64_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" - -[[package]] -name = "windows_i686_gnu" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" - -[[package]] -name = "windows_i686_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" - -[[package]] -name = "windows_i686_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" - -[[package]] -name = "windows_x86_64_gnu" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" - -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" - -[[package]] -name = "windows_x86_64_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" - -[[package]] -name = "wit-bindgen" -version = "0.57.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" - -[[package]] -name = "writeable" -version = "0.6.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" - -[[package]] -name = "yoke" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" -dependencies = [ - "stable_deref_trait", - "yoke-derive", - "zerofrom", -] - -[[package]] -name = "yoke-derive" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "33811428bee40dbceb6d545e95754741d17a6aef9a4849f0fd62e2ba4f412a78" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", - "synstructure", -] - -[[package]] -name = "zerocopy" -version = "0.8.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c17e8fafad82b542ff3717217ecdc736231b59e387768c9630123b4ce4d2db44" -dependencies = [ - "zerocopy-derive", -] - -[[package]] -name = "zerocopy-derive" -version = "0.8.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "595f56e044df4f46a0c9a626f65c3d99eb8488f7e8a8baa12dd76326d9710bf2" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "zerofrom" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" -dependencies = [ - "zerofrom-derive", -] - -[[package]] -name = "zerofrom-derive" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f75b4683f6c7f45248d4d64056a24298c6281e0993356d7d1b4a1a962ef10d4a" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", - "synstructure", -] - -[[package]] -name = "zeroize" -version = "1.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" - -[[package]] -name = "zerotrie" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" -dependencies = [ - "displaydoc", - "yoke", - "zerofrom", -] - -[[package]] -name = "zerovec" -version = "0.11.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" -dependencies = [ - "yoke", - "zerofrom", - "zerovec-derive", -] - -[[package]] -name = "zerovec-derive" -version = "0.11.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "zmij" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/crates/crab-ltx/fuzz/Cargo.toml b/crates/crab-ltx/fuzz/Cargo.toml deleted file mode 100644 index 0c4c337b7..000000000 --- a/crates/crab-ltx/fuzz/Cargo.toml +++ /dev/null @@ -1,55 +0,0 @@ -# Decoder fuzzing for `crab-ltx`. -# -# Its own workspace on purpose: fuzzing needs a nightly toolchain and -# `cargo-fuzz`, which the normal `cargo test` path must not depend on. The -# stable-toolchain replay of the same entry points lives in -# `tests/ltx/vectors.rs`. -[package] -name = "crab-ltx-fuzz" -version = "0.0.0" -edition = "2021" -publish = false - -[package.metadata] -cargo-fuzz = true - -[workspace] - -[dependencies] -libfuzzer-sys = "0.4" -crab-ltx = { path = "..", features = ["replica"] } - -[[bin]] -name = "ltx" -path = "fuzz_targets/ltx.rs" -test = false -doc = false -bench = false - -[[bin]] -name = "root" -path = "fuzz_targets/root.rs" -test = false -doc = false -bench = false - -[[bin]] -name = "directory" -path = "fuzz_targets/directory.rs" -test = false -doc = false -bench = false - -[[bin]] -name = "bundle" -path = "fuzz_targets/bundle.rs" -test = false -doc = false -bench = false - -[[bin]] -name = "node_frame" -path = "fuzz_targets/node_frame.rs" -test = false -doc = false -bench = false diff --git a/crates/crab-ltx/fuzz/README.md b/crates/crab-ltx/fuzz/README.md deleted file mode 100644 index 9dbab803d..000000000 --- a/crates/crab-ltx/fuzz/README.md +++ /dev/null @@ -1,38 +0,0 @@ -# `crab-ltx` fuzz targets - -One target per untrusted decoder. Every target drives the stable inspection -surface in `crab_ltx::internal`, which is the same code production runs; a -panic, hang, or unbounded allocation is a finding. - -| Target | Surface | -| --- | --- | -| `ltx` | LTX header, page frames, index, trailer, and rolling checksums | -| `root` | Root document JSON and root descriptor pages | -| `directory` | Authenticated radix directory nodes | -| `bundle` | Bundle envelope, footer, and rows | -| `node_frame` | Authenticated node-log frame | - -## Run - -Fuzzing needs a nightly toolchain and `cargo-fuzz`: - -```sh -cargo install cargo-fuzz -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-ltx-fuzz" \ - cargo +nightly fuzz run ltx -- -max_total_time=600 -``` - -The external vectors in `../tests/vectors/` seed every corpus: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-ltx-fuzz" \ - cargo +nightly fuzz run ltx ../tests/vectors -``` - -## Stable-toolchain replay - -CI does not require nightly. `../tests/ltx/vectors.rs` runs the same entry points -over every truncation, deterministic bit flip, and length prefix of the external -vectors, so the decoders stay total on the stable toolchain and a panic fails -`cargo test -p crab-ltx --features replica`. Keep the replay and these targets in -step when adding a decoder. diff --git a/crates/crab-ltx/fuzz/fuzz_targets/bundle.rs b/crates/crab-ltx/fuzz/fuzz_targets/bundle.rs deleted file mode 100644 index cec82b6ea..000000000 --- a/crates/crab-ltx/fuzz/fuzz_targets/bundle.rs +++ /dev/null @@ -1,17 +0,0 @@ -#![no_main] - -//! Bundle envelopes must reject malformed footers and rows within their limits. - -use crab_ltx::Limits; -use libfuzzer_sys::fuzz_target; - -fuzz_target!(|data: &[u8]| { - // A small bound keeps the target focused on parsing, not on allocation. - let limits = Limits { - max_capture_bytes: 64 * 1024, - max_file_bytes: 256 * 1024, - max_plan_bytes: 1024 * 1024, - ..Limits::default() - }; - let _ = crab_ltx::internal::inspect_bundle(data, limits); -}); diff --git a/crates/crab-ltx/fuzz/fuzz_targets/directory.rs b/crates/crab-ltx/fuzz/fuzz_targets/directory.rs deleted file mode 100644 index b9ecd678e..000000000 --- a/crates/crab-ltx/fuzz/fuzz_targets/directory.rs +++ /dev/null @@ -1,9 +0,0 @@ -#![no_main] - -//! Directory nodes must reject malformed headers, records, and aggregates. - -use libfuzzer_sys::fuzz_target; - -fuzz_target!(|data: &[u8]| { - let _ = crab_ltx::internal::inspect_directory_node(data); -}); diff --git a/crates/crab-ltx/fuzz/fuzz_targets/ltx.rs b/crates/crab-ltx/fuzz/fuzz_targets/ltx.rs deleted file mode 100644 index c6ad542aa..000000000 --- a/crates/crab-ltx/fuzz/fuzz_targets/ltx.rs +++ /dev/null @@ -1,10 +0,0 @@ -#![no_main] - -//! Any byte string must decode or fail, never panic. - -use libfuzzer_sys::fuzz_target; - -fuzz_target!(|data: &[u8]| { - let _ = crab_ltx::internal::inspect_ltx(data); - let _ = crab_ltx::internal::cut_upper_bound(4096, data.len() as u64); -}); diff --git a/crates/crab-ltx/fuzz/fuzz_targets/node_frame.rs b/crates/crab-ltx/fuzz/fuzz_targets/node_frame.rs deleted file mode 100644 index a07011f5b..000000000 --- a/crates/crab-ltx/fuzz/fuzz_targets/node_frame.rs +++ /dev/null @@ -1,16 +0,0 @@ -#![no_main] - -//! Authenticated node frames must reject malformed scope and body fields. - -use crab_ltx::Limits; -use libfuzzer_sys::fuzz_target; - -fuzz_target!(|data: &[u8]| { - let limits = Limits { - max_capture_bytes: 64 * 1024, - max_file_bytes: 256 * 1024, - max_plan_bytes: 1024 * 1024, - ..Limits::default() - }; - let _ = crab_ltx::internal::inspect_node_frame(data, limits); -}); diff --git a/crates/crab-ltx/fuzz/fuzz_targets/root.rs b/crates/crab-ltx/fuzz/fuzz_targets/root.rs deleted file mode 100644 index 48ea81298..000000000 --- a/crates/crab-ltx/fuzz/fuzz_targets/root.rs +++ /dev/null @@ -1,10 +0,0 @@ -#![no_main] - -//! Root documents and descriptor pages must reject malformed JSON and extents. - -use libfuzzer_sys::fuzz_target; - -fuzz_target!(|data: &[u8]| { - let _ = crab_ltx::internal::inspect_root(data); - let _ = crab_ltx::internal::inspect_segment_page(data); -}); diff --git a/crates/crab-ltx/perf/README.md b/crates/crab-ltx/perf/README.md deleted file mode 100644 index f70046898..000000000 --- a/crates/crab-ltx/perf/README.md +++ /dev/null @@ -1,1029 +0,0 @@ -# `crab-ltx` versus `celld-ltx` - -This directory contains a small, reproducible local-filesystem comparison of -the two in-process LTX implementations. It is intentionally outside the Crab -Cargo workspace: `crab-ltx` uses `rusqlite` 0.34 while the pinned Celld source -uses `rusqlite` 0.31, and Cargo cannot link two `libsqlite3-sys` versions in -one process. Each runner is therefore its own package and emits JSON. - -The Celld runner is pinned to the revision documented in -[`UPSTREAM.md`](../UPSTREAM.md): - -`10cb1303dac710dcb3b557e318e08c855261f68b` - -## Current conclusion - -This harness does **not** establish that Crab is universally faster than -Celld. Crab's default capture pays a parent-directory durability barrier that -the pinned Celld capture does not, and is slower in the direct comparison. -Crab's opt-in grouped barrier improves total throughput in the retained local -workloads, but it measures batch completion rather than independently durable -per-transaction latency; repeated runs have not established a universal 1.5x -speedup. Recovery is also workload-dependent, with Celld still able to win the -small case. Treat the phase data and durability contract as part of every -performance claim. - -On 2026-09-21, a release diagnostic on the same macOS host ran 128 -transactions with 4 KiB payloads, one warmup and three measured rounds per -process. In three alternating Crab/Celld pairs, Crab default total time was -1.51x, 1.49x, and 1.52x the pinned Celld default total time (slower). With -the runner-only Celld `--sync-parent` option, the Crab/Celld ratios were -0.965, 1.008, and 0.999: near parity when both pay directory barriers. -Crab batch-8 completed a separate three-round diagnostic at a 0.424 s -median versus the pinned Celld default at 0.583 s, but batch completion is -not independently durable per-capture latency. These small local samples -confirm the durability-cost explanation; they are not production SLOs or -evidence of a universal Crab win. - -### Current PR comparison (2026-09-25) - -The current `crab-ltx` release runner and Celld pinned at -`10cb1303dac710dcb3b557e318e08c855261f68b` ran on the same macOS 25.5 -external APFS SSD. Each mode ran seven alternating independent processes; -each process warmed one 128-transaction round and measured three more. The -table shows the median of each process's median, followed by the nearest-rank -p95 across the seven process medians. Times are for the whole 128-transaction -round, in milliseconds. - -| Payload | Mode | Capture p50 | Recovery p50 | Full round p50 / p95 | -| --- | --- | ---: | ---: | ---: | -| 4 KiB | Crab immediate | 744 | 25 | 813 / 965 | -| 4 KiB | Celld default | 403 | 13 | 461 / 773 | -| 4 KiB | Celld with diagnostic directory syncs | 727 | 26 | 786 / 937 | -| 16 KiB | Crab immediate | 805 | 28 | 919 / 946 | -| 16 KiB | Celld default | 420 | 24 | 503 / 542 | -| 16 KiB | Celld with diagnostic directory syncs | 784 | 29 | 875 / 939 | - -Celld's default syncs completed LTX file bytes but does not sync each renamed -file's directory entry. The runner-only diagnostic syncs L0 after every cut, -the new L0 ancestor names after the first cut, the new L1 name after -compaction, and the restored file's parent. With that diagnostic, Crab's full -round median is 3% slower at 4 KiB and 5% slower at 16 KiB. Against Celld's -unchanged default it is 76% and 82% slower, respectively. The 4 KiB p95 has -substantial run-to-run variance. The full-round comparison also includes -different SQLite versions and different recovery verification work, so it is -not an isolated capture-algorithm comparison. - -Crab `--durability-batch 8` measured 325 / 337 ms p50 / p95 at 4 KiB and -309 / 328 ms at 16 KiB for the same round. Those medians are 29% and 39% -below Celld's default full-round medians, but each group of eight cuts waits -for one shared barrier before local durability can be acknowledged. They do -not represent independent per-transaction durable latency. Raw JSON for all -five modes is at -`$HOME/Workspace/crabbuild-target/crab-1bab/ltx-celld-current-20260925/`. -Reproduce each mode with `--transactions 128 --rounds 3 --warmup 1` and -`--payload-bytes 4096` or `16384`, repeating in seven alternating processes; -add `--sync-parent` for the Celld diagnostic or `--durability-batch 8` for -Crab's grouped mode. - -## Run it - -The script uses release builds, one warmup round, and five measured rounds by -default. It stores build output under the mounted workspace volume when -`CARGO_TARGET_DIR` is not supplied. - -```bash -crates/crab-ltx/perf/run.sh -``` - -Override the workload without editing the harness: - -```bash -LTX_TRANSACTIONS=512 \ -LTX_PAYLOAD_BYTES=16384 \ -LTX_ROUNDS=7 \ -LTX_WARMUP=2 \ -crates/crab-ltx/perf/run.sh -``` - -To measure the opt-in grouped durability path on the Crab runner, invoke it -directly with `--durability-batch N`. It completes and renames each LTX file, -then uses a bounded parallel file flush followed by one shared parent-directory -sync whenever `N` captures are ready. The final partial batch is also flushed: - -```bash -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-ltx-perf" \ - cargo run --release \ - --manifest-path crates/crab-ltx/perf/crab/Cargo.toml -- \ - --transactions 128 --payload-bytes 4096 --rounds 5 --warmup 1 \ - --durability-batch 8 -``` - -`--durability-batch 1` is the default synchronous path. Values greater than -one bound each durability group to at most `N` captures; they do not make an -individual capture durable before that group's barrier succeeds. - -The binaries also run directly when a single side is useful: - -```bash -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-ltx-perf" \ - cargo run --release \ - --manifest-path crates/crab-ltx/perf/crab/Cargo.toml -- \ - --transactions 128 --payload-bytes 4096 --rounds 5 --warmup 1 -``` - -## Workload and measurements - -Every measured round creates a fresh temporary SQLite database and performs the -same sequence: - -1. Create `payloads(id INTEGER PRIMARY KEY, value BLOB NOT NULL)`. -2. Capture/sync the schema transaction. -3. Commit `N` one-row transactions. Each row contains a deterministic BLOB of - `--payload-bytes` bytes, followed by one capture/sync call. -4. Compact all captured L0 files into one local output. -5. Restore the compacted state into a fresh SQLite file and run - `PRAGMA integrity_check` plus a row-count check. - -The JSON reports median wall-clock microseconds across measured rounds. The -important fields are: - -- `workload_write_us`: SQLite commit time for schema plus the `N` inserts. -- `capture_us`: local WAL-to-LTX capture time, including local file syncs. -- `capture_*_us`: Crab's capture phase ledger. `capture_fsync_us` is the LTX - file sync; `capture_parent_sync_us` includes the rename's parent sync and, - on the first cut, the new directory-chain syncs. The other fields split - position resolution, WAL reads, page collection, encoding, and local writes. -- `capture_barrier_us`: only populated for the Crab deferred mode; it includes - the grouped file flush and final parent-directory barrier and is included in - `capture_us`. -- `verify_us`: Crab's explicit owned-input plan verification. Celld reports - zero because its compactor does not expose an equivalent call. -- `compact_us`: local LTX compaction, including source listing, reads, merge, - and output fsync for Celld; Crab's `compact_exact` merge and output install. -- `compact_verify_us`: Crab's verification of the compacted output. Celld's - destination-level continuity check is included in `compact_us`. -- `restore_us`: end-to-end restore wall time. Celld additionally reports its - plan, download, and apply sub-timings. -- `recovery_us`: the recovery subtotal. Crab defines it as - `verify_us + compact_us + compact_verify_us + restore_us`. Celld uses the - same formula, with its explicit verification fields set to zero. -- `total_us`: the full local-round headline: - `workload_write_us + capture_us + recovery_us`. -- `input_ltx_bytes` and `compacted_ltx_bytes`: storage amplification evidence. - -The harness checks the restored row count and SQLite integrity in every round; -it does not include those checks in the reported restore timer. - -For a phase comparison, use `recovery_us` rather than comparing `compact_us` -alone. Crab intentionally verifies and owns every input before the merge; the -verified image is then encoded as a snapshot, synced, and read back to match its -exact length and BLAKE3 digest before installation. `compact_verify_us` builds -the explicit plan required by Crab's restore API and independently decodes that -snapshot. Celld's pinned `ReplicaCompactor` validates the range shape and -destination continuity but does not expose equivalent input-plan or compacted- -plan verification phases. Use `total_us` as the only full local-round headline. - -The implementations do not have identical durability costs. Crab fsyncs the -LTX file and its parent directory before returning a capture batch. The pinned -Celld path fsyncs the file but uses a plain rename without a parent-directory -sync. Do not treat the capture-only gap as a portable performance win without -making that durability choice explicit. - -The September 21 comparison and the sidecar before/after matrix below predate -the fix that syncs the newly created `ltx/0`, `ltx`, and session-directory names -on the first locally durable cut. -Its numbers are historical rather than a current-build timing claim. Later -cuts in the same session reuse that directory-chain proof. - -The `replica-cost` JSON now separates the schema bootstrap's first immediate -capture (`bootstrap_capture_us`) and its complete parent-sync phase -(`bootstrap_parent_sync_us`) from measured commands. On the current build, -seven independent release processes with 4 KiB commands measured first-cut -capture at 6,665 / 9,591 µs p50 / p95 and parent sync at 3,003 / 6,024 µs. -This was macOS 25.5 on the same external APFS SSD, Rust 1.97.0, bundled -SQLite 3.49.1, and the in-memory object store. The phase includes the final -LTX rename's direct-parent sync and the three one-time ancestor syncs; it -does not isolate those four calls individually. Raw per-process JSON is at -`$HOME/Workspace/crabbuild-target/crab-1bab/ltx-firstcut-20260925/`. -Reproduce each process with the release binary and -`--payload-bytes 4096 --commands 12 --warmup 5`; repeat seven times. - -The Celld runner accepts `--sync-parent` as a diagnostic contract-normalization -mode. After each upstream `Db::sync()`, it syncs Celld's L0 directory before -recording capture completion. On the first cut it also syncs the newly created -`0`, `ltx`, and session-directory names; after compaction it syncs the new L1 -directory name. It syncs the destination directory after restore installation. -This is runner-only behavior, not pinned Celld behavior. The September 21 -`--sync-parent` ratios above predate these additional directory-chain syncs. - -The grouped Crab mode measures batch completion: all captures remain -unacknowledged until the final file-and-directory barrier succeeds. It is a -throughput comparison, not a measurement of independently durable -per-transaction acknowledgement latency. Compare Crab's default `capture()` -path when every capture must cross its own durability boundary. - -## Scope - -This is a local mechanics benchmark, not a claim about the complete durability -protocol. It does not measure immutable-root preparation, object-store latency, -network retries, Cell authority/owner-head CAS, acknowledgement ordering, -retention, scheduled multi-level compaction, or either implementation's paged -VFS. Those paths have different contracts and need a second harness with the -same object-store and authority model before they can be compared fairly. - -## Cell publication cost per command - -`replica-cost/` measures the other half of that scope: what one command costs -the Cell object store when it publishes an immutable root. Each measured command -commits one SQLite transaction, captures one LTX cut, and prepares one successor -root through `CellReplica` over an in-memory `object_store`, which reports the -objects and bytes a provider would receive. Supply `--endpoint` and the bucket -arguments below to measure real RustFS requests. All activation modes -use the same measured host, deferred capture and exact-cut pruning. The runner -restores the final root from the provider into a fresh destination and verifies -every committed payload byte before emitting a report. - -```bash -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-ltx-replica-cost" \ - cargo run --release \ - --manifest-path crates/crab-ltx/perf/replica-cost/Cargo.toml -- \ - --payload-bytes 4096 --commands 32 --warmup 4 -``` - -### Activation histories and the first mutation - -`--activation fresh|sparse|hydrated|resumed` selects one writer history. All -histories start with the same bootstrap SQL, root, deferred capture policy and -mutation sequence. `--churn-rows` fixes the initial working set. Churn now -starts at the last row and walks backward; this keeps the first update outside -the header read-ahead window for sufficiently large databases. Historical -churn reports walked forward from row one. The old -`--sparse` switch is removed; unsupported options fail before creating data. - -| Mode | Setup before the first measured mutation | -| --- | --- | -| `fresh` (default) | Continue the writer that produced the bootstrap root | -| `sparse` | Close that writer and open a fresh writable VFS from the exact root | -| `hydrated` | Open the sparse VFS, then fetch and install all inherited pages in batches of at most 64 | -| `resumed` | Hydrate, persist the drained continuation, close, and reopen on a fresh path with local checksum verification | - -Every mode must retain the bootstrap TXID/checksum before it can run a command. -`activation` records the selected mode, initial database page count/size, -elapsed setup time and storage observations. Resumed setup includes hydration, -continuation persistence and reopen; it is not isolated reopen latency. -`activation.first_command` always retains command zero, even when `--warmup` -excludes it from the existing steady-state `samples` and percentile summaries. -Use `--warmup 0` to include every command in those summaries too. - -Each command separates `commit_io`, `capture_io`, `preparation_io` and -`prune_io`; activation and final restore are outside these phase counters. -These are observed storage operations, outcomes and bytes, not a count of -provider-internal retry attempts. The timing field `elapsed_us` still measures -root preparation only. Add commit, capture, preparation and prune durations -for their measured total; it excludes fixture bookkeeping and is not public -response latency. Resident mutations can have zero commit/capture origin reads -while root preparation still reads and writes the provider. - -Run each mode in a separate process against a private RustFS bucket, alternating -mode order between repeats. This example seeds about 32 MiB of payload, then -updates, deletes and reinserts one 4-KiB row per command: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ -TMPDIR="$HOME/Workspace/crabbuild-target/crab-8bc8/tmp" RUSTC_WRAPPER= \ - cargo run --release --locked \ - --manifest-path crates/crab-ltx/perf/replica-cost/Cargo.toml -- \ - --activation resumed --churn-rows 8192 --random-payload \ - --payload-bytes 4096 --commands 30 --warmup 3 \ - --endpoint http://127.0.0.1:44010 --bucket crab-cell-issue-fleet \ - --access-key crab --secret-key crab -``` - -Hold initial page count, payload, command sequence and changed pages constant -before comparing histories. Check each report's exact provider restore and -`restored_rows`. This library fixture does not measure actor queues, ownership -CAS, eviction policy, multi-Cell interference, power-loss durability, or a -one-vCPU/one-GiB node's capacity. - -#### RustFS GA verification (2026-09-27) - -Twelve release processes passed: three per history, using the command above -with alternating mode order. Every initial database had 9,222 pages of 4 KiB, -every first mutation updated row 8,192, and every final provider restore -verified all 8,192 expected rows byte for byte. Each run retained 27 commands -after three warmup commands, plus the first command separately. - -| History | First commit ms, min–max | First capture ms, min–max | First commit/capture origin bytes, each run | Later update commit median ms, min–max across runs | -| --- | ---: | ---: | ---: | ---: | -| Fresh | 0.457–18.668 | 0.307–2.002 | 0 | 0.526–9.893 | -| Sparse | 9.009–26.437 | 0.507–0.935 | 507,531 | 1.570–6.754 | -| Hydrated | 28.691–99.794 | 0.557–0.579 | 0 | 0.824–30.463 | -| Resumed | 1.260–59.712 | 0.529–0.660 | 0 | 0.614–7.006 | - -The deterministic result is that only sparse first mutations fetched inherited -pages from the provider. Hydrated and resumed writers still incurred local -SQLite work, and every history still performed provider work to prepare roots. -These noisy timing ranges do not establish a latency ranking or a performance -improvement. The benchmark exposed a clean-resume checksum mismatch: opening -a restored database wrote a synthetic sequence frame that SQLite checkpointed -on close. The capture-boundary fix and repeated no-write resume regression -are included with this runner. - -The native macOS ARM64 process used Rust 1.98.0 and SQLite 3.49.1. RustFS ran -in a private 4-vCPU/8-GiB Colima VM, pinned to -`ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. -Provider connections/caches persisted between processes, each process used a -fresh object prefix, and other host builds could compete for resources. There -was no separate provider CPU/memory cap. This is library correctness and I/O -attribution evidence; public-action and independent-host limits remain open. - -Retained evidence directory: `activation-histories-ga-20260927` under the -checkout's external Cargo target. It contains all twelve JSON reports and a -manifest with source hashes, report hashes, order and timing. Parent revision: -`58f370537887932dee3ada482b9077cdaccc4c9b`, plus this clean-resume and runner -change. All recorded source hashes matched the tested files at closeout. -Binary SHA-256: `e967ff4a8dd9c07597618f8d638bcbb9277b9c77ea2bce28d160ae18599485db`. -Manifest SHA-256: `536563484e6bf0e366958f4a38ae573623689139c3a83ded4694dd996ab55f23`. - -### Sparse activation baseline (2026-09-25) - -Pass `--activation sparse` to bootstrap a root, close the source writer, and activate a -real sparse writer through `open_root().paged().prepare_writable()` and -`open_writable()`. This mode uses deferred capture, prepares an immutable -successor, and prunes its local cut after preparation. It reports SQLite commit, -capture, checksum-sidecar sync count/time, root preparation, WAL bytes read, and -peak WAL-image allocation separately. The sidecar sync measurement wraps only -the local `FileSystem`; it does not include SQLite VFS syncs. In the original -baseline, fresh mode used immediate capture and retained cuts. That historical -comparison also changed the barrier and cleanup workload; the matched runner -described below removes those differences. Neither mode measures runtime -response proof latency or grants authority merely by preparing a root. - -The runner also emits per-command `capture_*_us` fields for all phases in -`CaptureTiming`. They distinguish WAL transfer, page collection, cut encoding, -checkpoint maintenance, and LTX reinspection during batch collection. -`prune_us` measures cleanup after root preparation in both modes; historical -fresh reports have null because they retained cuts. `captured_bytes` is the total LTX length in that -command's batch, including any checkpoint cut. Preparation timing excludes -cleanup, and neither timing includes the runtime authority CAS. - -### Matched capture and provider restore (2026-09-26) - -The follow-up to `3ced0777a6f` opens fresh databases through `CellReplica` so -both histories share the instrumented host. Both capture deferred cuts and -prune after preparing each successor, including the bootstrap. Final restore -and payload verification run outside the timed phases; `restored_rows` must -equal the configured command count. - -Three release processes per history used local RustFS, 32 transactions, four -warmups, and 4 KiB command-seeded random payloads. Run order alternated between -fresh and sparse. Every process restored all 32 rows byte-for-byte and prepared -five immutable objects per measured command. - -| Repeat | Fresh capture p50 / p95, µs | Sparse capture p50 / p95, µs | Fresh prepare p50 / p95, µs | Sparse prepare p50 / p95, µs | -| ---: | ---: | ---: | ---: | ---: | -| 1 | 342 / 463 | 723 / 1,812 | 3,648 / 4,666 | 4,377 / 6,391 | -| 2 | 322 / 426 | 616 / 803 | 3,761 / 6,009 | 3,603 / 4,815 | -| 3 | 336 / 471 | 618 / 911 | 3,774 / 5,392 | 3,597 / 5,210 | - -These small-database, shared-host diagnostics do not isolate checksum -representation from all sparse VFS work, measure activation latency, or -establish service capacity. No before/after performance gain is inferred. -Deferred rename time remains in `capture_parent_sync_us`; that field is a -phase duration, not proof that a directory sync occurred. SQLite's own syncs -are separate from the deferred LTX barrier. - -Retained evidence is under the external per-checkout target's -`matched-capture-20260926/`: six JSON reports, stderr, source diff, and source -and binary SHA-256 values. Reproduce each history with a fresh process; add -`--activation sparse` for the restored history: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ -TMPDIR="$HOME/Workspace/crabbuild-target/crab-8bc8/tmp" \ - cargo run --release --locked \ - --manifest-path crates/crab-ltx/perf/replica-cost/Cargo.toml -- \ - --random-payload --payload-bytes 4096 --commands 32 --warmup 4 \ - --endpoint http://127.0.0.1:19010 --bucket crab-cell-issue-fleet \ - --access-key crab --secret-key crab -``` - -### Fixed-working-set update/delete churn (2026-09-26) - -`replica-cost --churn-rows N` seeds N rows, then updates, deletes and reinserts -one row in three separate transactions before moving to the next row. The -working set cycles while command-seeded payloads change. Omitting the option -retains append-only inserts. Each `samples` entry includes `mutation`, `row`, -`live_rows` and checkpoint run/busy/frame/backfill counts, alongside the existing -capture, preparation and cleanup phases. Compare the mutation classes separately. -`object_prefix` identifies the retained immutable graph; `churn_rows` identifies -the initial working set. It does not cap runtime command-history retention. - -Every transaction must change exactly one row. After closing the writer and -pruning its captured cuts, the runner restores the selected root from the -provider into a fresh destination. An independent key-to-payload-seed model -rejects lost rows, stale payloads and resurrected deleted keys. A run ending -immediately after deletion must also restore that absence. These checks run -outside measured phases. In append mode, `restored_rows` still equals commands; -in churn mode, it equals the surviving working set. - -Reproduce on an existing isolated RustFS bucket; use `--activation sparse` for the restored -writer history and a target directory belonging to this checkout: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ -TMPDIR="$HOME/Workspace/crabbuild-target/crab-8bc8/tmp" RUSTC_WRAPPER= \ - cargo run --release --locked \ - --manifest-path crates/crab-ltx/perf/replica-cost/Cargo.toml -- \ - --churn-rows 32 --random-payload --payload-bytes 32768 \ - --commands 300 --warmup 12 \ - --endpoint http://127.0.0.1:19010 --bucket crab-cell-issue-fleet \ - --access-key crab --secret-key crab -``` - -Six alternating release processes against local RustFS -`1.0.0-beta.8-glibc` ran this workload. Each restored all 32 surviving rows, -with 96 measured updates, 96 deletes and 96 reinserts after warmup. Two -checkpoints per run backfilled their reported frames with no busy result. - -| Repeat | Fresh capture p50 / p95, µs | Sparse capture p50 / p95, µs | Fresh prepare p50 / p95, µs | Sparse prepare p50 / p95, µs | -| ---: | ---: | ---: | ---: | ---: | -| 1 | 468 / 974 | 565 / 947 | 13,669 / 19,304 | 13,820 / 20,893 | -| 2 | 465 / 566 | 550 / 690 | 13,591 / 19,600 | 13,338 / 19,071 | -| 3 | 473 / 619 | 546 / 682 | 13,590 / 20,934 | 12,825 / 19,385 | - -These are unconstrained macOS processes with Docker-hosted RustFS, one writer -and roughly 1 MiB of live payload. They do not measure public application -responses, scheduled compaction, pinned readers, concurrent recovery or the -one-vCPU/one-GiB node profile. Preparation excludes authority CAS. The fixture -runs for seconds, so crossing two checkpoints does not establish sustained -service capacity or a tail-latency SLO. The different payload size, initial -database and root-chain length prevent a causal comparison with the append -measurements above. - -Evidence is under the checkout's external target in `churn-20260926/`: -source patch, parent `6169c9c0270`, binary SHA256, six reports and their hashes, -and stderr. The separate `*-tail-*`, `*-append` and `*-full-image-delete` -reports exercise final update/delete/reinsert states, the original append -workload, and cuts exceeding an 8 KiB incremental bound in both histories. - -### Streaming published-cut cleanup (2026-09-26) - -Three release processes per workload and implementation used local RustFS, -command-seeded random payloads, sparse deferred capture, six commands, and one -warmup. The large workload used 4 MiB inserts and a 1 MiB incremental-capture -limit, exercising full-image cuts and checkpoint batches. Both versions used -the same macOS/APFS host and RustFS instance. Baseline production source was -`0c898097e95`; the candidate replaces cleanup's full-file buffer with a 64 KiB -buffered reader and verifies its digest during decoding. - -| Captured bytes in batch | Before cleanup, median ms | Streaming cleanup, median ms | -| ---: | ---: | ---: | -| 16,941,374 | 63.204 | 52.233 | -| 25,412,106 | 94.847 | 88.384 | -| 33,882,834 | 128.005 | 127.587 | -| 42,353,558 | 160.156 | 143.721 | -| 50,824,284 | 194.740 | 153.671 | - -Each row is the median of three matching command positions, not a tail -percentile. The small 4 KiB payload workload produced 5,162–7,155-byte batches; -its pooled cleanup median was 161 microseconds for both implementations over -15 measured commands each. Variation is visible in the raw samples. These -measurements suggest a large-cut benefit but do not establish an application -latency improvement, a 1 GiB RSS bound, or sustained throughput. - -Raw reports, binary SHA-256 values, host metadata, and the candidate source diff -are retained under -`$HOME/Workspace/crabbuild-target/crab-8bc8/prune-streaming-20260926/`. -`summary.json` retains every per-command comparison. Reproduce the large run -against an existing local RustFS bucket: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ -TMPDIR="$HOME/Workspace/crabbuild-target/crab-8bc8/tmp" \ - cargo run --release --locked \ - --manifest-path crates/crab-ltx/perf/replica-cost/Cargo.toml -- \ - --activation sparse --random-payload --payload-bytes 4194304 \ - --max-capture-bytes 1048576 --commands 6 --warmup 1 \ - --endpoint http://127.0.0.1:19010 --bucket crab-cell-issue-fleet \ - --access-key crab --secret-key crab -``` - -Cleanup still decodes every page and retains decoder indexes on the SQL worker. -The transfer-bound regression rejects a whole-file read for a large random cut; -the failure cases retain accounting and allow retry after repair. Decoder index -memory and same-worker response interference remain separate audit gates. - -### Streaming footer and optional replica index (2026-09-26) - -The next decoder change removes unused replica entries from ordinary -verification, compares footer entries as a stream, and moves requested replica -indexes to their caller instead of cloning them. Three release processes per -implementation repeated the preceding RustFS workload. Baseline LTX behavior -matches `5c2abfd915e` (unchanged by `a16c8efcc2b`); candidate production changes -are retained as a source diff beside the reports. - -| Captured bytes in batch | Previous decoder cleanup, median ms | Changed decoder cleanup, median ms | -| ---: | ---: | ---: | -| 16,941,374 | 46.035 | 25.091 | -| 25,412,106 | 70.635 | 37.511 | -| 33,882,834 | 92.304 | 50.118 | -| 42,353,558 | 131.654 | 63.285 | -| 50,824,284 | 140.464 | 101.878 | - -Rows compare three samples at the same command position. Small-cut pooled -medians were 202 and 139 microseconds across 15 commands per implementation. -These are exploratory observations: the shared developer host had unrelated -compiler/VM activity, and baseline samples overlapped focused compilation. -The runs were sequential, not randomized or isolated, so the differences -cannot establish a causal gain or public-action percentiles. - -Whole-process maximum RSS for the large workload ranged from 86.5–92.9 MB -before and 83.6–91.7 MB after; those ranges do not establish a memory reduction. -They include SQLite, capture, publication, and cleanup, not decoder allocations -alone. Concurrent verification under the 1 GiB node profile remains unmeasured. -Raw JSON, `/usr/bin/time -l` output, binary digests, source provenance/diff, and -per-command comparison are retained under -`$HOME/Workspace/crabbuild-target/crab-8bc8/decoder-streaming-20260926/`. -Use the preceding command to reproduce the workload. - -### Large sparse checkpoint capture (2026-09-25) - -With `--activation sparse --payload-bytes 4194304 --max-capture-bytes 1048576 ---commands 3 --warmup 1`, seven independent release processes measured two -commands each. Each row below is the p50 / nearest-rank p95 of the seven -per-process medians, in microseconds. The before and after binaries ran on the -same macOS/APFS host; run-to-run variation makes this local evidence rather -than a response-latency SLO. - -| Phase | Before | After retaining every sealed cut | -| --- | ---: | ---: | -| SQLite commit | 4,327 / 4,550 | 4,482 / 5,258 | -| LTX capture | 48,935 / 50,122 | 38,801 / 47,617 | -| LTX reinspection during collection | 10,321 / 10,588 | 0 / 0 | -| LTX encode | 21,376 / 21,984 | 21,632 / 22,089 | -| In-memory root preparation | 2,860 / 2,926 | 2,881 / 2,967 | - -Checkpointing can seal a second cut before the command receives its batch. -The prior implementation cached the newest cut's metadata and re-read the -earlier LTX file to obtain its size, digest, and checksums. The writer already -computed those values while sealing that same cut. The new cache holds each -sealed result until collection, removing that reinspection. `VerifiedPlan` -still reads and verifies every cut before exact restore. Raw per-process JSON -is under `$HOME/Workspace/crabbuild-target/crab-1bab/ltx-slice4-profile-20260925/` -(`sparse-*.json` and `metadata-cache-*.json`). The production response phase -profile and provider durability receipt remain open. - -For a small-cut regression check, 21 independent processes per mode used -`--payload-bytes 4096` or `16384`, `--commands 12 --warmup 5`. Sparse deferred -capture measured 280 / 329 µs at 4 KiB and 301 / 351 µs at 16 KiB (p50 / -p95), compared with the earlier seven-process 285 / 335 and 314 / 374 µs. -Fresh immediate capture measured 4,278 / 5,578 µs and 4,572 / 5,509 µs, -compared with 4,255 / 5,362 and 4,465 / 5,565 µs. These are different -process counts and non-interleaved host runs, so small differences are not -attributable to this change; none shows a greater than 5% p95 regression. - -Seven independent release-process rounds per row on macOS 25.5, APFS on a USB -SSD, Apple silicon, Rust 1.97.0, bundled SQLite 3.49.1, and `object_store` -0.14.2 in-memory. Each small-payload round measured seven commands after five -warmups; each 4 MiB round measured two after one warmup. Cells show p50 / p95 -across the seven per-round medians, in microseconds. Peak RSS is the median -per-process maximum from `/usr/bin/time -l`. The large case sets -`--max-capture-bytes 1048576` and recorded four complete WAL reads per round. - -| Workload | Payload | SQLite commit | LTX capture | Sidecar sync | Root prepare | Sidecar syncs / round | WAL read / round | Peak RSS | -| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | -| Fresh immediate | 4 KiB | 318 / 442 | 4,255 / 5,362 | 0 / 0 | 172 / 207 | 0 | 149 KiB | 11.8 MiB | -| Sparse deferred | 4 KiB | 332 / 346 | 2,465 / 3,203 | 2,188 / 2,741 | 176 / 256 | 7 | 177 KiB | 12.1 MiB | -| Fresh immediate | 16 KiB | 244 / 256 | 4,001 / 4,824 | 0 / 0 | 173 / 314 | 0 | 234 KiB | 12.0 MiB | -| Sparse deferred | 16 KiB | 279 / 351 | 3,154 / 3,347 | 2,790 / 3,006 | 214 / 420 | 7 | 262 KiB | 12.3 MiB | -| Fresh immediate | 4 MiB | 4,800 / 5,357 | 45,164 / 47,473 | 0 / 0 | 2,498 / 2,795 | 0 | 24.4 MiB | 37.1 MiB | -| Sparse deferred | 4 MiB | 4,485 / 4,812 | 48,017 / 48,920 | 5,055 / 6,165 | 2,580 / 2,620 | 4 | 24.3 MiB | 37.8 MiB | - -The raw per-round JSON is outside tracked source at -`$HOME/Workspace/crabbuild-target/crab-1bab/ltx-baseline-20260925/`. To -reproduce a round after a release build, run the corresponding command seven -times, retaining each JSON output and `/usr/bin/time -l` maximum resident set -size: - -```bash -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-" \ - cargo build --release --locked \ - --manifest-path crates/crab-ltx/perf/replica-cost/Cargo.toml -/usr/bin/time -l "$HOME/Workspace/crabbuild-target/crab-/release/crab-ltx-replica-cost" \ - --activation sparse --payload-bytes 4096 --commands 12 --warmup 5 -``` - -Replace the payload with `16384` for the middle row. For 4 MiB, use -`--payload-bytes 4194304 --max-capture-bytes 1048576 --commands 3 --warmup 1`. -Use `--activation fresh` (the default) for fresh rows. The published RustFS loopback measurements -below are provider preparation cost; runtime fleet or exact-root response -latency needs a separate qualification receipt. - -Production telemetry now includes -`crab_cell_ltx_phase_seconds{phase="root_preparation"}` for each normal -`CellReplica::prepare` attempt, alongside capture phases and -`crab_cell_durability_wait_seconds{source="fleet|object"}`. Preparation includes -admission, immutable uploads, and verification; it can overlap follower proof. -Use the protected response profile to decide which phase controls acknowledgement -latency before selecting another optimization. - -After removing the active sidecar's per-cut sync, a second seven-round matrix -with the same commands and host measured: - -| Workload | Payload | Capture p50 / p95 before → after | Sidecar syncs / round before → after | WAL read / round before → after | -| --- | ---: | ---: | ---: | ---: | -| Fresh immediate | 4 KiB | 4,255 / 5,362 → 3,912 / 4,502 | 0 → 0 | 149 → 149 KiB | -| Sparse deferred | 4 KiB | 2,465 / 3,203 → 285 / 335 | 7 → 0 | 177 → 177 KiB | -| Fresh immediate | 16 KiB | 4,001 / 4,824 → 4,465 / 5,565 | 0 → 0 | 234 → 234 KiB | -| Sparse deferred | 16 KiB | 3,154 / 3,347 → 314 / 374 | 7 → 0 | 262 → 262 KiB | -| Fresh immediate | 4 MiB | 45,164 / 47,473 → 43,154 / 45,192 | 0 → 0 | 24.4 → 24.4 MiB | -| Sparse deferred | 4 MiB | 48,017 / 48,920 → 41,348 / 43,153 | 4 → 0 | 24.3 → 24.3 MiB | - -The second raw matrix is at -`$HOME/Workspace/crabbuild-target/crab-1bab/ltx-after-sidecar-20260925/`. -The unchanged fresh 16 KiB path moved by more than the desired 5% tolerance, -so these two sequential matrices alone cannot establish a precise global -latency regression bound. The sidecar sync count and sparse capture reduction -are direct local evidence; a response-latency claim still requires the runtime -qualification environment. - -Pass `--endpoint http://host:port --bucket --access-key ---secret-key ` to run the identical workload against an S3-compatible -provider. The table below was measured on 2026-09-24 (Apple silicon, release -build, one bootstrap root then 28 measured commands) against the in-memory store -and against a local RustFS server: - -| payload | objects/command | objects p95 | bytes/command | bytes p95 | in-memory us p50 / p95 | RustFS us p50 / p95 / p99 | -| --- | --- | --- | --- | --- | --- | --- | -| 4 KiB | 5 | 5 | 13,050 | 20,009 | 189 / 289 | 99,872 / 139,124 / 152,133 | -| 16 KiB | 5 | 5 | 19,057 | 29,331 | 222 / 393 | 87,197 / 94,588 / 98,919 | -| 256 KiB | 7 | 11 | 81,735 | 105,934 | 408 / 556 | 85,729 / 126,159 / 155,510 | -| 1 MiB | 8 | 11 | 159,666 | 178,523 | 604 / 848 | 97,929 / 120,578 / 132,084 | -| 4 MiB | 13 | 14 | 526,784 | 547,043 | 1,102 / 1,583 | 126,178 / 145,947 / 149,441 | - -Every command pays a segment body, its index, the rewritten directory nodes, the -root document, and any segment page. Object and byte counts are byte-identical -across the two stores, so they are provider independent; only latency moves, and -the RustFS rows are loopback latency, not a cloud bucket. Counts grow with the -number of directory leaves a payload touches: 5 objects for a small write up to -13 at the 4 MiB maximum a built-in primitive may write in one command. Byte cost -is dominated by the rewritten leaves and root document, and the payload's -compressibility matters more than its size. - -A hot Cell offering 100 commands per second would demand roughly 500-1,300 -logical immutable-object uploads per second at these measured costs. These -are not HTTP request counts: at that revision native LTX bodies used multipart even when small, -and retries, metadata reads, and authority CAS add requests. The runtime admits -against one pending-publication byte high-water mark per Cell plus the -32-segment compaction debt that folds the -root graph. Publication stays one serialized root per command because the object -path is the long-term durability authority and the node-log fleet proof releases -the command earlier; on this loopback RustFS path one command costs roughly -0.09-0.15 seconds of provider work, so a Cell without a fleet proof is -provider-latency bound, not CPU bound. Coalescing several commands into one root -could save metadata objects while retaining each command's captured bodies. -Fleet proof can release a response before object publication; the response -winner and sustained publication drain rate still need measurement. Multi-Cell concurrency, -cloud-bucket p99, and retention cost remain unmeasured. - -Audit qualification limits: `replica-cost` generates a periodic payload that -repeats every 251 bytes; it does not represent incompressible data. Its direct -`CellReplica::prepare` loop excludes runtime authority CAS, the follower race, -and scheduled compaction. With 28 measured commands, nearest-rank p99 is the -maximum sample. Retain these rows as a reproducible historical workload; -qualify entropy, sustained publication debt, and application latency separately. -The [LTX performance audit](../../crab-cell-runtime/docs/ltx-performance-audit.md) -records source-backed optimization candidates and their acceptance gates. - -### Small-body transfer experiment - -On 2026-09-25, seven independent release processes before (`a7091fd7138`) -and after (`85a3bf684d3`) the bounded single-PUT change used the same loopback -RustFS container. Each process ran `--activation sparse --payload-bytes 4096 --commands 32 ---warmup 4`: 28 measured preparations of the historical periodic payload. -Values below are the median of the seven per-run statistics, in microseconds: - -| Measurement | Before | After | -| --- | ---: | ---: | -| Root preparation p50 | 6,642 | 6,163 | -| Root preparation p95 | 10,951 | 12,142 | -| SQLite commit p50 | 463 | 550 | -| Capture p50 | 542 | 653 | -| Logical objects / command | 5 | 5 | -| Mean bytes / command | 13,046 | 13,046 | - -Raw samples and source/binary SHA-256 metadata are retained outside the -checkout under `$HOME/.codex/cell-vfs-ltx-scale/ltx-upload-a7091fd/`, with -`before-0.json` through `before-6.json` and matching `after-*` files. -The store image is the pinned RustFS `1.0.0-beta.8-glibc` from the Compose -example; the harness uses `object_store` 0.14.2 and bundled SQLite 3.49.1. - -This is an inconclusive latency comparison: phases ran sequentially on one -shared host, p95 worsened, and the unchanged commit/capture paths also slowed. -The baseline itself differs substantially from the earlier RustFS table. -Do not attribute that historical difference to this patch or claim a p99, -fleet throughput, or latency SLO. This harness does not install the persistent -directory cache, so it cannot measure removal of cache-index writes. - -Focused transport tests prove the narrower change: native, compacted, and -bundled bodies up to 256 KiB use single PUT; larger bodies keep multipart; -lost responses reconcile; conflicting bytes are refused; restored databases -are identical. A real RustFS public HTTP/peer test also passed mutations and -restoration after takeover. Public-action p95/p99 and sustained publication -drain remain the performance acceptance gates. - -### Backend calls and payload entropy - -The runner now uses the existing `Store::with_storage_observer` boundary. -Each sample's `preparation_io` records backend operation/outcome, calls, -accumulated duration, bytes read, and bytes written for root preparation. -Bootstrap, activation, and commit/capture reads are excluded. Calls retried -by `Store` appear separately; retries inside the provider client do not. -These are logical backend calls, not a wire-level HTTP request counter. -Concurrent call durations overlap and must not be added to infer wall time. - -The default report labels its historical data `periodic-251`. -`--random-payload` selects deterministic command-seeded xorshift64 bytes and -labels them `xorshift64-command-seeded`. Generation is outside the timed SQL -transaction. This is workload data, not a cryptographic random generator. - -```sh -TMPDIR="$HOME/Workspace/crabbuild-target/crab-8bc8/tmp" \ -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-8bc8" \ - cargo run --release --locked --manifest-path \ - crates/crab-ltx/perf/replica-cost/Cargo.toml -- \ - --activation sparse --random-payload --payload-bytes 300000 --commands 12 --warmup 4 \ - --endpoint http://127.0.0.1:19010 --bucket crab-cell-issue-fleet \ - --access-key crab --secret-key crab -``` - -Three real RustFS smoke runs at `6226f0445c1`, each with eight measured -preparations after four warmups, produced these aggregate counts: - -| Payload | HEAD | PUT | Multipart start / part / complete | Mean logical bytes / root | -| --- | ---: | ---: | ---: | ---: | -| 4 KiB periodic | 16 | 40 | 0 / 0 / 0 | 7,468 | -| 4 KiB random | 16 | 40 | 0 / 0 / 0 | 13,627 | -| 300,000 bytes random | 16 | 51 | 8 / 8 / 8 | 356,575 | - -All calls completed successfully. The small-root path still performs two -metadata presence checks and five immutable PUTs per measured command. The -larger body crosses the bounded single-PUT threshold. Raw `io-committed-*.json` -samples and source/binary metadata are retained beside the comparison above. -These short runs verify observation wiring and expose payload sensitivity; -they do not establish tail latency, runtime CAS cost, compaction cost, or -sustainable throughput. Keep the missing-root-metadata refusal invariant when -evaluating those two HEADs. - -### Sparse activation over real RustFS - -The scale example at `42b7eb8e0c9` measures cold and reused metadata at one, -four, and eight shared I/O slots. The library includes bounded leaf overlap -from `b2c3b51bede`. See the [runnable example](../examples/README.md#rustfs-scale-workload) -for setup, JSON fields, and cache semantics. This experiment compares admission -settings on the same implementation; it is not a previous-revision comparison. - -Two release processes on 2026-09-25 (local time) used SQLite `randomblob` -payloads of 32 MiB and 256 MiB. Each process produced three samples per -slot/cache pair, reversing slot order in the middle round. The table gives -median checksum-preparation milliseconds; calls/bytes were identical in all -three cold samples for each size and slot count. - -| Payload | Metadata cache | 1 slot | 4 slots | 8 slots | Origin calls / bytes | -| --- | --- | ---: | ---: | ---: | ---: | -| 32 MiB | Cold | 100.464 | 74.721 | 73.979 | 33 / 723,272 | -| 32 MiB | Reused | 52.158 | 78.214 | 83.789 | 0 / 0 | -| 256 MiB | Cold | 1,057.293 | 419.734 | 295.320 | 259 / 5,797,768 | -| 256 MiB | Reused | 119.020 | 57.479 | 71.758 | 0 / 0 | - -The databases contained 8,207 and 65,626 pages of 4 KiB. Every activation -queried its restored row successfully and materialized four pages. Both -processes deleted the source and passed byte-identical full restore and -compaction restore. Cold checksum preparation benefits from overlap in these -samples; the zero-origin phase still has substantial and variable local work. -Wider concurrency does not improve every phase: cold writable-open medians at -256 MiB were 107, 82, and 126 ms for one, four, and eight slots. Three samples -per setting do not establish tails or a universally optimal concurrency. - -The first query made two range calls totaling 524,994 bytes at 32 MiB and -265,332 bytes at 256 MiB. The current VFS requests up to 64 contiguous -same-object pages per miss. This exposes a demand-read versus read-ahead -tradeoff to qualify with scans and hydration before changing policy. - -Environment: native ARM64 release binary on macOS 26.5.2, external APFS volume, -Rust 1.97.0, SQLite 3.49.1, workspace `object_store` 0.14.1. RustFS ran in the -existing Colima ARM64 VM (8 CPUs, approximately 16 GiB), using -`1.0.0-beta.8-glibc` at digest -`sha256:040304b66e029a5cde4bed140b41513e925909839a9b912a40a98340610d1f66`. -Provider connections and provider-side caches were reused. This is a local -library probe without per-node cgroup limits, runtime ownership/CAS, followers, -concurrent application traffic, or a persistent directory cache. - -Raw JSON lines and phase summaries: `32mib.log` and `256mib.log` under -`$HOME/Workspace/crabbuild-target/crab-8bc8/activation-probe/`. Binary SHA-256: -`0be9ec6e3015dff76aea2a7c769c87a0087335d265cd86a32a66cd638954ac47`. -The same directory retains source/environment metadata and stderr. Keep this -microbenchmark separate from public-action and fleet qualification. - -### Concurrent GA activation under one vCPU (2026-09-27) - -The scale example's [activation burst](../examples/README.md#concurrent-activation-and-first-write-recovery) -was run with four distinct Cells at 32 MiB and 256 MiB per Cell. Both release -processes used the same Linux ARM64 binary, pinned RustFS 1.0 GA and kernel -limits of one vCPU, 1 GiB memory and no swap. Within each process all four -rounds reopened the same four authenticated roots with fresh metadata -identities and local files. Serial/burst order was `1, 4, 4, 1`. - -The measured interval ends when every Cell has read its first 1 MiB payload, -committed a local replacement, captured it and prepared its next immutable root. -Full verification starts after that interval. Times below are whole-round -observations, not percentiles: - -| Payload per Cell | Serial, round 0 | Four concurrent, round 1 | Four concurrent, round 2 | Serial, round 3 | -| --- | ---: | ---: | ---: | ---: | -| 32 MiB | 139.252 ms | 81.941 ms | 107.235 ms | 136.621 ms | -| 256 MiB | 243.549 ms | 393.284 ms | 542.418 ms | 290.040 ms | - -At 256 MiB, per-Cell checksum-phase medians were 28.770 / 167.682 / -355.429 / 31.504 ms in round order. That phase includes shared dirty-admission -waiting and local checksum-file work as well as object reads; it is not a -provider-only timer. It performed 259–260 observed reads and returned -5,797,768–5,797,912 bytes per Cell. At 32 MiB, it performed 33 reads and -returned 723,272 bytes. The larger case therefore exposes eager metadata cost -and interference that the smaller case does not qualify away. - -Both sizes passed all sixteen first-write recovery checks: after removing the -local database and captured cut, a fresh object-root restore matched every -payload against the independent source-plus-replacement model. Across both -processes this checks 4,608 payloads of 1 MiB each. All sixteen mutations per -process have distinct IDs and prepared roots. The 256 MiB case also exercises -eight successive bootstrap capture batches for each independent Cell graph. -The original full-source and compacted-root byte comparisons also passed. - -The replacement is a compressible 1 MiB value with a unique mutation ID; the -source payloads use SQLite `randomblob`. The measured write follows a complete -read of that row, so it does not measure a write-first cold fault. Prepared -object totals were 8 / 102,814 bytes at 32 MiB and 9 / 121,989 bytes at 256 MiB. -These are immutable proposals; the probe performs no authority CAS or public -application acknowledgement. - -Peak charged cgroup memory was 205.9 MiB at 32 MiB and 940.1 MiB at 256 MiB. -The latter process recorded 68 throttled CPU periods and 591,933 microseconds -of throttled time; the smaller process recorded none. Neither had OOM events. -These process-wide counters include bootstrap, the preceding eighteen single -activations, verification and final compaction. They cannot attribute the -burst slowdown to CPU throttling or estimate four resident Cells' RSS. -Default LTX Host admission had 32 I/O slots, one blocking slot, one dirty slot -and two recovery slots. SQLite operations used Tokio blocking dispatch; this -does not reproduce the runtime's fixed SQL-worker assignment or node ledger. - -Source: `411ab72b291` plus the opt-in activation-burst example patch, excluding -the separately staged app-to-host relocation. Rust 1.97.1 and SQLite 3.49.1; -RustFS image -`ghcr.io/rustfs/rustfs:1.0.0-glibc@sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`. -Four-CPU/eight-GiB Colima VM; scratch used a Docker local volume. Raw logs, -source patch/hashes, binary hash, container inspections, cgroup counters and -independently checked JSON summaries are retained under -`worker-profile-20260927/activation-burst/evidence/` in the checkout's external -target; the `256/` child holds the larger run. The example defaults remain -unchanged when `--activation-cells` is omitted. - -Do not raise production recovery concurrency from the 32 MiB result. First -measure the same mixed recovery/application workload through `CellNode` and -its admission ledger, including write-first faults, larger or fragmented -roots, unaffected resident Cells and phase-specific resource counters. These -two processes do not establish sustained capacity, service tails or independent -failure-domain recovery. - -### Write-first GA activation under one vCPU (2026-09-27) - -The burst probe now makes the update its first application SQL, omitting the -payload query that warmed the earlier measurement. Each JSON sample marks -`access_order: "write_first"`. The separate eighteen single-Cell activation -samples still exercise first reads. Both 32 MiB and 256 MiB cases used four -distinct authenticated graphs, fresh local files and metadata identities, -shared default Host admission, and verified one-vCPU/one-GiB/no-swap limits. - -Whole-round time until all four successor roots were prepared: - -| Payload per Cell | Serial, round 0 | Four concurrent, round 1 | Four concurrent, round 2 | Serial, round 3 | -| --- | ---: | ---: | ---: | ---: | -| 32 MiB | 186.978 ms | 108.865 ms | 696.506 ms | 216.590 ms | -| 256 MiB | 262.668 ms | 252.677 ms | 255.855 ms | 368.397 ms | - -Every first update fetched origin data: six reads and 1,580,738 bytes in the -32 MiB case; six to ten reads and 1,321,076–1,380,708 bytes in the larger case. -These counters include Store GET/range/HEAD operations, excluding provider -retries. Writable-open and checksum preparation have separate counters. The -larger checksum phase still reads about 5.8 MB per Cell before the mutation. - -The 696.506 ms round is retained. Its per-Cell checksum, writable-open and -mutation medians were 210.337, 77.084 and 229.205 ms. No CPU throttling was -recorded in that process; these overlapping durations do not establish the -cause of the delay. Two serial and two concurrent rounds cannot establish -percentiles or a concurrency default. The new processes create different random -source payloads from the preceding read-before-write runs, so those records -are not a controlled before/after latency comparison. - -All 32 first-write outcomes survived deletion of local database/capture state; -4,608 full 1 MiB payload digests matched the independent source-plus-mutation -model. Both whole-source and compacted-root byte comparisons passed. Replacements -remain compressible and have distinct mutation IDs and prepared roots. Peak -charged cgroup memory was 205.5 MiB and 932.5 MiB, with no OOM events. The larger -process recorded 69 throttled periods and 640,927 microseconds of throttled -time. Resource counters include setup, verification and compaction. - -Source: `bf672cf1c64` plus the write-first example change, with the staged -app-to-host relocation excluded. All other tracked build inputs matched that -commit. Native Linux ARM64 release build, Rust 1.97.1, the same pinned RustFS -1.0 GA image and four-CPU/eight-GiB Colima VM described above. Raw samples, -source and executable hashes, Docker inspection and kernel counters are retained -under `worker-profile-20260927/write-first/evidence/` in the external target. -An initial evidence-mount failure ran no workload; its log is also retained. - -This closes the probe's read-before-write measurement gap. The roots remain -immutable proposals without authority CAS or application acknowledgements. -Public `CellNode` actions, fixed-worker interference, fragmented roots, -sustained arrivals and failure-domain recovery remain separate qualification. - -### One-GiB write-first activation and memory pressure (2026-09-27) - -The same probe now has a larger-root diagnostic: four distinct Cells with -32 MiB, 128 MiB, or 1 GiB of random payload per Cell. Each size used a fresh -process with one vCPU, 1 GiB memory, and no swap, checked both through Docker -inspection and the container's cgroup files. The dedicated Colima VM had four -vCPUs and 8 GiB memory. RustFS 1.0 GA ran separately in that VM. This extends -the earlier 256 MiB write-first proof; it is not a before/after optimization. - -All four successor roots were prepared after these whole-round durations: - -| Payload per Cell | Serial, round 0 | Four concurrent, round 1 | Four concurrent, round 2 | Serial, round 3 | -| --- | ---: | ---: | ---: | ---: | -| 32 MiB | 198.619 ms | 172.046 ms | 256.476 ms | 716.875 ms | -| 128 MiB | 193.424 ms | 113.298 ms | 345.851 ms | 558.794 ms | -| 1 GiB | 2,049.918 ms | 1,609.130 ms | 2,804.233 ms | 789.726 ms | - -Checksum preparation remains proportional to the complete authenticated page -directory, even though the first application statement changes only one 1 MiB -payload. Across each size's sixteen write-first samples: - -| Payload per Cell | Checksum-phase median | Checksum-phase Store reads | Checksum-phase bytes | First-update median | -| --- | ---: | ---: | ---: | ---: | -| 32 MiB | 5.094 ms | 33 | 723,272 | 135.438 ms | -| 128 MiB | 44.315 ms | 129–130 | 2,891,848–2,899,104 | 33.734 ms | -| 1 GiB | 372.207 ms | 1,031–1,032 | 23,189,568–23,189,880 | 69.434 ms | - -At 1 GiB, the first update itself fetched 1,366,714 bytes in eight Store reads -for every sample. Preparation walks directory leaves to create the verified -checksum sidecar; it does not fetch the database's page bodies. These measured -phase costs make checksum preparation a candidate for further work. They do -not establish the cause of the slower burst rounds: phases overlap across -Cells, and the source payload and cache state differ between processes. - -All 48 first-write results survived deletion of their local database and cut -files. Independent restores checked 18,944 complete payload digests, including -16,384 in the 1 GiB case; all mutation IDs and prepared roots were distinct -within each process. Each process also passed whole-source and compacted-root -byte comparisons and exited zero. The eighteen separate first-read activation -samples per process remained enabled. - -Whole-process charged-memory peaks were 231.52, 536.97, and 1,026.57 MiB. The -1 GiB case recorded 134,516 `memory.events:max` events, with zero `oom` and -`oom_kill` events. The kernel-reported peak is retained exactly despite being -slightly above the configured limit. It includes filesystem cache, setup, -full-restore verification, and compaction; it is neither reader RSS nor a -per-Cell resident-memory requirement. No phase-local memory series was -collected, so these events cannot be attributed specifically to activation. -The result proves completion under pressure, not memory headroom. CPU throttle -time was 4,377 / 183,187 / 506,063 microseconds for the three processes. - -Source: `f9d54fc36a6f47eb873b071cd35e04c7885c87bf`; its LTX/runtime production -sources match main `311105eb864ca90fc08bf62d3bfa6ef5c8991e2a`. The native Linux -ARM64 release executable has SHA-256 -`4fa48edce05baf50ffc7b9bc804ef149befa1352453650e982b4f75cd396a0b4`. -The RustFS image ID is -`sha256:bffcab0c9d647aab0055d1c69d340b202d0909966b385932d4ead1aeb7602858`; -the Rust build/worker image ID is -`sha256:0e2bcaef56d041a486784e54104a81aebe0da44bd03019bd70bc0401e42e4a97`. -The isolated VM imported those exact cached images and used `pull_policy: never`. - -Reproduce with `qualification/worker-profile.compose.yaml` in -`crab-cell-runtime`: run the release example with `--activation-cells 4`, -`CRAB_LTX_WORKLOAD_ROOT=/scratch`, and `CRAB_CELL_LTX_TARGET_BYTES` set to -`33554432`, `134217728`, then `1073741824`, using a separate worker process -for each size. Preserve the worker limits and retain its kernel counters, -container inspection, exit status, source, executable hash, and complete log. -Raw evidence is retained in the external per-checkout target under -`activation-profile-f9d54fc-20260927/evidence/`; a small evidence copy is under -`$HOME/.codex/cell-catalog-guard/activation-profile/`. The log SHA-256 values are: - -- 32 MiB: `5d15b302e23d39f90bcb4c2e60bfb167a097170be8505bf04e0f48153d53163f` -- 128 MiB: `251b54d695130a3582ded135659e82ec95ca4dda2ed8211e313038c4b56689e1` -- 1 GiB: `54c02312fe0b17db680f4abf8de01bae2d258ff44fd03b9c7ebac2cfec2c6d94` - -The roots are immutable proposals, without Cell authority CAS, node SQL-worker -sharding, HTTP requests, or application acknowledgements. Two serial and two -concurrent rounds do not qualify tail latency, supported Cell sizes, service -throughput, or sustained capacity. Those public-host gates remain open. - -The runners also use the implementations' pinned bundled SQLite versions: -Crab currently links SQLite 3.49.1 while the pinned Celld revision links SQLite -3.45.0. `workload_write_us` and therefore `total_us` include that difference; -use the capture and recovery subtotals when attributing work specifically to -the LTX implementations. - -Run each workload matrix on the same machine, filesystem, SQLite page size, -build profile, and power state. Use the median as a compact summary, but retain -the per-round JSON when investigating variance. diff --git a/crates/crab-ltx/perf/celld/Cargo.lock b/crates/crab-ltx/perf/celld/Cargo.lock deleted file mode 100644 index 15dc61655..000000000 --- a/crates/crab-ltx/perf/celld/Cargo.lock +++ /dev/null @@ -1,1483 +0,0 @@ -# This file is automatically @generated by Cargo. -# It is not intended for manual editing. -version = 4 - -[[package]] -name = "adler2" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" - -[[package]] -name = "ahash" -version = "0.8.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a15f179cd60c4584b8a8c596927aadc462e27f2ca70c04e0071964a73ba7a75" -dependencies = [ - "cfg-if", - "once_cell", - "version_check", - "zerocopy", -] - -[[package]] -name = "android_system_properties" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" -dependencies = [ - "libc", -] - -[[package]] -name = "async-trait" -version = "0.1.92" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "autocfg" -version = "1.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" - -[[package]] -name = "base64" -version = "0.23.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" - -[[package]] -name = "bitflags" -version = "2.13.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" - -[[package]] -name = "bumpalo" -version = "3.20.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" - -[[package]] -name = "bytes" -version = "1.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" - -[[package]] -name = "cc" -version = "1.4.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "54413ede23c2daf518f35156dfde027feb2374004d63bd497f983c8db9c0e313" -dependencies = [ - "find-msvc-tools", - "shlex", -] - -[[package]] -name = "celld-ltx" -version = "0.0.0" -source = "git+https://github.com/denoland/celld.git?rev=10cb1303dac710dcb3b557e318e08c855261f68b#10cb1303dac710dcb3b557e318e08c855261f68b" -dependencies = [ - "async-trait", - "bytes", - "chrono", - "crc-fast", - "futures-util", - "http", - "lz4_flex", - "object_store", - "percent-encoding", - "rusqlite", - "thiserror", - "tokio", - "tracing", - "ureq", -] - -[[package]] -name = "cfg-if" -version = "1.0.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" - -[[package]] -name = "chrono" -version = "0.4.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" -dependencies = [ - "iana-time-zone", - "num-traits", - "windows-link", -] - -[[package]] -name = "core-foundation-sys" -version = "0.8.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" - -[[package]] -name = "crab-ltx-perf-celld" -version = "0.1.0" -dependencies = [ - "celld-ltx", - "rusqlite", - "serde", - "serde_json", - "tempfile", - "tokio", -] - -[[package]] -name = "crc-fast" -version = "1.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5" -dependencies = [ - "digest", - "spin", -] - -[[package]] -name = "crc32fast" -version = "1.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78" -dependencies = [ - "cfg-if", -] - -[[package]] -name = "crypto-common" -version = "0.1.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" -dependencies = [ - "generic-array", - "typenum", -] - -[[package]] -name = "digest" -version = "0.10.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" -dependencies = [ - "crypto-common", -] - -[[package]] -name = "displaydoc" -version = "0.2.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "either" -version = "1.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" - -[[package]] -name = "errno" -version = "0.3.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" -dependencies = [ - "libc", - "windows-sys 0.61.2", -] - -[[package]] -name = "fallible-iterator" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" - -[[package]] -name = "fallible-streaming-iterator" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" - -[[package]] -name = "fastrand" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" - -[[package]] -name = "find-msvc-tools" -version = "0.1.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef25905e51abafe4dcea6c15fec58c57b601cdbd0ee53d22ea1d3016c587d39b" - -[[package]] -name = "flate2" -version = "1.1.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" -dependencies = [ - "crc32fast", - "miniz_oxide", - "zlib-rs", -] - -[[package]] -name = "form_urlencoded" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" -dependencies = [ - "percent-encoding", -] - -[[package]] -name = "futures" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a31d2a3fbaaeb2af2368bbdd904aa8e812d3c04a1ee10d3171f52d556e5d0a3" -dependencies = [ - "futures-channel", - "futures-core", - "futures-executor", - "futures-io", - "futures-sink", - "futures-task", - "futures-util", -] - -[[package]] -name = "futures-channel" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" -dependencies = [ - "futures-core", - "futures-sink", -] - -[[package]] -name = "futures-core" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" - -[[package]] -name = "futures-executor" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "031b47cf1a3c6cc8bc2fc76cd437f521619387907d469316e7c0bc278f1f5432" -dependencies = [ - "futures-core", - "futures-task", - "futures-util", -] - -[[package]] -name = "futures-io" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" - -[[package]] -name = "futures-macro" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "futures-sink" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" - -[[package]] -name = "futures-task" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" - -[[package]] -name = "futures-util" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" -dependencies = [ - "futures-channel", - "futures-core", - "futures-io", - "futures-macro", - "futures-sink", - "futures-task", - "memchr", - "pin-project-lite", - "slab", -] - -[[package]] -name = "generic-array" -version = "0.14.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" -dependencies = [ - "typenum", - "version_check", -] - -[[package]] -name = "getrandom" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" -dependencies = [ - "cfg-if", - "libc", - "wasi", -] - -[[package]] -name = "getrandom" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" -dependencies = [ - "cfg-if", - "libc", - "r-efi", -] - -[[package]] -name = "hashbrown" -version = "0.14.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e5274423e17b7c9fc20b6e7e208532f9b19825d82dfd615708b70edd83df41f1" -dependencies = [ - "ahash", -] - -[[package]] -name = "hashlink" -version = "0.9.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ba4ff7128dee98c7dc9794b6a411377e1404dba1c97deb8d1a55297bd25d8af" -dependencies = [ - "hashbrown", -] - -[[package]] -name = "http" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" -dependencies = [ - "bytes", - "itoa", -] - -[[package]] -name = "httparse" -version = "1.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" - -[[package]] -name = "humantime" -version = "2.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" - -[[package]] -name = "iana-time-zone" -version = "0.1.65" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" -dependencies = [ - "android_system_properties", - "core-foundation-sys", - "iana-time-zone-haiku", - "js-sys", - "log", - "wasm-bindgen", - "windows-core", -] - -[[package]] -name = "iana-time-zone-haiku" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" -dependencies = [ - "cc", -] - -[[package]] -name = "icu_collections" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" -dependencies = [ - "displaydoc", - "potential_utf", - "utf8_iter", - "yoke", - "zerofrom", - "zerovec", -] - -[[package]] -name = "icu_locale_core" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" -dependencies = [ - "displaydoc", - "litemap", - "tinystr", - "writeable", - "zerovec", -] - -[[package]] -name = "icu_normalizer" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" -dependencies = [ - "icu_collections", - "icu_normalizer_data", - "icu_properties", - "icu_provider", - "smallvec", - "zerovec", -] - -[[package]] -name = "icu_normalizer_data" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" - -[[package]] -name = "icu_properties" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" -dependencies = [ - "displaydoc", - "icu_collections", - "icu_locale_core", - "icu_properties_data", - "icu_provider", - "zerotrie", - "zerovec", -] - -[[package]] -name = "icu_properties_data" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" - -[[package]] -name = "icu_provider" -version = "2.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" -dependencies = [ - "displaydoc", - "icu_locale_core", - "writeable", - "yoke", - "zerofrom", - "zerotrie", - "zerovec", -] - -[[package]] -name = "idna" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3b0875f23caa03898994f6ddc501886a45c7d3d62d04d2d90788d47be1b1e4de" -dependencies = [ - "idna_adapter", - "smallvec", - "utf8_iter", -] - -[[package]] -name = "idna_adapter" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb68373c0d6620ef8105e855e7745e18b0d00d3bdb07fb532e434244cdb9a714" -dependencies = [ - "icu_normalizer", - "icu_properties", -] - -[[package]] -name = "itertools" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2b192c782037fadd9cfa75548310488aabdbf3d2da73885b31bd0abd03351285" -dependencies = [ - "either", -] - -[[package]] -name = "itoa" -version = "1.0.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" - -[[package]] -name = "js-sys" -version = "0.3.105" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce57d20d1ea864ce2ac172ab472d409214f4fd359f0b2a2775abdf522e2af99e" -dependencies = [ - "cfg-if", - "futures-util", - "wasm-bindgen", -] - -[[package]] -name = "libc" -version = "0.2.189" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" - -[[package]] -name = "libsqlite3-sys" -version = "0.28.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c10584274047cb335c23d3e61bcef8e323adae7c5c8c760540f73610177fc3f" -dependencies = [ - "cc", - "pkg-config", - "vcpkg", -] - -[[package]] -name = "linux-raw-sys" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" - -[[package]] -name = "litemap" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" - -[[package]] -name = "lock_api" -version = "0.4.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "224399e74b87b5f3557511d98dff8b14089b3dadafcab6bb93eab67d3aace965" -dependencies = [ - "scopeguard", -] - -[[package]] -name = "log" -version = "0.4.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" - -[[package]] -name = "lz4_flex" -version = "0.11.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a" -dependencies = [ - "twox-hash", -] - -[[package]] -name = "memchr" -version = "2.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" - -[[package]] -name = "miniz_oxide" -version = "0.9.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" -dependencies = [ - "adler2", - "simd-adler32", -] - -[[package]] -name = "num-traits" -version = "0.2.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" -dependencies = [ - "autocfg", -] - -[[package]] -name = "object_store" -version = "0.12.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fbfbfff40aeccab00ec8a910b57ca8ecf4319b335c542f2edcd19dd25a1e2a00" -dependencies = [ - "async-trait", - "bytes", - "chrono", - "futures", - "http", - "humantime", - "itertools", - "parking_lot", - "percent-encoding", - "thiserror", - "tokio", - "tracing", - "url", - "wasm-bindgen-futures", - "web-time", -] - -[[package]] -name = "once_cell" -version = "1.21.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" - -[[package]] -name = "parking_lot" -version = "0.12.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "93857453250e3077bd71ff98b6a65ea6621a19bb0f559a85248955ac12c45a1a" -dependencies = [ - "lock_api", - "parking_lot_core", -] - -[[package]] -name = "parking_lot_core" -version = "0.9.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2621685985a2ebf1c516881c026032ac7deafcda1a2c9b7850dc81e3dfcb64c1" -dependencies = [ - "cfg-if", - "libc", - "redox_syscall", - "smallvec", - "windows-link", -] - -[[package]] -name = "percent-encoding" -version = "2.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" - -[[package]] -name = "pin-project-lite" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" - -[[package]] -name = "pkg-config" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" - -[[package]] -name = "potential_utf" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" -dependencies = [ - "zerovec", -] - -[[package]] -name = "proc-macro2" -version = "1.0.107" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "quote" -version = "1.0.47" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" -dependencies = [ - "proc-macro2", -] - -[[package]] -name = "r-efi" -version = "6.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" - -[[package]] -name = "redox_syscall" -version = "0.5.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" -dependencies = [ - "bitflags", -] - -[[package]] -name = "ring" -version = "0.17.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7" -dependencies = [ - "cc", - "cfg-if", - "getrandom 0.2.17", - "libc", - "untrusted", - "windows-sys 0.52.0", -] - -[[package]] -name = "rusqlite" -version = "0.31.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b838eba278d213a8beaf485bd313fd580ca4505a00d5871caeb1457c55322cae" -dependencies = [ - "bitflags", - "fallible-iterator", - "fallible-streaming-iterator", - "hashlink", - "libsqlite3-sys", - "smallvec", -] - -[[package]] -name = "rustix" -version = "1.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "891efababe418670775f199f0d233d84843c227a0949a883ce15b37c78d6629d" -dependencies = [ - "bitflags", - "errno", - "libc", - "linux-raw-sys", - "windows-sys 0.61.2", -] - -[[package]] -name = "rustls" -version = "0.23.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d41d731c7d2f962d1ccc364cec258de3c0e93b38c2fb3ba97ac74513048d634" -dependencies = [ - "log", - "once_cell", - "ring", - "rustls-pki-types", - "rustls-webpki", - "subtle", - "zeroize", -] - -[[package]] -name = "rustls-pki-types" -version = "1.15.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" -dependencies = [ - "zeroize", -] - -[[package]] -name = "rustls-webpki" -version = "0.103.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" -dependencies = [ - "ring", - "rustls-pki-types", - "untrusted", -] - -[[package]] -name = "rustversion" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" - -[[package]] -name = "scopeguard" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" - -[[package]] -name = "serde" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" -dependencies = [ - "serde_core", - "serde_derive", -] - -[[package]] -name = "serde_core" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" -dependencies = [ - "serde_derive", -] - -[[package]] -name = "serde_derive" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "serde_json" -version = "1.0.151" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" -dependencies = [ - "itoa", - "memchr", - "serde", - "serde_core", - "zmij", -] - -[[package]] -name = "shlex" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" - -[[package]] -name = "simd-adler32" -version = "0.3.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" - -[[package]] -name = "slab" -version = "0.4.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" - -[[package]] -name = "smallvec" -version = "1.16.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba467056f1b547ed52077911161fc86985becbc60e8e1857c8a144dab0def891" - -[[package]] -name = "spin" -version = "0.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" - -[[package]] -name = "stable_deref_trait" -version = "1.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" - -[[package]] -name = "subtle" -version = "2.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" - -[[package]] -name = "syn" -version = "2.0.119" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "syn" -version = "3.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "synstructure" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "901704edd0dfe137f1987838ee4f259e4e063c31371bdb423f7ae38ec6f77f02" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tempfile" -version = "3.27.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" -dependencies = [ - "fastrand", - "getrandom 0.4.3", - "once_cell", - "rustix", - "windows-sys 0.61.2", -] - -[[package]] -name = "thiserror" -version = "2.0.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" -dependencies = [ - "thiserror-impl", -] - -[[package]] -name = "thiserror-impl" -version = "2.0.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tinystr" -version = "0.8.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" -dependencies = [ - "displaydoc", - "zerovec", -] - -[[package]] -name = "tokio" -version = "1.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" -dependencies = [ - "bytes", - "pin-project-lite", - "tokio-macros", -] - -[[package]] -name = "tokio-macros" -version = "2.7.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tracing" -version = "0.1.44" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" -dependencies = [ - "pin-project-lite", - "tracing-attributes", - "tracing-core", -] - -[[package]] -name = "tracing-attributes" -version = "0.1.31" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "tracing-core" -version = "0.1.36" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" -dependencies = [ - "once_cell", -] - -[[package]] -name = "twox-hash" -version = "2.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" - -[[package]] -name = "typenum" -version = "1.20.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" - -[[package]] -name = "unicode-ident" -version = "1.0.26" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" - -[[package]] -name = "untrusted" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" - -[[package]] -name = "ureq" -version = "3.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a7ac20be9b7726e0bbdbf974c059676d9acb1cd414961f570a4e8231cacd7fc" -dependencies = [ - "base64", - "flate2", - "log", - "percent-encoding", - "rustls", - "rustls-pki-types", - "ureq-proto", - "utf8-zero", - "webpki-roots", -] - -[[package]] -name = "ureq-proto" -version = "0.6.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f86fd172ccca569e458f61b6bdd6220965a9ef36e672a6852953b51a0e1583be" -dependencies = [ - "base64", - "http", - "httparse", - "log", -] - -[[package]] -name = "url" -version = "2.5.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff67a8a4397373c3ef660812acab3268222035010ab8680ec4215f38ba3d0eed" -dependencies = [ - "form_urlencoded", - "idna", - "percent-encoding", - "serde", -] - -[[package]] -name = "utf8-zero" -version = "0.8.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8c0a043c9540bae7c578c88f91dda8bd82e59ae27c21baca69c8b191aaf5a6e" - -[[package]] -name = "utf8_iter" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" - -[[package]] -name = "vcpkg" -version = "0.2.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" - -[[package]] -name = "version_check" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" - -[[package]] -name = "wasi" -version = "0.11.1+wasi-snapshot-preview1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" - -[[package]] -name = "wasm-bindgen" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "aecb87a33d3b0c5e3b7aa46336eaf486cffafbd281b195e4c8b80d50df2351bf" -dependencies = [ - "cfg-if", - "once_cell", - "rustversion", - "wasm-bindgen-macro", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-futures" -version = "0.4.78" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ef4c5d3d2cdf5c54f4231181768f5510842e350db025faf1f7163b1030ed928" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "wasm-bindgen-macro" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a690d511e3c1a8b3a55e33511e3c2c00c78415cd23650f32b808627f5696b9ed" -dependencies = [ - "quote", - "wasm-bindgen-macro-support", -] - -[[package]] -name = "wasm-bindgen-macro-support" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "411e4887f0071ef2d2164a9d5fdf2d20efbef78fccd3a78b0c10a1dc5295e48a" -dependencies = [ - "bumpalo", - "proc-macro2", - "quote", - "syn 3.0.6", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-shared" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "81941cd78d0c92026c33e5e01312845a4cb1e9af3407f9134b100dd03144103e" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "web-time" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a6580f308b1fad9207618087a65c04e7a10bc77e02c8e84e9b00dd4b12fa0bb" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "webpki-roots" -version = "1.0.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7dcd9d09a39985f5344844e66b0c530a33843579125f23e21e9f0f220850f22a" -dependencies = [ - "rustls-pki-types", -] - -[[package]] -name = "windows-core" -version = "0.62.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" -dependencies = [ - "windows-implement", - "windows-interface", - "windows-link", - "windows-result", - "windows-strings", -] - -[[package]] -name = "windows-implement" -version = "0.60.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "windows-interface" -version = "0.59.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "windows-link" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" - -[[package]] -name = "windows-result" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-strings" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-sys" -version = "0.52.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" -dependencies = [ - "windows-targets", -] - -[[package]] -name = "windows-sys" -version = "0.61.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-targets" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" -dependencies = [ - "windows_aarch64_gnullvm", - "windows_aarch64_msvc", - "windows_i686_gnu", - "windows_i686_gnullvm", - "windows_i686_msvc", - "windows_x86_64_gnu", - "windows_x86_64_gnullvm", - "windows_x86_64_msvc", -] - -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" - -[[package]] -name = "windows_aarch64_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" - -[[package]] -name = "windows_i686_gnu" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" - -[[package]] -name = "windows_i686_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" - -[[package]] -name = "windows_i686_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" - -[[package]] -name = "windows_x86_64_gnu" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" - -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" - -[[package]] -name = "windows_x86_64_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" - -[[package]] -name = "writeable" -version = "0.6.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" - -[[package]] -name = "yoke" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" -dependencies = [ - "stable_deref_trait", - "yoke-derive", - "zerofrom", -] - -[[package]] -name = "yoke-derive" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "33811428bee40dbceb6d545e95754741d17a6aef9a4849f0fd62e2ba4f412a78" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", - "synstructure", -] - -[[package]] -name = "zerocopy" -version = "0.8.57" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d35102a9f36d089ccae9e4c6802bc118be4487b80aaffc0ab4e0cf5ce92d2873" -dependencies = [ - "zerocopy-derive", -] - -[[package]] -name = "zerocopy-derive" -version = "0.8.57" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "146c01f5ab44258da43cf276c74a2763db2ff3969c9c652c3f2de07041d0b2bc" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "zerofrom" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" -dependencies = [ - "zerofrom-derive", -] - -[[package]] -name = "zerofrom-derive" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f75b4683f6c7f45248d4d64056a24298c6281e0993356d7d1b4a1a962ef10d4a" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", - "synstructure", -] - -[[package]] -name = "zeroize" -version = "1.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" - -[[package]] -name = "zerotrie" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" -dependencies = [ - "displaydoc", - "yoke", - "zerofrom", -] - -[[package]] -name = "zerovec" -version = "0.11.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" -dependencies = [ - "yoke", - "zerofrom", - "zerovec-derive", -] - -[[package]] -name = "zerovec-derive" -version = "0.11.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "zlib-rs" -version = "0.6.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112" - -[[package]] -name = "zmij" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/crates/crab-ltx/perf/celld/Cargo.toml b/crates/crab-ltx/perf/celld/Cargo.toml deleted file mode 100644 index 0b7e6ccf3..000000000 --- a/crates/crab-ltx/perf/celld/Cargo.toml +++ /dev/null @@ -1,16 +0,0 @@ -[package] -name = "crab-ltx-perf-celld" -version = "0.1.0" -edition = "2021" -publish = false - -[workspace] -resolver = "2" - -[dependencies] -celld-ltx = { git = "https://github.com/denoland/celld.git", rev = "10cb1303dac710dcb3b557e318e08c855261f68b", default-features = false } -rusqlite = { version = "0.31", features = ["bundled"] } -serde = { version = "1", features = ["derive"] } -serde_json = "1" -tempfile = "3" -tokio = { version = "1", features = ["macros", "rt-multi-thread", "sync"] } diff --git a/crates/crab-ltx/perf/celld/src/main.rs b/crates/crab-ltx/perf/celld/src/main.rs deleted file mode 100644 index a66ef4ab5..000000000 --- a/crates/crab-ltx/perf/celld/src/main.rs +++ /dev/null @@ -1,364 +0,0 @@ -use celld_ltx::{Db, FileReplicaClient, TXID}; -use rusqlite::Connection; -use serde::Serialize; -use std::error::Error; -use std::path::{Path, PathBuf}; -use std::sync::Arc; -use std::time::{Duration, Instant}; -use tokio::sync::Semaphore; - -const CELLD_SOURCE: &str = "celld 10cb1303dac710dcb3b557e318e08c855261f68b"; - -#[derive(Debug, Clone, Copy)] -struct Config { - transactions: usize, - payload_bytes: usize, - rounds: usize, - warmup: usize, - sync_parent: bool, -} - -#[derive(Debug, Clone, Serialize)] -struct Sample { - round: usize, - workload_write_us: u64, - capture_us: u64, - verify_us: u64, - compact_us: u64, - compact_verify_us: u64, - restore_us: u64, - restore_plan_us: u64, - restore_download_us: u64, - restore_apply_us: u64, - recovery_us: u64, - total_us: u64, - segments: usize, - input_ltx_bytes: u64, - compacted_ltx_bytes: u64, - source_database_bytes: u64, - final_txid: u64, -} - -#[derive(Debug, Serialize)] -struct Report { - implementation: &'static str, - source: &'static str, - config: ConfigOutput, - samples: Vec, - median: Summary, -} - -#[derive(Debug, Serialize)] -struct ConfigOutput { - transactions: usize, - payload_bytes: usize, - measured_rounds: usize, - warmup_rounds: usize, - sync_parent: bool, -} - -#[derive(Debug, Serialize)] -struct Summary { - workload_write_us: u64, - capture_us: u64, - verify_us: u64, - compact_us: u64, - compact_verify_us: u64, - restore_us: u64, - restore_plan_us: u64, - restore_download_us: u64, - restore_apply_us: u64, - recovery_us: u64, - total_us: u64, - segments: usize, - input_ltx_bytes: u64, - compacted_ltx_bytes: u64, - source_database_bytes: u64, - final_txid: u64, -} - -#[tokio::main(flavor = "multi_thread")] -async fn main() -> Result<(), Box> { - let config = Config::from_args(std::env::args().skip(1))?; - let mut samples = Vec::with_capacity(config.rounds); - for round in 0..config.warmup + config.rounds { - let sample = run_round(config, round).await?; - if round >= config.warmup { - samples.push(sample); - } - } - - let report = Report { - implementation: "celld-ltx", - source: CELLD_SOURCE, - config: ConfigOutput { - transactions: config.transactions, - payload_bytes: config.payload_bytes, - measured_rounds: config.rounds, - warmup_rounds: config.warmup, - sync_parent: config.sync_parent, - }, - median: Summary::from_samples(&samples), - samples, - }; - println!("{}", serde_json::to_string_pretty(&report)?); - Ok(()) -} - -impl Config { - fn from_args(args: I) -> Result> - where - I: IntoIterator, - { - let args = args.into_iter().collect::>(); - let transactions = option(&args, "--transactions")?.unwrap_or(128); - let payload_bytes = option(&args, "--payload-bytes")?.unwrap_or(4096); - let rounds = option(&args, "--rounds")?.unwrap_or(5); - let warmup = option(&args, "--warmup")?.unwrap_or(1); - let sync_parent = args.iter().any(|arg| arg == "--sync-parent"); - if transactions == 0 || payload_bytes == 0 || rounds == 0 { - return Err("transactions, payload-bytes, and rounds must be positive".into()); - } - Ok(Self { - transactions, - payload_bytes, - rounds, - warmup, - sync_parent, - }) - } -} - -fn option(args: &[String], name: &str) -> Result, Box> { - let Some(index) = args.iter().position(|arg| arg == name) else { - return Ok(None); - }; - let value = args - .get(index + 1) - .ok_or_else(|| format!("missing value for {name}"))? - .parse::() - .map_err(|error| format!("invalid value for {name}: {error}"))?; - Ok(Some(value)) -} - -async fn run_round(config: Config, round: usize) -> Result> { - let directory = tempfile::tempdir()?; - let source = directory.path().join("source.sqlite"); - let mut ltx_db = Db::open(&source)?; - let mut writer = Connection::open(&source)?; - writer.busy_timeout(Duration::from_secs(1))?; - writer.pragma_update(None, "wal_autocheckpoint", 0)?; - writer.pragma_update(None, "synchronous", "FULL")?; - writer.pragma_update(None, "foreign_keys", true)?; - - let mut workload_write_us = 0; - let mut capture_us = 0; - let started = Instant::now(); - writer.execute_batch("CREATE TABLE payloads (id INTEGER PRIMARY KEY, value BLOB NOT NULL)")?; - workload_write_us += elapsed_us(started); - let started = Instant::now(); - ltx_db.sync()?; - if config.sync_parent { - sync_ltx_parent(ltx_db.meta_path(), 0)?; - sync_ltx_ancestors(ltx_db.meta_path())?; - } - capture_us += elapsed_us(started); - - for id in 0..config.transactions { - let payload = payload(id, config.payload_bytes); - let started = Instant::now(); - let transaction = writer.transaction()?; - transaction.execute( - "INSERT INTO payloads (id, value) VALUES (?1, ?2)", - rusqlite::params![id as i64, payload.as_slice()], - )?; - transaction.commit()?; - workload_write_us += elapsed_us(started); - - let started = Instant::now(); - ltx_db.sync()?; - if config.sync_parent { - sync_ltx_parent(ltx_db.meta_path(), 0)?; - } - capture_us += elapsed_us(started); - } - - let meta_path = ltx_db.meta_path().to_path_buf(); - ltx_db.close()?; - drop(writer); - let source_database_bytes = std::fs::metadata(&source)?.len(); - let input_files = list_ltx_files(&meta_path, 0)?; - let input_ltx_bytes = input_files.iter().map(|(_, bytes)| *bytes).sum::(); - let segments = input_files.len(); - let client = FileReplicaClient::new(meta_path.to_string_lossy().into_owned()); - - let started = Instant::now(); - let output = celld_ltx::replica_compactor::ReplicaCompactor::new(&client) - .with_local_path(&meta_path) - .with_verification(true) - .compact(1) - .await? - .ok_or("Celld compaction produced no output")?; - if config.sync_parent { - sync_ltx_parent(&meta_path, 1)?; - sync_parent(&meta_path.join("ltx").join("1"))?; - } - let compact_us = elapsed_us(started); - let compacted_ltx_bytes = u64::try_from(output.info.size) - .map_err(|_| "Celld compaction returned a negative output size")?; - - let restored = directory.path().join("restored.sqlite"); - let started = Instant::now(); - let timing = celld_ltx::replica::restore_timed_with_download_slots( - &client, - &restored, - TXID::ZERO, - Arc::new(Semaphore::new(1)), - ) - .await?; - if config.sync_parent { - sync_parent(&restored)?; - } - let restore_us = elapsed_us(started); - validate_restore(&restored, config.transactions)?; - - let recovery_us = recovery_us(0, compact_us, 0, restore_us); - let total_us = total_us(workload_write_us, capture_us, recovery_us); - - Ok(Sample { - round, - workload_write_us, - capture_us, - verify_us: 0, - compact_us, - compact_verify_us: 0, - restore_us, - restore_plan_us: timing.plan_us, - restore_download_us: timing.download_us, - restore_apply_us: timing.apply_us, - recovery_us, - total_us, - segments, - input_ltx_bytes, - compacted_ltx_bytes, - source_database_bytes, - final_txid: output.info.max_txid.0, - }) -} - -fn sync_ltx_parent(meta_path: &Path, level: u32) -> Result<(), Box> { - std::fs::File::open(meta_path.join("ltx").join(level.to_string()))?.sync_all()?; - Ok(()) -} - -fn sync_ltx_ancestors(meta_path: &Path) -> Result<(), Box> { - let ltx = meta_path.join("ltx"); - sync_parent(<x.join("0"))?; - sync_parent(<x)?; - sync_parent(meta_path) -} - -fn sync_parent(path: &Path) -> Result<(), Box> { - let parent = path.parent().filter(|path| !path.as_os_str().is_empty()); - std::fs::File::open(parent.unwrap_or_else(|| Path::new(".")))?.sync_all()?; - Ok(()) -} - -fn list_ltx_files(root: &Path, level: u32) -> Result, Box> { - let directory = root.join("ltx").join(level.to_string()); - let mut files = Vec::new(); - if !directory.exists() { - return Ok(files); - } - for entry in std::fs::read_dir(directory)? { - let entry = entry?; - let path = entry.path(); - if path.extension().is_some_and(|extension| extension == "ltx") { - files.push((path.clone(), std::fs::metadata(path)?.len())); - } - } - files.sort_by(|left, right| left.0.cmp(&right.0)); - Ok(files) -} - -fn validate_restore(path: &Path, transactions: usize) -> Result<(), Box> { - let connection = Connection::open(path)?; - let count: i64 = connection.query_row("SELECT COUNT(*) FROM payloads", [], |row| row.get(0))?; - if count != transactions as i64 { - return Err(format!("restored {count} rows, expected {transactions}").into()); - } - let integrity: String = connection.query_row("PRAGMA integrity_check", [], |row| row.get(0))?; - if integrity != "ok" { - return Err(format!("restored database integrity check: {integrity}").into()); - } - Ok(()) -} - -fn payload(id: usize, bytes: usize) -> Vec { - (0..bytes) - .map(|offset| ((id.wrapping_mul(31).wrapping_add(offset)) % 251) as u8) - .collect() -} - -fn elapsed_us(started: Instant) -> u64 { - started.elapsed().as_micros().min(u128::from(u64::MAX)) as u64 -} - -impl Summary { - fn from_samples(samples: &[Sample]) -> Self { - Self { - workload_write_us: median(samples.iter().map(|sample| sample.workload_write_us)), - capture_us: median(samples.iter().map(|sample| sample.capture_us)), - verify_us: median(samples.iter().map(|sample| sample.verify_us)), - compact_us: median(samples.iter().map(|sample| sample.compact_us)), - compact_verify_us: median(samples.iter().map(|sample| sample.compact_verify_us)), - restore_us: median(samples.iter().map(|sample| sample.restore_us)), - restore_plan_us: median(samples.iter().map(|sample| sample.restore_plan_us)), - restore_download_us: median(samples.iter().map(|sample| sample.restore_download_us)), - restore_apply_us: median(samples.iter().map(|sample| sample.restore_apply_us)), - recovery_us: median(samples.iter().map(|sample| sample.recovery_us)), - total_us: median(samples.iter().map(|sample| sample.total_us)), - segments: median(samples.iter().map(|sample| sample.segments as u64)) as usize, - input_ltx_bytes: median(samples.iter().map(|sample| sample.input_ltx_bytes)), - compacted_ltx_bytes: median(samples.iter().map(|sample| sample.compacted_ltx_bytes)), - source_database_bytes: median( - samples.iter().map(|sample| sample.source_database_bytes), - ), - final_txid: median(samples.iter().map(|sample| sample.final_txid)), - } - } -} - -fn recovery_us(verify_us: u64, compact_us: u64, compact_verify_us: u64, restore_us: u64) -> u64 { - verify_us - .saturating_add(compact_us) - .saturating_add(compact_verify_us) - .saturating_add(restore_us) -} - -fn total_us(workload_write_us: u64, capture_us: u64, recovery_us: u64) -> u64 { - workload_write_us - .saturating_add(capture_us) - .saturating_add(recovery_us) -} - -fn median(values: impl Iterator) -> u64 { - let mut values = values.collect::>(); - values.sort_unstable(); - values[values.len() / 2] -} - -#[cfg(test)] -mod tests { - use super::{recovery_us, total_us}; - - #[test] - fn recovery_subtotal_includes_each_phase_once() { - assert_eq!(recovery_us(11, 13, 17, 19), 60); - } - - #[test] - fn total_includes_workload_capture_and_recovery_once() { - assert_eq!(total_us(23, 29, 31), 83); - } -} diff --git a/crates/crab-ltx/perf/crab/Cargo.lock b/crates/crab-ltx/perf/crab/Cargo.lock deleted file mode 100644 index 9a72f09ef..000000000 --- a/crates/crab-ltx/perf/crab/Cargo.lock +++ /dev/null @@ -1,452 +0,0 @@ -# This file is automatically @generated by Cargo. -# It is not intended for manual editing. -version = 4 - -[[package]] -name = "arrayvec" -version = "0.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" - -[[package]] -name = "bitflags" -version = "2.13.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" - -[[package]] -name = "blake3" -version = "1.8.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d9e454fc11f76977dc803893aff6304ed33d6a26efae8696573bea74baa27ae" -dependencies = [ - "arrayvec", - "cc", - "cfg-if", - "constant_time_eq", - "cpufeatures", -] - -[[package]] -name = "cc" -version = "1.4.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "54413ede23c2daf518f35156dfde027feb2374004d63bd497f983c8db9c0e313" -dependencies = [ - "find-msvc-tools", - "shlex", -] - -[[package]] -name = "cfg-if" -version = "1.0.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" - -[[package]] -name = "constant_time_eq" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b" - -[[package]] -name = "cpufeatures" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ca28b0ae3115b884660db4118d803791fd6756b6e88f39c0f3f7859060d7566" -dependencies = [ - "libc", -] - -[[package]] -name = "crab-ltx" -version = "0.1.0" -dependencies = [ - "blake3", - "crc-fast", - "lz4_flex", - "rusqlite", - "tempfile", - "thiserror", -] - -[[package]] -name = "crab-ltx-perf-crab" -version = "0.1.0" -dependencies = [ - "crab-ltx", - "serde", - "serde_json", - "tempfile", -] - -[[package]] -name = "crc-fast" -version = "1.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5" -dependencies = [ - "digest", - "spin", -] - -[[package]] -name = "crypto-common" -version = "0.1.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" -dependencies = [ - "generic-array", - "typenum", -] - -[[package]] -name = "digest" -version = "0.10.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" -dependencies = [ - "crypto-common", -] - -[[package]] -name = "errno" -version = "0.3.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" -dependencies = [ - "libc", - "windows-sys", -] - -[[package]] -name = "fallible-iterator" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" - -[[package]] -name = "fallible-streaming-iterator" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" - -[[package]] -name = "fastrand" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" - -[[package]] -name = "find-msvc-tools" -version = "0.1.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef25905e51abafe4dcea6c15fec58c57b601cdbd0ee53d22ea1d3016c587d39b" - -[[package]] -name = "foldhash" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" - -[[package]] -name = "generic-array" -version = "0.14.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" -dependencies = [ - "typenum", - "version_check", -] - -[[package]] -name = "getrandom" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" -dependencies = [ - "cfg-if", - "libc", - "r-efi", -] - -[[package]] -name = "hashbrown" -version = "0.15.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" -dependencies = [ - "foldhash", -] - -[[package]] -name = "hashlink" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7382cf6263419f2d8df38c55d7da83da5c18aef87fc7a7fc1fb1e344edfe14c1" -dependencies = [ - "hashbrown", -] - -[[package]] -name = "itoa" -version = "1.0.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" - -[[package]] -name = "libc" -version = "0.2.189" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" - -[[package]] -name = "libsqlite3-sys" -version = "0.32.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fbb8270bb4060bd76c6e96f20c52d80620f1d82a3470885694e41e0f81ef6fe7" -dependencies = [ - "cc", - "pkg-config", - "vcpkg", -] - -[[package]] -name = "linux-raw-sys" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" - -[[package]] -name = "lz4_flex" -version = "0.11.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a" -dependencies = [ - "twox-hash", -] - -[[package]] -name = "memchr" -version = "2.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" - -[[package]] -name = "once_cell" -version = "1.21.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" - -[[package]] -name = "pkg-config" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" - -[[package]] -name = "proc-macro2" -version = "1.0.107" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "quote" -version = "1.0.47" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" -dependencies = [ - "proc-macro2", -] - -[[package]] -name = "r-efi" -version = "6.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" - -[[package]] -name = "rusqlite" -version = "0.34.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "37e34486da88d8e051c7c0e23c3f15fd806ea8546260aa2fec247e97242ec143" -dependencies = [ - "bitflags", - "fallible-iterator", - "fallible-streaming-iterator", - "hashlink", - "libsqlite3-sys", - "smallvec", -] - -[[package]] -name = "rustix" -version = "1.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "891efababe418670775f199f0d233d84843c227a0949a883ce15b37c78d6629d" -dependencies = [ - "bitflags", - "errno", - "libc", - "linux-raw-sys", - "windows-sys", -] - -[[package]] -name = "serde" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" -dependencies = [ - "serde_core", - "serde_derive", -] - -[[package]] -name = "serde_core" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" -dependencies = [ - "serde_derive", -] - -[[package]] -name = "serde_derive" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" -dependencies = [ - "proc-macro2", - "quote", - "syn", -] - -[[package]] -name = "serde_json" -version = "1.0.151" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" -dependencies = [ - "itoa", - "memchr", - "serde", - "serde_core", - "zmij", -] - -[[package]] -name = "shlex" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" - -[[package]] -name = "smallvec" -version = "1.16.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba467056f1b547ed52077911161fc86985becbc60e8e1857c8a144dab0def891" - -[[package]] -name = "spin" -version = "0.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" - -[[package]] -name = "syn" -version = "3.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "tempfile" -version = "3.27.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" -dependencies = [ - "fastrand", - "getrandom", - "once_cell", - "rustix", - "windows-sys", -] - -[[package]] -name = "thiserror" -version = "2.0.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" -dependencies = [ - "thiserror-impl", -] - -[[package]] -name = "thiserror-impl" -version = "2.0.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" -dependencies = [ - "proc-macro2", - "quote", - "syn", -] - -[[package]] -name = "twox-hash" -version = "2.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" - -[[package]] -name = "typenum" -version = "1.20.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" - -[[package]] -name = "unicode-ident" -version = "1.0.26" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" - -[[package]] -name = "vcpkg" -version = "0.2.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" - -[[package]] -name = "version_check" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" - -[[package]] -name = "windows-link" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" - -[[package]] -name = "windows-sys" -version = "0.61.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" -dependencies = [ - "windows-link", -] - -[[package]] -name = "zmij" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/crates/crab-ltx/perf/crab/Cargo.toml b/crates/crab-ltx/perf/crab/Cargo.toml deleted file mode 100644 index afea92bdb..000000000 --- a/crates/crab-ltx/perf/crab/Cargo.toml +++ /dev/null @@ -1,14 +0,0 @@ -[package] -name = "crab-ltx-perf-crab" -version = "0.1.0" -edition = "2024" -publish = false - -[workspace] -resolver = "2" - -[dependencies] -crab-ltx = { path = "../../", default-features = false } -serde = { version = "1", features = ["derive"] } -serde_json = "1" -tempfile = "3" diff --git a/crates/crab-ltx/perf/crab/src/main.rs b/crates/crab-ltx/perf/crab/src/main.rs deleted file mode 100644 index dbc5ced28..000000000 --- a/crates/crab-ltx/perf/crab/src/main.rs +++ /dev/null @@ -1,483 +0,0 @@ -use crab_ltx::{CaptureTiming, Db, Limits, Position, VerifiedPlan, compact_exact, restore_exact}; -use serde::Serialize; -use std::error::Error; -use std::path::Path; -use std::time::Instant; - -const CRAB_SOURCE: &str = "workspace crab-ltx"; - -#[derive(Debug, Clone, Copy)] -struct Config { - transactions: usize, - payload_bytes: usize, - rounds: usize, - warmup: usize, - durability_batch: usize, -} - -#[derive(Debug, Clone, Serialize)] -struct Sample { - round: usize, - workload_write_us: u64, - capture_us: u64, - capture_position_us: u64, - capture_wal_read_us: u64, - capture_page_collection_us: u64, - capture_verification_us: u64, - capture_encode_us: u64, - capture_local_write_us: u64, - capture_fsync_us: u64, - capture_parent_sync_us: u64, - capture_barrier_us: u64, - verify_us: u64, - compact_us: u64, - compact_verify_us: u64, - restore_us: u64, - recovery_us: u64, - total_us: u64, - segments: usize, - input_ltx_bytes: u64, - compacted_ltx_bytes: u64, - source_database_bytes: u64, - final_txid: u64, -} - -#[derive(Debug, Serialize)] -struct Report { - implementation: &'static str, - source: &'static str, - config: ConfigOutput, - samples: Vec, - median: Summary, -} - -#[derive(Debug, Serialize)] -struct ConfigOutput { - transactions: usize, - payload_bytes: usize, - measured_rounds: usize, - warmup_rounds: usize, - durability_batch: usize, -} - -#[derive(Debug, Serialize)] -struct Summary { - workload_write_us: u64, - capture_us: u64, - capture_position_us: u64, - capture_wal_read_us: u64, - capture_page_collection_us: u64, - capture_verification_us: u64, - capture_encode_us: u64, - capture_local_write_us: u64, - capture_fsync_us: u64, - capture_parent_sync_us: u64, - capture_barrier_us: u64, - verify_us: u64, - compact_us: u64, - compact_verify_us: u64, - restore_us: u64, - recovery_us: u64, - total_us: u64, - segments: usize, - input_ltx_bytes: u64, - compacted_ltx_bytes: u64, - source_database_bytes: u64, - final_txid: u64, -} - -fn main() -> Result<(), Box> { - let config = Config::from_args(std::env::args().skip(1))?; - let mut samples = Vec::with_capacity(config.rounds); - for round in 0..config.warmup + config.rounds { - let sample = run_round(config, round)?; - if round >= config.warmup { - samples.push(sample); - } - } - - let report = Report { - implementation: "crab-ltx", - source: CRAB_SOURCE, - config: ConfigOutput { - transactions: config.transactions, - payload_bytes: config.payload_bytes, - measured_rounds: config.rounds, - warmup_rounds: config.warmup, - durability_batch: config.durability_batch, - }, - median: Summary::from_samples(&samples), - samples, - }; - println!("{}", serde_json::to_string_pretty(&report)?); - Ok(()) -} - -impl Config { - fn from_args(args: I) -> Result> - where - I: IntoIterator, - { - let args = args.into_iter().collect::>(); - let transactions = option(&args, "--transactions")?.unwrap_or(128); - let payload_bytes = option(&args, "--payload-bytes")?.unwrap_or(4096); - let rounds = option(&args, "--rounds")?.unwrap_or(5); - let warmup = option(&args, "--warmup")?.unwrap_or(1); - let durability_batch = option(&args, "--durability-batch")?.unwrap_or(1); - if transactions == 0 || payload_bytes == 0 || rounds == 0 || durability_batch == 0 { - return Err( - "transactions, payload-bytes, rounds, and durability-batch must be positive".into(), - ); - } - Ok(Self { - transactions, - payload_bytes, - rounds, - warmup, - durability_batch, - }) - } -} - -fn option(args: &[String], name: &str) -> Result, Box> { - let Some(index) = args.iter().position(|arg| arg == name) else { - return Ok(None); - }; - let value = args - .get(index + 1) - .ok_or_else(|| format!("missing value for {name}"))? - .parse::() - .map_err(|error| format!("invalid value for {name}: {error}"))?; - Ok(Some(value)) -} - -fn run_round(config: Config, round: usize) -> Result> { - let directory = tempfile::tempdir()?; - let source = directory.path().join("source.sqlite"); - let mut database = Db::open(&source, Limits::default())?; - let mut segments = Vec::new(); - let mut position = Position::default(); - let mut capture_phases = CapturePhases::default(); - let mut capture_barrier_us = 0; - let mut pending_captures = 0; - let mut workload_write_us = 0; - let mut capture_us = 0; - - let started = Instant::now(); - database.transaction(|transaction| { - transaction - .execute_batch("CREATE TABLE payloads (id INTEGER PRIMARY KEY, value BLOB NOT NULL)") - })?; - workload_write_us += elapsed_us(started); - let started = Instant::now(); - append_capture( - &mut database, - &mut segments, - &mut position, - &mut capture_phases, - config.durability_batch, - &mut pending_captures, - &mut capture_barrier_us, - )?; - capture_us += elapsed_us(started); - - for id in 0..config.transactions { - let payload = payload(id, config.payload_bytes); - let started = Instant::now(); - database.transaction(|transaction| { - transaction.execute( - "INSERT INTO payloads (id, value) VALUES (?1, ?2)", - crab_ltx::rusqlite::params![id as i64, payload.as_slice()], - )?; - Ok(()) - })?; - workload_write_us += elapsed_us(started); - - let started = Instant::now(); - append_capture( - &mut database, - &mut segments, - &mut position, - &mut capture_phases, - config.durability_batch, - &mut pending_captures, - &mut capture_barrier_us, - )?; - capture_us += elapsed_us(started); - } - - if pending_captures > 0 { - let started = Instant::now(); - database.durability_barrier()?; - let elapsed = elapsed_us(started); - capture_barrier_us = capture_barrier_us.saturating_add(elapsed); - capture_us = capture_us.saturating_add(elapsed); - } - - database.close()?; - let source_database_bytes = std::fs::metadata(&source)?.len(); - let limits = Limits::default(); - - let started = Instant::now(); - let plan = VerifiedPlan::new(&segments, position, limits)?; - let verify_us = elapsed_us(started); - - let compact_path = directory.path().join("compacted.ltx"); - let started = Instant::now(); - let compacted = compact_exact(&plan, &compact_path)?; - let compact_us = elapsed_us(started); - - let started = Instant::now(); - let compact_plan = VerifiedPlan::new(std::slice::from_ref(&compacted), position, limits)?; - let compact_verify_us = elapsed_us(started); - - let restored = directory.path().join("restored.sqlite"); - let started = Instant::now(); - restore_exact(&compact_plan, &restored)?; - let restore_us = elapsed_us(started); - validate_restore(&restored, config.transactions)?; - - let recovery_us = recovery_us(verify_us, compact_us, compact_verify_us, restore_us); - let total_us = total_us(workload_write_us, capture_us, recovery_us); - - Ok(Sample { - round, - workload_write_us, - capture_us, - capture_position_us: capture_phases.position_us(), - capture_wal_read_us: capture_phases.wal_read_us(), - capture_page_collection_us: capture_phases.page_collection_us(), - capture_verification_us: capture_phases.verification_us(), - capture_encode_us: capture_phases.encode_us(), - capture_local_write_us: capture_phases.local_write_us(), - capture_fsync_us: capture_phases.fsync_us(), - capture_parent_sync_us: capture_phases.parent_sync_us(), - capture_barrier_us, - verify_us, - compact_us, - compact_verify_us, - restore_us, - recovery_us, - total_us, - segments: segments.len(), - input_ltx_bytes: segments - .iter() - .map(|segment| segment.info().size_bytes) - .sum(), - compacted_ltx_bytes: compacted.info().size_bytes, - source_database_bytes, - final_txid: position.txid, - }) -} - -fn append_capture( - database: &mut Db, - segments: &mut Vec, - position: &mut Position, - phases: &mut CapturePhases, - durability_batch: usize, - pending_captures: &mut usize, - capture_barrier_us: &mut u64, -) -> crab_ltx::Result<()> { - let batch = if durability_batch > 1 { - database.capture_deferred()? - } else { - database.capture()? - }; - *position = batch.position; - phases.add(batch.timing); - segments.extend(batch.segments); - if durability_batch > 1 { - *pending_captures = pending_captures.saturating_add(1); - if *pending_captures == durability_batch { - let started = Instant::now(); - database.durability_barrier()?; - *capture_barrier_us = capture_barrier_us.saturating_add(elapsed_us(started)); - *pending_captures = 0; - } - } - Ok(()) -} - -#[derive(Debug, Default)] -struct CapturePhases { - position_nanos: u64, - wal_read_nanos: u64, - page_collection_nanos: u64, - verification_nanos: u64, - encode_nanos: u64, - local_write_nanos: u64, - fsync_nanos: u64, - parent_sync_nanos: u64, -} - -impl CapturePhases { - fn add(&mut self, timing: CaptureTiming) { - self.position_nanos = self - .position_nanos - .saturating_add(timing.position_resolution_nanos); - self.wal_read_nanos = self.wal_read_nanos.saturating_add(timing.wal_read_nanos); - self.page_collection_nanos = self - .page_collection_nanos - .saturating_add(timing.page_collection_nanos); - self.verification_nanos = self - .verification_nanos - .saturating_add(timing.verification_nanos); - self.encode_nanos = self.encode_nanos.saturating_add(timing.encode_nanos); - self.local_write_nanos = self - .local_write_nanos - .saturating_add(timing.local_write_nanos); - self.fsync_nanos = self.fsync_nanos.saturating_add(timing.fsync_nanos); - self.parent_sync_nanos = self - .parent_sync_nanos - .saturating_add(timing.parent_sync_nanos); - } - - fn position_us(&self) -> u64 { - nanos_to_us(self.position_nanos) - } - - fn wal_read_us(&self) -> u64 { - nanos_to_us(self.wal_read_nanos) - } - - fn page_collection_us(&self) -> u64 { - nanos_to_us(self.page_collection_nanos) - } - - fn verification_us(&self) -> u64 { - nanos_to_us(self.verification_nanos) - } - - fn encode_us(&self) -> u64 { - nanos_to_us(self.encode_nanos) - } - - fn local_write_us(&self) -> u64 { - nanos_to_us(self.local_write_nanos) - } - - fn fsync_us(&self) -> u64 { - nanos_to_us(self.fsync_nanos) - } - - fn parent_sync_us(&self) -> u64 { - nanos_to_us(self.parent_sync_nanos) - } -} - -fn validate_restore(path: &Path, transactions: usize) -> Result<(), Box> { - let connection = crab_ltx::rusqlite::Connection::open(path)?; - let count: i64 = connection.query_row("SELECT COUNT(*) FROM payloads", [], |row| row.get(0))?; - if count != transactions as i64 { - return Err(format!("restored {count} rows, expected {transactions}").into()); - } - let integrity: String = connection.query_row("PRAGMA integrity_check", [], |row| row.get(0))?; - if integrity != "ok" { - return Err(format!("restored database integrity check: {integrity}").into()); - } - Ok(()) -} - -fn payload(id: usize, bytes: usize) -> Vec { - (0..bytes) - .map(|offset| ((id.wrapping_mul(31).wrapping_add(offset)) % 251) as u8) - .collect() -} - -fn elapsed_us(started: Instant) -> u64 { - started.elapsed().as_micros().min(u128::from(u64::MAX)) as u64 -} - -fn nanos_to_us(nanos: u64) -> u64 { - nanos / 1_000 -} - -impl Summary { - fn from_samples(samples: &[Sample]) -> Self { - Self { - workload_write_us: median(samples.iter().map(|sample| sample.workload_write_us)), - capture_us: median(samples.iter().map(|sample| sample.capture_us)), - capture_position_us: median(samples.iter().map(|sample| sample.capture_position_us)), - capture_wal_read_us: median(samples.iter().map(|sample| sample.capture_wal_read_us)), - capture_page_collection_us: median( - samples - .iter() - .map(|sample| sample.capture_page_collection_us), - ), - capture_verification_us: median( - samples.iter().map(|sample| sample.capture_verification_us), - ), - capture_encode_us: median(samples.iter().map(|sample| sample.capture_encode_us)), - capture_local_write_us: median( - samples.iter().map(|sample| sample.capture_local_write_us), - ), - capture_fsync_us: median(samples.iter().map(|sample| sample.capture_fsync_us)), - capture_parent_sync_us: median( - samples.iter().map(|sample| sample.capture_parent_sync_us), - ), - capture_barrier_us: median(samples.iter().map(|sample| sample.capture_barrier_us)), - verify_us: median(samples.iter().map(|sample| sample.verify_us)), - compact_us: median(samples.iter().map(|sample| sample.compact_us)), - compact_verify_us: median(samples.iter().map(|sample| sample.compact_verify_us)), - restore_us: median(samples.iter().map(|sample| sample.restore_us)), - recovery_us: median(samples.iter().map(|sample| sample.recovery_us)), - total_us: median(samples.iter().map(|sample| sample.total_us)), - segments: median(samples.iter().map(|sample| sample.segments as u64)) as usize, - input_ltx_bytes: median(samples.iter().map(|sample| sample.input_ltx_bytes)), - compacted_ltx_bytes: median(samples.iter().map(|sample| sample.compacted_ltx_bytes)), - source_database_bytes: median( - samples.iter().map(|sample| sample.source_database_bytes), - ), - final_txid: median(samples.iter().map(|sample| sample.final_txid)), - } - } -} - -fn recovery_us(verify_us: u64, compact_us: u64, compact_verify_us: u64, restore_us: u64) -> u64 { - verify_us - .saturating_add(compact_us) - .saturating_add(compact_verify_us) - .saturating_add(restore_us) -} - -fn total_us(workload_write_us: u64, capture_us: u64, recovery_us: u64) -> u64 { - workload_write_us - .saturating_add(capture_us) - .saturating_add(recovery_us) -} - -fn median(values: impl Iterator) -> u64 { - let mut values = values.collect::>(); - values.sort_unstable(); - values[values.len() / 2] -} - -#[cfg(test)] -mod tests { - use super::{Config, recovery_us, total_us}; - - #[test] - fn durability_batch_defaults_to_synchronous_capture() { - let config = Config::from_args(Vec::new()).unwrap(); - - assert_eq!(config.durability_batch, 1); - } - - #[test] - fn durability_batch_must_be_positive() { - let error = Config::from_args(["--durability-batch".into(), "0".into()]).unwrap_err(); - - assert!(error.to_string().contains("durability-batch")); - } - - #[test] - fn recovery_subtotal_includes_each_phase_once() { - assert_eq!(recovery_us(11, 13, 17, 19), 60); - } - - #[test] - fn total_includes_workload_capture_and_recovery_once() { - assert_eq!(total_us(23, 29, 31), 83); - } -} diff --git a/crates/crab-ltx/perf/replica-cost/Cargo.lock b/crates/crab-ltx/perf/replica-cost/Cargo.lock deleted file mode 100644 index b4cc5306f..000000000 --- a/crates/crab-ltx/perf/replica-cost/Cargo.lock +++ /dev/null @@ -1,2398 +0,0 @@ -# This file is automatically @generated by Cargo. -# It is not intended for manual editing. -version = 4 - -[[package]] -name = "android_system_properties" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" -dependencies = [ - "libc", -] - -[[package]] -name = "arrayvec" -version = "0.7.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" - -[[package]] -name = "async-trait" -version = "0.1.92" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "atomic-waker" -version = "1.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" - -[[package]] -name = "autocfg" -version = "1.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" - -[[package]] -name = "aws-lc-rs" -version = "1.18.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b281d307588d634de920874890732659e2e7672f72b5e10e81badc1a8a83621e" -dependencies = [ - "aws-lc-sys", - "zeroize", -] - -[[package]] -name = "aws-lc-sys" -version = "0.45.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9bff6c3b54fad79a2e60b8102caf565819711497c1f5f092f49508e2f5c31b27" -dependencies = [ - "cc", - "cmake", - "dunce", - "fs_extra", - "pkg-config", -] - -[[package]] -name = "base64" -version = "0.22.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" - -[[package]] -name = "base64" -version = "0.23.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" - -[[package]] -name = "bitflags" -version = "2.13.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" - -[[package]] -name = "blake3" -version = "1.8.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d9e454fc11f76977dc803893aff6304ed33d6a26efae8696573bea74baa27ae" -dependencies = [ - "arrayvec", - "cc", - "cfg-if", - "constant_time_eq", - "cpufeatures", -] - -[[package]] -name = "block-buffer" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2f6c7dbe95a6ed67ad9f18e57daf93a2f034c524b99fd2b76d18fdfeb6660aa" -dependencies = [ - "hybrid-array", -] - -[[package]] -name = "bumpalo" -version = "3.20.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" - -[[package]] -name = "bytes" -version = "1.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" - -[[package]] -name = "cc" -version = "1.4.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "54413ede23c2daf518f35156dfde027feb2374004d63bd497f983c8db9c0e313" -dependencies = [ - "find-msvc-tools", - "jobserver", - "libc", - "shlex", -] - -[[package]] -name = "cfg-if" -version = "1.0.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" - -[[package]] -name = "cfg_aliases" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" - -[[package]] -name = "chacha20" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "65c35e4b699c7e15ccbe7ee35c005e4fc0a278d22238a2857e6ce2dadeda1b06" -dependencies = [ - "cfg-if", - "cpufeatures", - "rand_core 0.10.1", -] - -[[package]] -name = "chrono" -version = "0.4.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" -dependencies = [ - "iana-time-zone", - "num-traits", - "serde", - "windows-link", -] - -[[package]] -name = "cmake" -version = "0.1.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c0f78a02292a74a88ac736019ab962ece0bc380e3f977bf72e376c5d78ff0678" -dependencies = [ - "cc", -] - -[[package]] -name = "combine" -version = "4.6.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cfc320937d09e6de266b31b9afb480f197d7a861be86be7cb2ea7e5d1bfffc5e" -dependencies = [ - "bytes", - "memchr", -] - -[[package]] -name = "constant_time_eq" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b" - -[[package]] -name = "core-foundation" -version = "0.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b2a6cd9ae233e7f62ba4e9353e81a88df7fc8a5987b8d445b4d90c879bd156f6" -dependencies = [ - "core-foundation-sys", - "libc", -] - -[[package]] -name = "core-foundation-sys" -version = "0.8.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" - -[[package]] -name = "cpufeatures" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ca28b0ae3115b884660db4118d803791fd6756b6e88f39c0f3f7859060d7566" -dependencies = [ - "libc", -] - -[[package]] -name = "crab-ltx" -version = "0.1.0" -dependencies = [ - "async-trait", - "blake3", - "bytes", - "crab-storage", - "crc-fast", - "futures-util", - "lz4_flex", - "object_store", - "rusqlite", - "serde", - "serde_json", - "tempfile", - "thiserror", - "tokio", - "tokio-util", -] - -[[package]] -name = "crab-ltx-replica-cost" -version = "0.1.0" -dependencies = [ - "crab-ltx", - "crab-storage", - "object_store", - "serde", - "serde_json", - "tempfile", - "tokio", -] - -[[package]] -name = "crab-storage" -version = "0.1.0" -dependencies = [ - "async-trait", - "blake3", - "bytes", - "crab-types", - "futures-util", - "hyper", - "object_store", - "rand 0.9.5", - "reqwest 0.12.28", - "serde", - "serde_json", - "thiserror", - "tokio", - "tokio-util", - "tracing", - "url", -] - -[[package]] -name = "crab-types" -version = "0.1.0" -dependencies = [ - "schemars", - "serde", -] - -[[package]] -name = "crc-fast" -version = "1.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5" -dependencies = [ - "digest 0.10.7", - "spin", -] - -[[package]] -name = "crypto-common" -version = "0.1.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" -dependencies = [ - "generic-array", - "typenum", -] - -[[package]] -name = "crypto-common" -version = "0.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce6e4c961d6cd6c9a86db418387425e8bdeaf05b3c8bc1411e6dca4c252f1453" -dependencies = [ - "hybrid-array", -] - -[[package]] -name = "digest" -version = "0.10.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" -dependencies = [ - "crypto-common 0.1.7", -] - -[[package]] -name = "digest" -version = "0.11.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2" -dependencies = [ - "block-buffer", - "crypto-common 0.2.2", -] - -[[package]] -name = "displaydoc" -version = "0.2.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "dunce" -version = "1.0.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813" - -[[package]] -name = "dyn-clone" -version = "1.0.20" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555" - -[[package]] -name = "either" -version = "1.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" - -[[package]] -name = "equivalent" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" - -[[package]] -name = "errno" -version = "0.3.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" -dependencies = [ - "libc", - "windows-sys 0.61.2", -] - -[[package]] -name = "fallible-iterator" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" - -[[package]] -name = "fallible-streaming-iterator" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" - -[[package]] -name = "fastrand" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" - -[[package]] -name = "find-msvc-tools" -version = "0.1.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef25905e51abafe4dcea6c15fec58c57b601cdbd0ee53d22ea1d3016c587d39b" - -[[package]] -name = "fnv" -version = "1.0.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" - -[[package]] -name = "foldhash" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" - -[[package]] -name = "form_urlencoded" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" -dependencies = [ - "percent-encoding", -] - -[[package]] -name = "fs_extra" -version = "1.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" - -[[package]] -name = "futures-channel" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" -dependencies = [ - "futures-core", - "futures-sink", -] - -[[package]] -name = "futures-core" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" - -[[package]] -name = "futures-io" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" - -[[package]] -name = "futures-macro" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "futures-sink" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" - -[[package]] -name = "futures-task" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" - -[[package]] -name = "futures-util" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" -dependencies = [ - "futures-core", - "futures-io", - "futures-macro", - "futures-sink", - "futures-task", - "memchr", - "pin-project-lite", - "slab", -] - -[[package]] -name = "generic-array" -version = "0.14.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" -dependencies = [ - "typenum", - "version_check", -] - -[[package]] -name = "getrandom" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" -dependencies = [ - "cfg-if", - "js-sys", - "libc", - "wasi", - "wasm-bindgen", -] - -[[package]] -name = "getrandom" -version = "0.3.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" -dependencies = [ - "cfg-if", - "libc", - "r-efi 5.3.0", - "wasip2", -] - -[[package]] -name = "getrandom" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" -dependencies = [ - "cfg-if", - "js-sys", - "libc", - "r-efi 6.0.0", - "rand_core 0.10.1", - "wasm-bindgen", -] - -[[package]] -name = "h2" -version = "0.4.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef8e5e5a340588f4452631496976cf8636d4a7ecf600239fdc27615d2530bc16" -dependencies = [ - "atomic-waker", - "bytes", - "fnv", - "futures-core", - "futures-sink", - "http", - "indexmap", - "slab", - "tokio", - "tokio-util", - "tracing", -] - -[[package]] -name = "hashbrown" -version = "0.15.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" -dependencies = [ - "foldhash", -] - -[[package]] -name = "hashbrown" -version = "0.17.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" - -[[package]] -name = "hashlink" -version = "0.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7382cf6263419f2d8df38c55d7da83da5c18aef87fc7a7fc1fb1e344edfe14c1" -dependencies = [ - "hashbrown 0.15.5", -] - -[[package]] -name = "http" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" -dependencies = [ - "bytes", - "itoa", -] - -[[package]] -name = "http-body" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" -dependencies = [ - "bytes", - "http", -] - -[[package]] -name = "http-body-util" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" -dependencies = [ - "bytes", - "futures-core", - "http", - "http-body", - "pin-project-lite", -] - -[[package]] -name = "httparse" -version = "1.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" - -[[package]] -name = "humantime" -version = "2.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" - -[[package]] -name = "hybrid-array" -version = "0.4.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "27f864f10dfb56725ce5ce5472bc52252c8f93a4ab86327122cebf62c5f59a17" -dependencies = [ - "typenum", -] - -[[package]] -name = "hyper" -version = "1.11.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "27b501faa50e7a26c3d3560ca625132f4078a17771f4810baf70475ae48cbe43" -dependencies = [ - "atomic-waker", - "bytes", - "futures-channel", - "futures-core", - "h2", - "http", - "http-body", - "httparse", - "itoa", - "pin-project-lite", - "smallvec", - "tokio", - "want", -] - -[[package]] -name = "hyper-rustls" -version = "0.27.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dfa8e654703247911e29c23fbeaa261834bd9bb74efba2f9acddc37bfb127f53" -dependencies = [ - "http", - "hyper", - "hyper-util", - "rustls", - "tokio", - "tokio-rustls", - "tower-service", -] - -[[package]] -name = "hyper-util" -version = "0.1.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddc03d96684f9226b8a787cdb71488417b53ab5ea8fdb1dac946cb9431cc8bff" -dependencies = [ - "base64 0.23.1", - "bytes", - "futures-channel", - "futures-util", - "http", - "http-body", - "httparse", - "hyper", - "ipnet", - "libc", - "percent-encoding", - "pin-project-lite", - "socket2", - "tokio", - "tower-service", - "tracing", -] - -[[package]] -name = "iana-time-zone" -version = "0.1.65" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" -dependencies = [ - "android_system_properties", - "core-foundation-sys", - "iana-time-zone-haiku", - "js-sys", - "log", - "wasm-bindgen", - "windows-core", -] - -[[package]] -name = "iana-time-zone-haiku" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" -dependencies = [ - "cc", -] - -[[package]] -name = "icu_collections" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" -dependencies = [ - "displaydoc", - "potential_utf", - "utf8_iter", - "yoke", - "zerofrom", - "zerovec", -] - -[[package]] -name = "icu_locale_core" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" -dependencies = [ - "displaydoc", - "litemap", - "tinystr", - "writeable", - "zerovec", -] - -[[package]] -name = "icu_normalizer" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" -dependencies = [ - "icu_collections", - "icu_normalizer_data", - "icu_properties", - "icu_provider", - "smallvec", - "zerovec", -] - -[[package]] -name = "icu_normalizer_data" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" - -[[package]] -name = "icu_properties" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" -dependencies = [ - "displaydoc", - "icu_collections", - "icu_locale_core", - "icu_properties_data", - "icu_provider", - "zerotrie", - "zerovec", -] - -[[package]] -name = "icu_properties_data" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" - -[[package]] -name = "icu_provider" -version = "2.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" -dependencies = [ - "displaydoc", - "icu_locale_core", - "writeable", - "yoke", - "zerofrom", - "zerotrie", - "zerovec", -] - -[[package]] -name = "idna" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3b0875f23caa03898994f6ddc501886a45c7d3d62d04d2d90788d47be1b1e4de" -dependencies = [ - "idna_adapter", - "smallvec", - "utf8_iter", -] - -[[package]] -name = "idna_adapter" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb68373c0d6620ef8105e855e7745e18b0d00d3bdb07fb532e434244cdb9a714" -dependencies = [ - "icu_normalizer", - "icu_properties", -] - -[[package]] -name = "indexmap" -version = "2.14.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" -dependencies = [ - "equivalent", - "hashbrown 0.17.1", -] - -[[package]] -name = "ipnet" -version = "2.12.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "791930b43c0d5973160d90a8f3894509f2b273430f5c5c73b668636d0287c5c0" - -[[package]] -name = "itertools" -version = "0.15.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b4baf93f58d4425749ca49a51c50ebab072c5df6994d08fed93541c331481dc" -dependencies = [ - "either", -] - -[[package]] -name = "itoa" -version = "1.0.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" - -[[package]] -name = "jni" -version = "0.22.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5efd9a482cf3a427f00d6b35f14332adc7902ce91efb778580e180ff90fa3498" -dependencies = [ - "cfg-if", - "combine", - "jni-macros", - "jni-sys", - "log", - "simd_cesu8", - "thiserror", - "walkdir", - "windows-link", -] - -[[package]] -name = "jni-macros" -version = "0.22.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a00109accc170f0bdb141fed3e393c565b6f5e072365c3bd58f5b062591560a3" -dependencies = [ - "proc-macro2", - "quote", - "rustc_version", - "simd_cesu8", - "syn 2.0.119", -] - -[[package]] -name = "jni-sys" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c6377a88cb3910bee9b0fa88d4f42e1d2da8e79915598f65fb0c7ee14c878af2" -dependencies = [ - "jni-sys-macros", -] - -[[package]] -name = "jni-sys-macros" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "38c0b942f458fe50cdac086d2f946512305e5631e720728f2a61aabcd47a6264" -dependencies = [ - "quote", - "syn 2.0.119", -] - -[[package]] -name = "jobserver" -version = "0.1.35" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3" -dependencies = [ - "getrandom 0.4.3", - "libc", -] - -[[package]] -name = "js-sys" -version = "0.3.105" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce57d20d1ea864ce2ac172ab472d409214f4fd359f0b2a2775abdf522e2af99e" -dependencies = [ - "cfg-if", - "futures-util", - "wasm-bindgen", -] - -[[package]] -name = "libc" -version = "0.2.189" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" - -[[package]] -name = "libsqlite3-sys" -version = "0.32.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fbb8270bb4060bd76c6e96f20c52d80620f1d82a3470885694e41e0f81ef6fe7" -dependencies = [ - "cc", - "pkg-config", - "vcpkg", -] - -[[package]] -name = "linux-raw-sys" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" - -[[package]] -name = "litemap" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" - -[[package]] -name = "lock_api" -version = "0.4.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "224399e74b87b5f3557511d98dff8b14089b3dadafcab6bb93eab67d3aace965" -dependencies = [ - "scopeguard", -] - -[[package]] -name = "log" -version = "0.4.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" - -[[package]] -name = "lru-slab" -version = "0.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4050469837a6ff301cd14c1f8f24f88549e6d548f24f64e2148eb0f72cebc51f" - -[[package]] -name = "lz4_flex" -version = "0.11.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a" -dependencies = [ - "twox-hash", -] - -[[package]] -name = "md-5" -version = "0.11.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "69b6441f590336821bb897fb28fc622898ccceb1d6cea3fde5ea86b090c4de98" -dependencies = [ - "cfg-if", - "digest 0.11.3", -] - -[[package]] -name = "memchr" -version = "2.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" - -[[package]] -name = "mio" -version = "1.2.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b18443e9c262bfe8fa82f51666e2642c53393f7e5c27b3e1aeab922cff5b9d8" -dependencies = [ - "libc", - "wasi", - "windows-sys 0.61.2", -] - -[[package]] -name = "nix" -version = "0.31.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d" -dependencies = [ - "bitflags", - "cfg-if", - "cfg_aliases", - "libc", -] - -[[package]] -name = "num-traits" -version = "0.2.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" -dependencies = [ - "autocfg", -] - -[[package]] -name = "object_store" -version = "0.14.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1796bc93603f78c5760a69f2d58badc9618d22adade0a95385bb2adbae4eb94" -dependencies = [ - "async-trait", - "aws-lc-rs", - "base64 0.23.1", - "bytes", - "chrono", - "crc-fast", - "form_urlencoded", - "futures-channel", - "futures-core", - "futures-util", - "http", - "http-body-util", - "httparse", - "humantime", - "hyper", - "itertools", - "md-5", - "nix", - "parking_lot", - "percent-encoding", - "quick-xml", - "rand 0.10.3", - "reqwest 0.13.5", - "rustls-pki-types", - "serde", - "serde_json", - "serde_urlencoded", - "thiserror", - "tokio", - "tracing", - "url", - "walkdir", - "wasm-bindgen-futures", - "web-time", - "windows-sys 0.61.2", -] - -[[package]] -name = "once_cell" -version = "1.21.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" - -[[package]] -name = "openssl-probe" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7c87def4c32ab89d880effc9e097653c8da5d6ef28e6b539d313baaacfbafcbe" - -[[package]] -name = "parking_lot" -version = "0.12.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "93857453250e3077bd71ff98b6a65ea6621a19bb0f559a85248955ac12c45a1a" -dependencies = [ - "lock_api", - "parking_lot_core", -] - -[[package]] -name = "parking_lot_core" -version = "0.9.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2621685985a2ebf1c516881c026032ac7deafcda1a2c9b7850dc81e3dfcb64c1" -dependencies = [ - "cfg-if", - "libc", - "redox_syscall", - "smallvec", - "windows-link", -] - -[[package]] -name = "percent-encoding" -version = "2.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" - -[[package]] -name = "pin-project-lite" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" - -[[package]] -name = "pkg-config" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" - -[[package]] -name = "potential_utf" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" -dependencies = [ - "zerovec", -] - -[[package]] -name = "ppv-lite86" -version = "0.2.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" -dependencies = [ - "zerocopy", -] - -[[package]] -name = "proc-macro2" -version = "1.0.107" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "quick-xml" -version = "0.41.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1" -dependencies = [ - "memchr", - "serde", -] - -[[package]] -name = "quinn" -version = "0.11.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4051e23e9185c255a7e33ef59cdbca87a22d359052eecd22fc6b901fb37d9d11" -dependencies = [ - "bytes", - "cfg_aliases", - "pin-project-lite", - "quinn-proto", - "quinn-udp", - "rustc-hash", - "rustls", - "socket2", - "thiserror", - "tokio", - "tracing", - "web-time", -] - -[[package]] -name = "quinn-proto" -version = "0.11.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9746dbde176634f4f2f1faf2404e30a31b2bc1e9cafb5329c95d8177a18c9fc" -dependencies = [ - "aws-lc-rs", - "bytes", - "getrandom 0.4.3", - "lru-slab", - "rand 0.10.3", - "rand_pcg", - "ring", - "rustc-hash", - "rustls", - "rustls-pki-types", - "slab", - "thiserror", - "tinyvec", - "tracing", - "web-time", -] - -[[package]] -name = "quinn-udp" -version = "0.5.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694" -dependencies = [ - "cfg_aliases", - "libc", - "once_cell", - "socket2", - "tracing", - "windows-sys 0.61.2", -] - -[[package]] -name = "quote" -version = "1.0.47" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" -dependencies = [ - "proc-macro2", -] - -[[package]] -name = "r-efi" -version = "5.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" - -[[package]] -name = "r-efi" -version = "6.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" - -[[package]] -name = "rand" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" -dependencies = [ - "rand_chacha", - "rand_core 0.9.5", -] - -[[package]] -name = "rand" -version = "0.10.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "65c9fb96cbc91e3478eaae79a69fcd3f1ae4ad052e471fe6732fff548984b4af" -dependencies = [ - "chacha20", - "getrandom 0.4.3", - "rand_core 0.10.1", -] - -[[package]] -name = "rand_chacha" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" -dependencies = [ - "ppv-lite86", - "rand_core 0.9.5", -] - -[[package]] -name = "rand_core" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" -dependencies = [ - "getrandom 0.3.4", -] - -[[package]] -name = "rand_core" -version = "0.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "63b8176103e19a2643978565ca18b50549f6101881c443590420e4dc998a3c69" - -[[package]] -name = "rand_pcg" -version = "0.10.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "caa0f4137e1c0a72f4c651489402276c8e8e1cf081f3b0ba156d2cbeef09e86a" -dependencies = [ - "rand_core 0.10.1", -] - -[[package]] -name = "redox_syscall" -version = "0.5.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" -dependencies = [ - "bitflags", -] - -[[package]] -name = "reqwest" -version = "0.12.28" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147" -dependencies = [ - "base64 0.22.1", - "bytes", - "futures-core", - "http", - "http-body", - "http-body-util", - "hyper", - "hyper-util", - "js-sys", - "log", - "percent-encoding", - "pin-project-lite", - "serde", - "serde_json", - "serde_urlencoded", - "sync_wrapper", - "tokio", - "tower", - "tower-http", - "tower-service", - "url", - "wasm-bindgen", - "wasm-bindgen-futures", - "web-sys", -] - -[[package]] -name = "reqwest" -version = "0.13.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "16a1cfa75cc186dd73d5818e510e042e40927bccc9c236b061cea97e1eb08029" -dependencies = [ - "base64 0.23.1", - "bytes", - "futures-core", - "futures-util", - "h2", - "http", - "http-body", - "http-body-util", - "hyper", - "hyper-rustls", - "hyper-util", - "js-sys", - "log", - "percent-encoding", - "pin-project-lite", - "quinn", - "rustls", - "rustls-pki-types", - "rustls-platform-verifier", - "sync_wrapper", - "tokio", - "tokio-rustls", - "tokio-util", - "tower", - "tower-http", - "tower-service", - "url", - "wasm-bindgen", - "wasm-bindgen-futures", - "wasm-streams", - "web-sys", -] - -[[package]] -name = "ring" -version = "0.17.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7" -dependencies = [ - "cc", - "cfg-if", - "getrandom 0.2.17", - "libc", - "untrusted", - "windows-sys 0.52.0", -] - -[[package]] -name = "rusqlite" -version = "0.34.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "37e34486da88d8e051c7c0e23c3f15fd806ea8546260aa2fec247e97242ec143" -dependencies = [ - "bitflags", - "fallible-iterator", - "fallible-streaming-iterator", - "hashlink", - "libsqlite3-sys", - "smallvec", -] - -[[package]] -name = "rustc-hash" -version = "2.1.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d" - -[[package]] -name = "rustc_version" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cfcb3a22ef46e85b45de6ee7e79d063319ebb6594faafcf1c225ea92ab6e9b92" -dependencies = [ - "semver", -] - -[[package]] -name = "rustix" -version = "1.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "891efababe418670775f199f0d233d84843c227a0949a883ce15b37c78d6629d" -dependencies = [ - "bitflags", - "errno", - "libc", - "linux-raw-sys", - "windows-sys 0.61.2", -] - -[[package]] -name = "rustls" -version = "0.23.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d41d731c7d2f962d1ccc364cec258de3c0e93b38c2fb3ba97ac74513048d634" -dependencies = [ - "aws-lc-rs", - "once_cell", - "rustls-pki-types", - "rustls-webpki", - "subtle", - "zeroize", -] - -[[package]] -name = "rustls-native-certs" -version = "0.8.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dab5152771c58876a2146916e53e35057e1a4dfa2b9df0f0305b07f611fdea4d" -dependencies = [ - "openssl-probe", - "rustls-pki-types", - "schannel", - "security-framework", -] - -[[package]] -name = "rustls-pki-types" -version = "1.15.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" -dependencies = [ - "web-time", - "zeroize", -] - -[[package]] -name = "rustls-platform-verifier" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1167586491e2b18b8bfbb293e8180ec17c201c4f076d7cb3070ca964e7598f98" -dependencies = [ - "core-foundation", - "core-foundation-sys", - "jni", - "log", - "once_cell", - "rustls", - "rustls-native-certs", - "rustls-platform-verifier-android", - "rustls-webpki", - "security-framework", - "security-framework-sys", - "webpki-root-certs", - "windows-sys 0.61.2", -] - -[[package]] -name = "rustls-platform-verifier-android" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eec689c0bc40ff2458a5977b6619cb718087084a18e02a131c599b62d05e1a5f" - -[[package]] -name = "rustls-webpki" -version = "0.103.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" -dependencies = [ - "aws-lc-rs", - "ring", - "rustls-pki-types", - "untrusted", -] - -[[package]] -name = "rustversion" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" - -[[package]] -name = "ryu" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" - -[[package]] -name = "same-file" -version = "1.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502" -dependencies = [ - "winapi-util", -] - -[[package]] -name = "schannel" -version = "0.1.29" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "91c1b7e4904c873ef0710c1f407dde2e6287de2bebc1bbbf7d430bb7cbffd939" -dependencies = [ - "windows-sys 0.61.2", -] - -[[package]] -name = "schemars" -version = "0.8.22" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3fbf2ae1b8bc8e02df939598064d22402220cd5bbcca1c76f7d6a310974d5615" -dependencies = [ - "dyn-clone", - "schemars_derive", - "serde", - "serde_json", -] - -[[package]] -name = "schemars_derive" -version = "0.8.22" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32e265784ad618884abaea0600a9adf15393368d840e0222d101a072f3f7534d" -dependencies = [ - "proc-macro2", - "quote", - "serde_derive_internals", - "syn 2.0.119", -] - -[[package]] -name = "scopeguard" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" - -[[package]] -name = "security-framework" -version = "3.7.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" -dependencies = [ - "bitflags", - "core-foundation", - "core-foundation-sys", - "libc", - "security-framework-sys", -] - -[[package]] -name = "security-framework-sys" -version = "2.17.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ce2691df843ecc5d231c0b14ece2acc3efb62c0a398c7e1d875f3983ce020e3" -dependencies = [ - "core-foundation-sys", - "libc", -] - -[[package]] -name = "semver" -version = "1.0.28" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" - -[[package]] -name = "serde" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" -dependencies = [ - "serde_core", - "serde_derive", -] - -[[package]] -name = "serde_core" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" -dependencies = [ - "serde_derive", -] - -[[package]] -name = "serde_derive" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "serde_derive_internals" -version = "0.29.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "18d26a20a969b9e3fdf2fc2d9f21eda6c40e2de84c9408bb5d3b05d499aae711" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "serde_json" -version = "1.0.151" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" -dependencies = [ - "itoa", - "memchr", - "serde", - "serde_core", - "zmij", -] - -[[package]] -name = "serde_urlencoded" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3491c14715ca2294c4d6a88f15e84739788c1d030eed8c110436aafdaa2f3fd" -dependencies = [ - "form_urlencoded", - "itoa", - "ryu", - "serde", -] - -[[package]] -name = "shlex" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" - -[[package]] -name = "simd_cesu8" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "11031e251abf8611c80f460e19dbdeb54a66db918e49c65a7065b46ac7aec520" -dependencies = [ - "rustc_version", - "simdutf8", -] - -[[package]] -name = "simdutf8" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" - -[[package]] -name = "slab" -version = "0.4.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" - -[[package]] -name = "smallvec" -version = "1.16.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba467056f1b547ed52077911161fc86985becbc60e8e1857c8a144dab0def891" - -[[package]] -name = "socket2" -version = "0.6.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" -dependencies = [ - "libc", - "windows-sys 0.61.2", -] - -[[package]] -name = "spin" -version = "0.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" - -[[package]] -name = "stable_deref_trait" -version = "1.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" - -[[package]] -name = "subtle" -version = "2.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" - -[[package]] -name = "syn" -version = "2.0.119" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "syn" -version = "3.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "sync_wrapper" -version = "1.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0bf256ce5efdfa370213c1dabab5935a12e49f2c58d15e9eac2870d3b4f27263" -dependencies = [ - "futures-core", -] - -[[package]] -name = "synstructure" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "901704edd0dfe137f1987838ee4f259e4e063c31371bdb423f7ae38ec6f77f02" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tempfile" -version = "3.27.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" -dependencies = [ - "fastrand", - "getrandom 0.4.3", - "once_cell", - "rustix", - "windows-sys 0.61.2", -] - -[[package]] -name = "thiserror" -version = "2.0.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09e52cb86a36cede5cb101bf8908837b3e4c6e5e59fe7fd85c23fb56200d189e" -dependencies = [ - "thiserror-impl", -] - -[[package]] -name = "thiserror-impl" -version = "2.0.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fe5197923287db20a58125f0bc85c062f7f2c892de97b18c356f9efb14b28524" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tinystr" -version = "0.8.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" -dependencies = [ - "displaydoc", - "zerovec", -] - -[[package]] -name = "tinyvec" -version = "1.13.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fd3ca314f692efd6c868f8408f53fe444634a845f96c028b97d35f6a1f79f0ee" - -[[package]] -name = "tokio" -version = "1.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" -dependencies = [ - "bytes", - "libc", - "mio", - "pin-project-lite", - "socket2", - "tokio-macros", - "windows-sys 0.61.2", -] - -[[package]] -name = "tokio-macros" -version = "2.7.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tokio-rustls" -version = "0.26.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b0c85f2c3ef0b1cd58b36682f4b17aaa995f0e5db534d85692b4903abce21f67" -dependencies = [ - "rustls", - "tokio", -] - -[[package]] -name = "tokio-util" -version = "0.7.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" -dependencies = [ - "bytes", - "futures-core", - "futures-sink", - "futures-util", - "libc", - "pin-project-lite", - "tokio", -] - -[[package]] -name = "tower" -version = "0.5.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4" -dependencies = [ - "futures-core", - "futures-util", - "pin-project-lite", - "sync_wrapper", - "tokio", - "tower-layer", - "tower-service", -] - -[[package]] -name = "tower-http" -version = "0.6.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840" -dependencies = [ - "bitflags", - "bytes", - "futures-util", - "http", - "http-body", - "pin-project-lite", - "tower", - "tower-layer", - "tower-service", - "url", -] - -[[package]] -name = "tower-layer" -version = "0.3.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "121c2a6cda46980bb0fcd1647ffaf6cd3fc79a013de288782836f6df9c48780e" - -[[package]] -name = "tower-service" -version = "0.3.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8df9b6e13f2d32c91b9bd719c00d1958837bc7dec474d94952798cc8e69eeec3" - -[[package]] -name = "tracing" -version = "0.1.44" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" -dependencies = [ - "pin-project-lite", - "tracing-attributes", - "tracing-core", -] - -[[package]] -name = "tracing-attributes" -version = "0.1.31" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "tracing-core" -version = "0.1.36" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" -dependencies = [ - "once_cell", -] - -[[package]] -name = "try-lock" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" - -[[package]] -name = "twox-hash" -version = "2.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" - -[[package]] -name = "typenum" -version = "1.20.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" - -[[package]] -name = "unicode-ident" -version = "1.0.26" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" - -[[package]] -name = "untrusted" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" - -[[package]] -name = "url" -version = "2.5.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff67a8a4397373c3ef660812acab3268222035010ab8680ec4215f38ba3d0eed" -dependencies = [ - "form_urlencoded", - "idna", - "percent-encoding", - "serde", -] - -[[package]] -name = "utf8_iter" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" - -[[package]] -name = "vcpkg" -version = "0.2.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" - -[[package]] -name = "version_check" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" - -[[package]] -name = "walkdir" -version = "2.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29790946404f91d9c5d06f9874efddea1dc06c5efe94541a7d6863108e3a5e4b" -dependencies = [ - "same-file", - "winapi-util", -] - -[[package]] -name = "want" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bfa7760aed19e106de2c7c0b581b509f2f25d3dacaf737cb82ac61bc6d760b0e" -dependencies = [ - "try-lock", -] - -[[package]] -name = "wasi" -version = "0.11.1+wasi-snapshot-preview1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" - -[[package]] -name = "wasip2" -version = "1.0.4+wasi-0.2.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" -dependencies = [ - "wit-bindgen", -] - -[[package]] -name = "wasm-bindgen" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "aecb87a33d3b0c5e3b7aa46336eaf486cffafbd281b195e4c8b80d50df2351bf" -dependencies = [ - "cfg-if", - "once_cell", - "rustversion", - "wasm-bindgen-macro", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-futures" -version = "0.4.78" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ef4c5d3d2cdf5c54f4231181768f5510842e350db025faf1f7163b1030ed928" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "wasm-bindgen-macro" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a690d511e3c1a8b3a55e33511e3c2c00c78415cd23650f32b808627f5696b9ed" -dependencies = [ - "quote", - "wasm-bindgen-macro-support", -] - -[[package]] -name = "wasm-bindgen-macro-support" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "411e4887f0071ef2d2164a9d5fdf2d20efbef78fccd3a78b0c10a1dc5295e48a" -dependencies = [ - "bumpalo", - "proc-macro2", - "quote", - "syn 3.0.6", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-shared" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "81941cd78d0c92026c33e5e01312845a4cb1e9af3407f9134b100dd03144103e" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "wasm-streams" -version = "0.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d1ec4f6517c9e11ae630e200b2b65d193279042e28edd4a2cda233e46670bbb" -dependencies = [ - "futures-util", - "js-sys", - "wasm-bindgen", - "wasm-bindgen-futures", - "web-sys", -] - -[[package]] -name = "web-sys" -version = "0.3.105" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9fbddc4a036f00ec4f18c83445bd3115cb306a91da554919a099d9222fe4a7f8" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "web-time" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a6580f308b1fad9207618087a65c04e7a10bc77e02c8e84e9b00dd4b12fa0bb" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "webpki-root-certs" -version = "1.0.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b96554aa2acc8ccdb7e1c9a58a7a68dd5d13bccc69cd124cb09406db612a1c9b" -dependencies = [ - "rustls-pki-types", -] - -[[package]] -name = "winapi-util" -version = "0.1.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22" -dependencies = [ - "windows-sys 0.61.2", -] - -[[package]] -name = "windows-core" -version = "0.62.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" -dependencies = [ - "windows-implement", - "windows-interface", - "windows-link", - "windows-result", - "windows-strings", -] - -[[package]] -name = "windows-implement" -version = "0.60.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "windows-interface" -version = "0.59.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "windows-link" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" - -[[package]] -name = "windows-result" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-strings" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-sys" -version = "0.52.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" -dependencies = [ - "windows-targets", -] - -[[package]] -name = "windows-sys" -version = "0.61.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-targets" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" -dependencies = [ - "windows_aarch64_gnullvm", - "windows_aarch64_msvc", - "windows_i686_gnu", - "windows_i686_gnullvm", - "windows_i686_msvc", - "windows_x86_64_gnu", - "windows_x86_64_gnullvm", - "windows_x86_64_msvc", -] - -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" - -[[package]] -name = "windows_aarch64_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" - -[[package]] -name = "windows_i686_gnu" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" - -[[package]] -name = "windows_i686_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" - -[[package]] -name = "windows_i686_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" - -[[package]] -name = "windows_x86_64_gnu" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" - -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" - -[[package]] -name = "windows_x86_64_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" - -[[package]] -name = "wit-bindgen" -version = "0.57.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" - -[[package]] -name = "writeable" -version = "0.6.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" - -[[package]] -name = "yoke" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" -dependencies = [ - "stable_deref_trait", - "yoke-derive", - "zerofrom", -] - -[[package]] -name = "yoke-derive" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "33811428bee40dbceb6d545e95754741d17a6aef9a4849f0fd62e2ba4f412a78" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", - "synstructure", -] - -[[package]] -name = "zerocopy" -version = "0.8.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c17e8fafad82b542ff3717217ecdc736231b59e387768c9630123b4ce4d2db44" -dependencies = [ - "zerocopy-derive", -] - -[[package]] -name = "zerocopy-derive" -version = "0.8.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "595f56e044df4f46a0c9a626f65c3d99eb8488f7e8a8baa12dd76326d9710bf2" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "zerofrom" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" -dependencies = [ - "zerofrom-derive", -] - -[[package]] -name = "zerofrom-derive" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f75b4683f6c7f45248d4d64056a24298c6281e0993356d7d1b4a1a962ef10d4a" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", - "synstructure", -] - -[[package]] -name = "zeroize" -version = "1.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" - -[[package]] -name = "zerotrie" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" -dependencies = [ - "displaydoc", - "yoke", - "zerofrom", -] - -[[package]] -name = "zerovec" -version = "0.11.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" -dependencies = [ - "yoke", - "zerofrom", - "zerovec-derive", -] - -[[package]] -name = "zerovec-derive" -version = "0.11.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "zmij" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/crates/crab-ltx/perf/replica-cost/Cargo.toml b/crates/crab-ltx/perf/replica-cost/Cargo.toml deleted file mode 100644 index 0af01a433..000000000 --- a/crates/crab-ltx/perf/replica-cost/Cargo.toml +++ /dev/null @@ -1,20 +0,0 @@ -# Measures the immutable publication cost of one Cell command. -# -# Its own workspace, like the other runners here, so the crate under test is -# built with the `replica` feature without touching the main workspace target. -[package] -name = "crab-ltx-replica-cost" -version = "0.1.0" -edition = "2021" -publish = false - -[workspace] - -[dependencies] -crab-ltx = { path = "../../", features = ["replica"] } -crab-storage = { path = "../../../crab-storage" } -object_store = "0.14" -serde = { version = "1", features = ["derive"] } -serde_json = "1" -tempfile = "3" -tokio = { version = "1", features = ["macros", "rt-multi-thread", "sync", "time"] } diff --git a/crates/crab-ltx/perf/replica-cost/src/activation.rs b/crates/crab-ltx/perf/replica-cost/src/activation.rs deleted file mode 100644 index d40371bea..000000000 --- a/crates/crab-ltx/perf/replica-cost/src/activation.rs +++ /dev/null @@ -1,72 +0,0 @@ -use crab_ltx::{CellReplica, Db, RootRef}; -use serde::Serialize; -use std::path::Path; -use std::str::FromStr; - -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Serialize)] -#[serde(rename_all = "snake_case")] -pub(super) enum Activation { - #[default] - Fresh, - Sparse, - Hydrated, - Resumed, -} - -impl FromStr for Activation { - type Err = &'static str; - - fn from_str(value: &str) -> Result { - match value { - "fresh" => Ok(Self::Fresh), - "sparse" => Ok(Self::Sparse), - "hydrated" => Ok(Self::Hydrated), - "resumed" => Ok(Self::Resumed), - _ => Err("activation must be fresh, sparse, hydrated, or resumed"), - } - } -} - -impl Activation { - pub(super) async fn open( - self, - replica: &CellReplica, - root: &RootRef, - database: Db, - directory: &Path, - ) -> Result> { - if self == Self::Fresh { - return Ok(database); - } - database.close()?; - let active = directory.join("sparse.sqlite"); - let writable = replica - .open_root(root) - .await? - .paged() - .prepare_writable(&active) - .await?; - let mut database = - tokio::task::spawn_blocking(move || writable.open_writable(&active)).await??; - if self == Self::Sparse { - return Ok(database); - } - while !database.hydration()?.is_some_and(|state| state.complete()) { - let batch = database - .prepare_hydration(64)? - .ok_or("sparse activation lost hydration state")? - .fetch() - .await?; - database.install_hydration(batch)?; - } - if self == Self::Hydrated { - return Ok(database); - } - // This fixture owns the exact root and performs no intervening writes. - // Resume still verifies every local page against its saved checksum. - database.persist_continuation()?; - let source = database.path().to_owned(); - database.close()?; - Ok(replica.open_resumed(&source, &directory.join("resumed.sqlite"))?) - } -} diff --git a/crates/crab-ltx/perf/replica-cost/src/filesystem.rs b/crates/crab-ltx/perf/replica-cost/src/filesystem.rs deleted file mode 100644 index 5b9a9b9d6..000000000 --- a/crates/crab-ltx/perf/replica-cost/src/filesystem.rs +++ /dev/null @@ -1,121 +0,0 @@ -use crab_ltx::environment::{DirectFileSystem, FileIo, FileSystem}; -use std::io; -use std::path::{Path, PathBuf}; -use std::sync::atomic::{AtomicU64, Ordering}; -use std::time::Instant; - -#[derive(Default)] -pub(super) struct ChecksumSyncs { - calls: AtomicU64, - nanos: AtomicU64, -} - -impl ChecksumSyncs { - pub(super) fn take(&self) -> (u64, u64) { - ( - self.calls.swap(0, Ordering::Relaxed), - self.nanos.swap(0, Ordering::Relaxed) / 1_000, - ) - } -} - -pub(super) struct MeasuredFileSystem { - pub(super) syncs: std::sync::Arc, -} - -struct MeasuredFile { - inner: Box, - checksum: bool, - syncs: std::sync::Arc, -} - -impl MeasuredFileSystem { - fn wrap(&self, path: &Path, file: Box) -> Box { - Box::new(MeasuredFile { - inner: file, - checksum: path - .as_os_str() - .to_string_lossy() - .ends_with(".crab-ltx-checksums"), - syncs: self.syncs.clone(), - }) - } -} - -impl FileIo for MeasuredFile { - fn write_all(&mut self, bytes: &[u8]) -> io::Result<()> { - self.inner.write_all(bytes) - } - fn write_all_at(&mut self, offset: u64, bytes: &[u8]) -> io::Result<()> { - self.inner.write_all_at(offset, bytes) - } - fn read_exact_at(&mut self, offset: u64, len: usize) -> io::Result> { - self.inner.read_exact_at(offset, len) - } - fn sync_all(&mut self) -> io::Result<()> { - let started = Instant::now(); - let result = self.inner.sync_all(); - if self.checksum { - self.syncs.calls.fetch_add(1, Ordering::Relaxed); - self.syncs.nanos.fetch_add( - started.elapsed().as_nanos().min(u128::from(u64::MAX)) as u64, - Ordering::Relaxed, - ); - } - result - } - fn file_len(&self) -> io::Result { - self.inner.file_len() - } - fn set_len(&mut self, len: u64) -> io::Result<()> { - self.inner.set_len(len) - } -} - -impl FileSystem for MeasuredFileSystem { - fn open(&self, path: &Path) -> io::Result> { - Ok(self.wrap(path, DirectFileSystem.open(path)?)) - } - fn open_rw(&self, path: &Path) -> io::Result> { - Ok(self.wrap(path, DirectFileSystem.open_rw(path)?)) - } - fn create(&self, path: &Path) -> io::Result> { - Ok(self.wrap(path, DirectFileSystem.create(path)?)) - } - fn file_len(&self, path: &Path) -> io::Result { - DirectFileSystem.file_len(path) - } - fn create_dir_all(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.create_dir_all(path) - } - fn rename(&self, from: &Path, to: &Path) -> io::Result<()> { - DirectFileSystem.rename(from, to) - } - fn rename_uncommitted(&self, from: &Path, to: &Path) -> io::Result<()> { - DirectFileSystem.rename_uncommitted(from, to) - } - fn remove_file(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.remove_file(path) - } - fn canonicalize(&self, path: &Path) -> io::Result { - DirectFileSystem.canonicalize(path) - } - fn exists(&self, path: &Path) -> io::Result { - DirectFileSystem.exists(path) - } - fn create_dir(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.create_dir(path) - } - fn sync_parent(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.sync_parent(path) - } - fn persist_new(&self, path: &Path, bytes: &[u8]) -> io::Result<()> { - DirectFileSystem.persist_new(path, bytes) - } - fn persist_file_new(&self, source: &Path, destination: &Path) -> io::Result<()> { - DirectFileSystem.persist_file_new(source, destination) - } - fn cleanup_private_temporaries(&self, root: &Path) -> io::Result<()> { - DirectFileSystem.cleanup_private_temporaries(root) - } -} diff --git a/crates/crab-ltx/perf/replica-cost/src/main.rs b/crates/crab-ltx/perf/replica-cost/src/main.rs deleted file mode 100644 index 9598d3390..000000000 --- a/crates/crab-ltx/perf/replica-cost/src/main.rs +++ /dev/null @@ -1,599 +0,0 @@ -//! Reports the immutable publication cost of one Cell command. -//! -//! Each command commits one SQLite transaction, captures one LTX cut, and -//! publishes one immutable Cell root through `CellReplica`. The runner reports -//! objects, bytes, and wall time per command so the runtime's per-command -//! object-store budget has a measured baseline. -//! -//! The default store is in-memory and synchronous, so its latency is a floor. -//! `--endpoint` (with `--bucket`, `--access-key`, and `--secret-key`) points the -//! same workload at an S3-compatible provider such as RustFS. Object and byte -//! counts are provider independent; only latency changes. - -mod activation; -mod filesystem; -mod storage; - -use activation::Activation; -use crab_ltx::{CellReplica, CellStorageLayout, Host, Limits}; -use crab_storage::{ObjectStoreCredentials, Store}; -use object_store::{memory::InMemory, path::Path, ObjectStore}; -use serde::Serialize; -use std::collections::BTreeMap; -use std::sync::Arc; -use std::time::{Duration, Instant}; - -#[derive(Debug, Clone)] -struct Config { - payload_bytes: usize, - commands: usize, - warmup: usize, - activation: Activation, - random_payload: bool, - churn_rows: Option, - max_capture_bytes: Option, - endpoint: Option, - bucket: String, - access_key: String, - secret_key: String, -} - -#[derive(Clone, Debug, Serialize)] -struct Sample { - command: usize, - mutation: &'static str, - row: usize, - live_rows: usize, - objects: u64, - bytes: u64, - commit_us: u64, - capture_us: u64, - capture_preparation_us: u64, - capture_schema_check_us: u64, - capture_wal_existence_us: u64, - capture_position_resolution_us: u64, - capture_wal_read_us: u64, - capture_page_collection_us: u64, - capture_verification_us: u64, - capture_encode_us: u64, - capture_local_write_us: u64, - capture_fsync_us: u64, - capture_parent_sync_us: u64, - capture_checkpoint_us: u64, - checkpoint_runs: u32, - checkpoint_busy: u32, - checkpoint_frames: u64, - checkpoint_backfilled: u64, - checksum_sync_calls: u64, - checksum_sync_us: u64, - wal_read_bytes: u64, - wal_image_bytes: u64, - wal_snapshot_reads: u32, - wal_full_reads: u32, - elapsed_us: u64, - prune_us: u64, - captured_bytes: u64, - commit_io: Vec, - capture_io: Vec, - preparation_io: Vec, - prune_io: Vec, -} - -#[derive(Debug, Serialize)] -struct ActivationSample { - mode: Activation, - elapsed_us: u64, - io: Vec, - initial_database_pages: u32, - initial_page_size: u32, - first_command: Sample, -} - -#[derive(Debug, Serialize)] -struct Report { - implementation: &'static str, - store: &'static str, - object_prefix: String, - activation: ActivationSample, - churn_rows: Option, - sqlite_version: &'static str, - payload_bytes: usize, - payload_pattern: &'static str, - measured_commands: usize, - restored_rows: usize, - bootstrap_capture_us: u64, - bootstrap_parent_sync_us: u64, - objects_per_command: u64, - bytes_per_command: u64, - objects_p95: u64, - bytes_p95: u64, - elapsed_us_p50: u64, - elapsed_us_p95: u64, - elapsed_us_p99: u64, - elapsed_us_max: u64, - commit_us_p50: u64, - commit_us_p95: u64, - capture_us_p50: u64, - capture_us_p95: u64, - checksum_sync_calls: u64, - checksum_sync_us_p50: u64, - checksum_sync_us_p95: u64, - wal_read_bytes: u64, - wal_image_bytes_max: u64, - wal_snapshot_reads: u64, - wal_full_reads: u64, - samples: Vec, -} - -#[tokio::main(flavor = "multi_thread")] -async fn main() -> Result<(), Box> { - let config = Config::from_args(std::env::args().skip(1))?; - let directory = tempfile::TempDir::new()?; - let database_path = directory.path().join("cell.sqlite"); - let mut limits = Limits::default(); - if let Some(max_capture_bytes) = config.max_capture_bytes { - limits.max_capture_bytes = max_capture_bytes; - } - let (store, store_label, prefix) = open_store(&config)?; - let backend = Arc::new(storage::StorageCosts::default()); - let store = store.with_storage_observer(backend.clone()); - let syncs = Arc::new(filesystem::ChecksumSyncs::default()); - let host = Host::default().with_filesystem(Arc::new(filesystem::MeasuredFileSystem { - syncs: syncs.clone(), - })); - let replica = CellReplica::new( - CellStorageLayout::new(store, Path::from(prefix.as_str()), [7; 16]), - [5; 32], - [6; 16], - limits, - )? - .with_host(host); - let mut database = replica.open_new(&database_path)?; - - let mut expected = BTreeMap::new(); - database.transaction(|transaction| { - transaction - .execute_batch("CREATE TABLE payload(id INTEGER PRIMARY KEY, value BLOB NOT NULL)")?; - for seed in 0..config.churn_rows.unwrap_or(0) { - transaction.execute( - "INSERT INTO payload(id, value) VALUES(?1, ?2)", - crab_ltx::rusqlite::params![ - seed + 1, - payload(seed, config.payload_bytes, config.random_payload) - ], - )?; - expected.insert(seed + 1, seed); - } - Ok(()) - })?; - let bootstrap_started = Instant::now(); - let first = database.capture()?; - let bootstrap_capture_us = bootstrap_started.elapsed().as_micros() as u64; - let bootstrap_parent_sync_us = first.timing.parent_sync_nanos / 1_000; - let initial = first - .segments - .last() - .ok_or("bootstrap capture missing")? - .info(); - let initial_database_pages = initial.database_pages; - let initial_page_size = initial.page_size; - let initial_root = replica.prepare(None, &first, 1, 1).await?.root(); - let mut root = Some(initial_root); - database.prune_captured(&first)?; - let _bootstrap = replica.take_publication_cost(); - - let _bootstrap_io = backend.take(); - let activation_started = Instant::now(); - let mut database = config - .activation - .open(&replica, &initial_root, database, directory.path()) - .await?; - let activation_us = activation_started.elapsed().as_micros() as u64; - let activation_io = backend.take(); - if database.position() != first.position { - return Err("activation changed the selected root position".into()); - } - let _activation_syncs = syncs.take(); - - let mut first_command = None; - let mut samples = Vec::with_capacity(config.commands); - for command in 0..config.commands { - let (mutation, row, sql) = match config.churn_rows { - Some(rows) => { - // Start at the tail so activation's header read-ahead cannot - // prefetch the first mutation's target in a large working set. - let row = rows - command / 3 % rows; - match command % 3 { - 0 => ("update", row, "UPDATE payload SET value = ?2 WHERE id = ?1"), - 1 => ("delete", row, "DELETE FROM payload WHERE id = ?1"), - _ => ( - "reinsert", - row, - "INSERT INTO payload(id, value) VALUES(?1, ?2)", - ), - } - } - None => ( - "insert", - command + 1, - "INSERT INTO payload(id, value) VALUES(?1, ?2)", - ), - }; - let seed = command - .checked_add(config.churn_rows.unwrap_or(0)) - .ok_or("payload seed overflow")?; - let payload = payload(seed, config.payload_bytes, config.random_payload); - let commit_started = Instant::now(); - let changed = database.transaction(|transaction| { - if mutation == "delete" { - transaction.execute(sql, [row]) - } else { - transaction.execute(sql, crab_ltx::rusqlite::params![row, payload]) - } - })?; - if changed != 1 { - return Err("workload must change exactly one row".into()); - } - let commit_us = commit_started.elapsed().as_micros() as u64; - let commit_io = backend.take(); - if mutation == "delete" { - expected.remove(&row); - } else { - expected.insert(row, seed); - } - let capture_started = Instant::now(); - // Compare activation histories under the runtime's same capture and - // cleanup boundaries; immediate capture would add another barrier. - let batch = database.capture_deferred()?; - let capture_us = capture_started.elapsed().as_micros() as u64; - let (checksum_sync_calls, checksum_sync_us) = syncs.take(); - let capture_io = backend.take(); - let started = Instant::now(); - let prepared = replica - .prepare(root.as_ref(), &batch, command as u64 + 2, 1) - .await?; - let elapsed = started.elapsed(); - let preparation_io = backend.take(); - let cost = replica.take_publication_cost(); - root = Some(prepared.root()); - let started = Instant::now(); - database.prune_captured(&batch)?; - let prune_us = started.elapsed().as_micros() as u64; - let prune_io = backend.take(); - let sample = Sample { - command, - mutation, - row, - live_rows: expected.len(), - objects: cost.objects, - bytes: cost.bytes, - commit_us, - capture_us, - capture_preparation_us: batch.timing.preparation_nanos / 1_000, - capture_schema_check_us: batch.timing.schema_check_nanos / 1_000, - capture_wal_existence_us: batch.timing.wal_existence_nanos / 1_000, - capture_position_resolution_us: batch.timing.position_resolution_nanos / 1_000, - capture_wal_read_us: batch.timing.wal_read_nanos / 1_000, - capture_page_collection_us: batch.timing.page_collection_nanos / 1_000, - capture_verification_us: batch.timing.verification_nanos / 1_000, - capture_encode_us: batch.timing.encode_nanos / 1_000, - capture_local_write_us: batch.timing.local_write_nanos / 1_000, - capture_fsync_us: batch.timing.fsync_nanos / 1_000, - capture_parent_sync_us: batch.timing.parent_sync_nanos / 1_000, - capture_checkpoint_us: batch.timing.checkpoint_nanos / 1_000, - checkpoint_runs: batch.timing.checkpoint_runs, - checkpoint_busy: batch.timing.checkpoint_busy, - checkpoint_frames: batch.timing.checkpoint_frames, - checkpoint_backfilled: batch.timing.checkpoint_backfilled, - checksum_sync_calls, - checksum_sync_us, - wal_read_bytes: batch.timing.wal_read_bytes, - wal_image_bytes: batch.timing.wal_image_bytes, - wal_snapshot_reads: batch.timing.wal_snapshot_reads, - wal_full_reads: batch.timing.wal_full_reads, - elapsed_us: elapsed.as_micros() as u64, - prune_us, - captured_bytes: batch - .segments - .iter() - .map(|segment| segment.info().size_bytes) - .sum(), - commit_io, - capture_io, - preparation_io, - prune_io, - }; - if command == 0 { - first_command = Some(sample.clone()); - } - if command >= config.warmup { - samples.push(sample); - } - } - database.close()?; - - // Validate the final selected root independently of the writer and its - // pruned local cuts. This work stays outside all measured phases. - let restored = directory.path().join("restored.sqlite"); - replica - .open_root(root.as_ref().ok_or("final root missing")?) - .await? - .restore(&restored) - .await?; - let connection = crab_ltx::rusqlite::Connection::open_with_flags( - restored, - crab_ltx::rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY, - )?; - let mut statement = connection.prepare("SELECT id, value FROM payload ORDER BY id")?; - let mut rows = statement.query([])?; - let mut restored_rows = 0; - while let Some(row) = rows.next()? { - let id: usize = row.get(0)?; - let value: Vec = row.get(1)?; - let seed = expected - .remove(&id) - .ok_or("restored root resurrected a deleted or unknown row")?; - if value != payload(seed, config.payload_bytes, config.random_payload) { - return Err("restored root changed a committed payload".into()); - } - restored_rows += 1; - } - if !expected.is_empty() { - return Err("restored root lost committed payloads".into()); - } - - if samples.is_empty() { - return Err("at least one measured command is required".into()); - } - let activation = ActivationSample { - mode: config.activation, - elapsed_us: activation_us, - io: activation_io, - initial_database_pages, - initial_page_size, - first_command: first_command.ok_or("first command missing")?, - }; - let bootstrap = (bootstrap_capture_us, bootstrap_parent_sync_us); - let report = summarize( - config, - store_label, - prefix, - &samples, - bootstrap, - restored_rows, - activation, - ); - println!("{}", serde_json::to_string_pretty(&report)?); - Ok(()) -} - -/// Opens the configured store and returns it with its label and prefix. -fn open_store( - config: &Config, -) -> Result<(Store, &'static str, String), Box> { - let run = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH)? - .as_millis(); - let prefix = format!("cost/{}-{run}", std::process::id()); - let Some(endpoint) = config.endpoint.as_deref() else { - let backend: Arc = Arc::new(InMemory::new()); - return Ok((Store::new(backend), "object_store in-memory", prefix)); - }; - let allow_http = match endpoint.split_once("://") { - Some(("http", _)) => true, - Some(("https", _)) => false, - _ => return Err("endpoint must use http or https".into()), - }; - let store = crab_storage::build_explicit_store( - &config.bucket, - ObjectStoreCredentials::Aws { - access_key_id: config.access_key.clone(), - secret_access_key: config.secret_key.clone(), - session_token: None, - region: "us-east-1".into(), - }, - Some(endpoint), - allow_http, - )?; - // The label is bounded and static so the report stays comparable. - let label = if allow_http { - "S3-compatible endpoint (http)" - } else { - "S3-compatible endpoint (https)" - }; - Ok((store, label, prefix)) -} - -fn payload(command: usize, bytes: usize, random: bool) -> Vec { - if !random { - // Retain the periodic fixture so historical cost rows are reproducible. - return (0..bytes) - .map(|index| ((command * 131 + index) % 251) as u8) - .collect(); - } - let mut state = (command as u64).wrapping_add(1); - let mut payload = vec![0; bytes]; - for chunk in payload.chunks_mut(8) { - // A fixed command seed makes this high-entropy workload repeatable. - // This generator is only test data, never a security primitive. - state ^= state << 13; - state ^= state >> 7; - state ^= state << 17; - chunk.copy_from_slice(&state.to_le_bytes()[..chunk.len()]); - } - payload -} - -fn percentile(values: &[u64], percent: usize) -> u64 { - let mut sorted = values.to_vec(); - sorted.sort_unstable(); - sorted[(sorted.len() * percent).div_ceil(100).saturating_sub(1)] -} - -fn summarize( - config: Config, - store: &'static str, - object_prefix: String, - samples: &[Sample], - bootstrap: (u64, u64), - restored_rows: usize, - activation: ActivationSample, -) -> Report { - let objects: Vec = samples.iter().map(|sample| sample.objects).collect(); - let bytes: Vec = samples.iter().map(|sample| sample.bytes).collect(); - let elapsed: Vec = samples.iter().map(|sample| sample.elapsed_us).collect(); - let commits: Vec = samples.iter().map(|sample| sample.commit_us).collect(); - let captures: Vec = samples.iter().map(|sample| sample.capture_us).collect(); - let checksum_syncs: Vec = samples - .iter() - .map(|sample| sample.checksum_sync_us) - .collect(); - let total_objects: u64 = objects.iter().sum(); - let total_bytes: u64 = bytes.iter().sum(); - Report { - implementation: "crab-ltx CellReplica", - store, - object_prefix, - sqlite_version: crab_ltx::rusqlite::version(), - activation, - churn_rows: config.churn_rows, - payload_bytes: config.payload_bytes, - payload_pattern: if config.random_payload { - "xorshift64-command-seeded" - } else { - "periodic-251" - }, - measured_commands: samples.len(), - restored_rows, - bootstrap_capture_us: bootstrap.0, - bootstrap_parent_sync_us: bootstrap.1, - objects_per_command: total_objects / samples.len() as u64, - bytes_per_command: total_bytes / samples.len() as u64, - objects_p95: percentile(&objects, 95), - bytes_p95: percentile(&bytes, 95), - elapsed_us_p50: percentile(&elapsed, 50), - elapsed_us_p95: percentile(&elapsed, 95), - elapsed_us_p99: percentile(&elapsed, 99), - elapsed_us_max: *elapsed.iter().max().unwrap_or(&0), - commit_us_p50: percentile(&commits, 50), - commit_us_p95: percentile(&commits, 95), - capture_us_p50: percentile(&captures, 50), - capture_us_p95: percentile(&captures, 95), - checksum_sync_calls: samples - .iter() - .map(|sample| sample.checksum_sync_calls) - .sum(), - checksum_sync_us_p50: percentile(&checksum_syncs, 50), - checksum_sync_us_p95: percentile(&checksum_syncs, 95), - wal_read_bytes: samples.iter().map(|sample| sample.wal_read_bytes).sum(), - wal_image_bytes_max: samples - .iter() - .map(|sample| sample.wal_image_bytes) - .max() - .unwrap_or(0), - wal_snapshot_reads: samples - .iter() - .map(|sample| u64::from(sample.wal_snapshot_reads)) - .sum(), - wal_full_reads: samples - .iter() - .map(|sample| u64::from(sample.wal_full_reads)) - .sum(), - samples: samples.to_vec(), - } -} - -impl Config { - fn from_args( - args: impl IntoIterator, - ) -> Result> { - let args: Vec = args.into_iter().collect(); - let payload_bytes = option(&args, "--payload-bytes")?.unwrap_or(4096); - let commands = option(&args, "--commands")?.unwrap_or(64); - let warmup = option(&args, "--warmup")?.unwrap_or(4); - validate_options(&args)?; - let activation = value(&args, "--activation")? - .map(|value| value.parse()) - .transpose()? - .unwrap_or_default(); - let random_payload = args.iter().any(|arg| arg == "--random-payload"); - let churn_rows = option(&args, "--churn-rows")?; - if churn_rows.is_some_and(|rows| !(1..=100_000).contains(&rows)) { - return Err("churn-rows must be 1..100000".into()); - } - let max_capture_bytes = option(&args, "--max-capture-bytes")?.map(|bytes| bytes as u64); - if payload_bytes == 0 || commands == 0 || warmup >= commands { - return Err( - "payload-bytes and commands must be positive, and warmup < commands".into(), - ); - } - let endpoint = value(&args, "--endpoint")?; - let bucket = value(&args, "--bucket")?.unwrap_or_else(|| "crab-ltx-cost".into()); - let access_key = value(&args, "--access-key")?.unwrap_or_else(|| "crab-ltx-test".into()); - let secret_key = value(&args, "--secret-key")?.unwrap_or_else(|| "crab-ltx-test".into()); - Ok(Self { - payload_bytes, - commands, - warmup, - activation, - random_payload, - churn_rows, - max_capture_bytes, - endpoint, - bucket, - access_key, - secret_key, - }) - } -} - -fn validate_options(args: &[String]) -> Result<(), Box> { - let mut args = args.iter(); - while let Some(option) = args.next() { - match option.as_str() { - "--random-payload" => {} - "--activation" - | "--payload-bytes" - | "--commands" - | "--warmup" - | "--churn-rows" - | "--max-capture-bytes" - | "--endpoint" - | "--bucket" - | "--access-key" - | "--secret-key" => { - if args.next().is_none_or(|value| value.starts_with("--")) { - return Err(format!("{option} needs a value").into()); - } - } - _ => return Err("unknown option; select writer history with --activation".into()), - } - } - Ok(()) -} - -fn value(args: &[String], name: &str) -> Result, Box> { - let Some(position) = args.iter().position(|arg| arg == name) else { - return Ok(None); - }; - Ok(Some( - args.get(position + 1) - .ok_or_else(|| format!("{name} needs a value"))? - .clone(), - )) -} - -fn option(args: &[String], name: &str) -> Result, Box> { - let Some(position) = args.iter().position(|arg| arg == name) else { - return Ok(None); - }; - let value = args - .get(position + 1) - .ok_or_else(|| format!("{name} needs a value"))?; - Ok(Some(value.parse()?)) -} - -#[allow(dead_code)] -fn _duration_helper(duration: Duration) -> u64 { - duration.as_micros() as u64 -} diff --git a/crates/crab-ltx/perf/replica-cost/src/storage.rs b/crates/crab-ltx/perf/replica-cost/src/storage.rs deleted file mode 100644 index 3d371c75e..000000000 --- a/crates/crab-ltx/perf/replica-cost/src/storage.rs +++ /dev/null @@ -1,67 +0,0 @@ -use crab_storage::{StorageObservation, StorageObserver, StorageOperation, StorageOutcome}; -use serde::Serialize; -use std::sync::atomic::{AtomicU64, Ordering}; - -#[derive(Default)] -struct Counter { - calls: AtomicU64, - nanos: AtomicU64, - bytes_read: AtomicU64, - bytes_written: AtomicU64, -} - -#[derive(Default)] -pub(super) struct StorageCosts([[Counter; StorageOutcome::ALL.len()]; StorageOperation::ALL.len()]); - -#[derive(Clone, Debug, Serialize)] -pub(super) struct BackendCost { - operation: &'static str, - outcome: &'static str, - calls: u64, - elapsed_us: u64, - bytes_read: u64, - bytes_written: u64, -} - -impl StorageCosts { - // The runner takes a snapshot only after its serial preparation has ended. - pub(super) fn take(&self) -> Vec { - let mut costs = Vec::new(); - for operation in StorageOperation::ALL { - for outcome in StorageOutcome::ALL { - let counter = &self.0[operation.index()][outcome.index()]; - let calls = counter.calls.swap(0, Ordering::Relaxed); - if calls != 0 { - costs.push(BackendCost { - operation: operation.label(), - outcome: outcome.label(), - calls, - elapsed_us: counter.nanos.swap(0, Ordering::Relaxed) / 1_000, - bytes_read: counter.bytes_read.swap(0, Ordering::Relaxed), - bytes_written: counter.bytes_written.swap(0, Ordering::Relaxed), - }); - } - } - } - costs - } -} - -impl StorageObserver for StorageCosts { - fn started(&self, _: StorageOperation) {} - - fn finished(&self, observation: StorageObservation) { - let counter = &self.0[observation.operation.index()][observation.outcome.index()]; - counter.calls.fetch_add(1, Ordering::Relaxed); - counter.nanos.fetch_add( - observation.duration.as_nanos().min(u128::from(u64::MAX)) as u64, - Ordering::Relaxed, - ); - counter - .bytes_read - .fetch_add(observation.bytes_read, Ordering::Relaxed); - counter - .bytes_written - .fetch_add(observation.bytes_written, Ordering::Relaxed); - } -} diff --git a/crates/crab-ltx/perf/replica-cost/tests/workload.rs b/crates/crab-ltx/perf/replica-cost/tests/workload.rs deleted file mode 100644 index fdada0880..000000000 --- a/crates/crab-ltx/perf/replica-cost/tests/workload.rs +++ /dev/null @@ -1,72 +0,0 @@ -use serde_json::Value; -use std::process::Command; - -#[test] -fn writer_histories_preserve_first_mutation_and_restore_deleted_rows() { - for mode in ["fresh", "sparse", "hydrated", "resumed"] { - let output = Command::new(env!("CARGO_BIN_EXE_crab-ltx-replica-cost")) - .args([ - "--activation", - mode, - "--churn-rows", - "128", - "--random-payload", - "--payload-bytes", - "4096", - "--commands", - "2", - "--warmup", - "1", - ]) - .output() - .unwrap(); - assert!( - output.status.success(), - "{mode}: {}", - String::from_utf8_lossy(&output.stderr) - ); - let report: Value = serde_json::from_slice(&output.stdout).unwrap(); - let activation = &report["activation"]; - let first = &activation["first_command"]; - assert_eq!(activation["mode"], mode); - assert!(activation["initial_database_pages"].as_u64().unwrap() > 64); - assert_eq!(first["command"], 0); - assert_eq!(first["row"], 128); - assert_eq!(first["mutation"], "update"); - assert_eq!(report["samples"].as_array().unwrap().len(), 1); - assert_eq!(report["samples"][0]["command"], 1); - assert_eq!(report["samples"][0]["mutation"], "delete"); - assert_eq!(report["restored_rows"], 127); - let reads = first["commit_io"] - .as_array() - .unwrap() - .iter() - .chain(first["capture_io"].as_array().unwrap()) - .map(|cost| cost["bytes_read"].as_u64().unwrap()) - .sum::(); - if mode == "sparse" { - assert!( - reads > 0, - "tail mutation must fault pages absent at activation" - ); - } else { - assert_eq!(reads, 0, "{mode}: materialized mutation read the provider"); - } - } -} - -#[test] -fn unsupported_activation_options_cannot_silently_measure_a_fresh_writer() { - for args in [ - vec!["--sparse"], - vec!["--activation", "unknown"], - vec!["--activation"], - ] { - let output = Command::new(env!("CARGO_BIN_EXE_crab-ltx-replica-cost")) - .args(args) - .output() - .unwrap(); - assert!(!output.status.success()); - assert!(output.stdout.is_empty()); - } -} diff --git a/crates/crab-ltx/perf/run.sh b/crates/crab-ltx/perf/run.sh deleted file mode 100755 index f470cd5ec..000000000 --- a/crates/crab-ltx/perf/run.sh +++ /dev/null @@ -1,30 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -script_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd) -target_dir=${CARGO_TARGET_DIR:-${HOME}/Workspace/crabbuild-target/crab-ltx-perf} -transactions=${LTX_TRANSACTIONS:-128} -payload_bytes=${LTX_PAYLOAD_BYTES:-4096} -rounds=${LTX_ROUNDS:-5} -warmup=${LTX_WARMUP:-1} - -common_args=( - --transactions "$transactions" - --payload-bytes "$payload_bytes" - --rounds "$rounds" - --warmup "$warmup" -) - -crab_report=$(mktemp) -celld_report=$(mktemp) -trap 'rm -f "$crab_report" "$celld_report"' EXIT - -CARGO_TARGET_DIR="$target_dir" cargo run --quiet --release \ - --manifest-path "$script_dir/crab/Cargo.toml" -- "${common_args[@]}" >"$crab_report" -CARGO_TARGET_DIR="$target_dir" cargo run --quiet --release \ - --manifest-path "$script_dir/celld/Cargo.toml" -- "${common_args[@]}" >"$celld_report" - -printf '%s\n' "=== crab-ltx ===" -cat "$crab_report" -printf '%s\n' "=== celld-ltx ===" -cat "$celld_report" diff --git a/crates/crab-ltx/src/bundle.rs b/crates/crab-ltx/src/bundle.rs deleted file mode 100644 index 9220d278d..000000000 --- a/crates/crab-ltx/src/bundle.rs +++ /dev/null @@ -1,610 +0,0 @@ -//! Celld-inspired verbatim LTX envelopes with bounded, checksum-verified rows. -//! Apache-2.0; adapted from bundle.rs at the revision in UPSTREAM.md. - -use crate::{CrabError, Limits, Result, SegmentInfo}; -use bytes::Bytes; -use crab_storage::{MultipartUploadSource, StorageError}; -use serde::{Deserialize, Serialize}; -use std::{ - collections::BTreeSet, - fs::{File, OpenOptions}, - io::{Read, Seek, SeekFrom, Write}, - path::{Path, PathBuf}, - sync::Arc, -}; -use tempfile::TempPath; - -/// Keeps a server-owned verified bundle artifact alive while an overlay uses it. -pub trait BundleLease: Send + Sync {} - -/// One immutable segment and its repository/epoch identity, before bundling. -pub struct BundleEntry { - /// Canonical repository identity the segment belongs to. - pub repository: String, - /// Canonical epoch identity the segment was captured under. - pub epoch: String, - /// Manifest expectations the segment must satisfy. - pub info: SegmentInfo, - /// Complete immutable segment bytes. - pub bytes: Vec, -} - -impl BundleEntry { - /// Creates an entry using the canonical identity selected by `CellReplica`. - #[must_use] - pub fn for_cell( - cell: [u8; 32], - incarnation: [u8; 16], - info: SegmentInfo, - bytes: Vec, - ) -> Self { - Self { - repository: crate::hex::encode_hex(&cell), - epoch: crate::hex::encode_hex(&incarnation), - info, - bytes, - } - } -} - -/// A verified segment's byte extent in a bundle; identity is not authorization. -#[derive(Clone, Debug, Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -pub struct BundleRow { - /// Canonical repository identity the row was verified against. - pub repository: String, - /// Canonical epoch identity the row was verified against. - pub epoch: String, - /// Manifest expectations the row's segment satisfied. - pub info: SegmentInfo, - /// Byte offset of the segment inside the bundle. - pub offset: u64, -} - -enum BundleBody { - Memory(Bytes), - File(Arc), -} - -struct BundleFile { - path: PathBuf, - remove_on_drop: bool, -} - -impl Drop for BundleFile { - fn drop(&mut self) { - if self.remove_on_drop { - let _ = std::fs::remove_file(&self.path); - } - } -} - -impl Clone for BundleBody { - fn clone(&self) -> Self { - match self { - Self::Memory(bytes) => Self::Memory(bytes.clone()), - Self::File(file) => Self::File(Arc::clone(file)), - } - } -} - -/// An owned, validated envelope containing verbatim checksum-bearing LTX files. -/// -/// The format is Crab's `CRB1`, not Celld's `CLB1`: rows retain string epochs, -/// exact ranges and whole-file checksums. A manifest must pin its locations. -pub struct Bundle { - body: BundleBody, - rows: Vec, - digest: [u8; 32], - length: u64, -} - -/// File-backed builder for a verified bundle. -/// -/// Segment bodies are written as they arrive; only the bounded row table is -/// retained in memory. `finish` seals the footer and revalidates the complete -/// envelope before returning the owned temporary bundle. -pub struct BundleBuilder { - path: TempPath, - rows: Vec, - identities: BTreeSet<(String, String, u64, u64)>, - payload_len: u64, - limits: Limits, - poisoned: bool, -} - -impl BundleBuilder { - /// Creates a private temporary bundle in an existing runtime scratch dir. - pub fn new_temp(directory: &Path, limits: Limits) -> Result { - let limits = limits.validate()?; - let file = tempfile::Builder::new() - .prefix(".crab-bundle-") - .tempfile_in(directory)?; - Ok(Self { - path: file.into_temp_path(), - rows: Vec::new(), - identities: BTreeSet::new(), - payload_len: 0, - limits, - poisoned: false, - }) - } - - /// Appends one fully verified segment without retaining its body. - pub fn push(&mut self, entry: BundleEntry) -> Result<()> { - if self.poisoned { - return Err(CrabError::InvalidState("bundle builder is poisoned")); - } - if self.rows.len() >= self.limits.max_segments - || entry.repository.is_empty() - || entry.repository.len() > 4096 - || !valid_epoch(&entry.epoch) - { - return Err(CrabError::Limit(crate::LimitKind::BundleEntries)); - } - crate::recovery::verify_segment(&entry.bytes, &entry.info, self.limits)?; - let identity = ( - entry.repository.clone(), - entry.epoch.clone(), - entry.info.min_txid, - entry.info.max_txid, - ); - if !self.identities.insert(identity) { - return Err(CrabError::LTXCorrupted); - } - let next_len = self - .payload_len - .checked_add(entry.info.size_bytes) - .ok_or(CrabError::Limit(crate::LimitKind::BundleBytes))?; - if next_len > self.limits.max_plan_bytes { - return Err(CrabError::Limit(crate::LimitKind::BundleBytes)); - } - let write_result = OpenOptions::new() - .append(true) - .open(&self.path) - .and_then(|mut file| file.write_all(&entry.bytes)); - if let Err(error) = write_result { - self.poisoned = true; - return Err(error.into()); - } - self.rows.push(BundleRow { - repository: entry.repository, - epoch: entry.epoch, - info: entry.info, - offset: self.payload_len, - }); - self.payload_len = next_len; - Ok(()) - } - - /// Seals, syncs, validates, and transfers ownership of the bundle file. - pub fn finish(self) -> Result { - if self.poisoned || self.rows.is_empty() { - return Err(CrabError::InvalidState("bundle builder has no valid rows")); - } - let footer = serde_json::to_vec(&self.rows)?; - let footer_len = u32::try_from(footer.len()) - .map_err(|_| CrabError::Limit(crate::LimitKind::BundleFooter))?; - let total = self - .payload_len - .checked_add(footer.len() as u64) - .and_then(|length| length.checked_add(8)) - .ok_or(CrabError::Limit(crate::LimitKind::BundleBytes))?; - if total > self.limits.max_plan_bytes { - return Err(CrabError::Limit(crate::LimitKind::BundleBytes)); - } - let mut file = OpenOptions::new().append(true).open(&self.path)?; - file.write_all(&footer)?; - file.write_all(&footer_len.to_le_bytes())?; - file.write_all(b"CRB1")?; - file.sync_all()?; - Bundle::decode_temp_file(self.path, self.limits) - } -} - -impl Bundle { - /// Encodes bounded segments; duplicate identities and malformed LTX are rejected. - pub fn encode(entries: Vec, limits: Limits) -> Result { - let limits = limits.validate()?; - if entries.is_empty() || entries.len() > limits.max_segments { - return Err(CrabError::Limit(crate::LimitKind::BundleEntries)); - } - let mut bytes = Vec::new(); - let mut rows = Vec::new(); - for entry in entries { - if entry.repository.is_empty() - || entry.repository.len() > 4096 - || !valid_epoch(&entry.epoch) - { - return Err(CrabError::LTXCorrupted); - } - crate::recovery::verify_segment(&entry.bytes, &entry.info, limits)?; - if (bytes.len() as u64).saturating_add(entry.info.size_bytes) > limits.max_plan_bytes { - return Err(CrabError::Limit(crate::LimitKind::BundleBytes)); - } - rows.push(BundleRow { - repository: entry.repository, - epoch: entry.epoch, - info: entry.info, - offset: bytes.len() as u64, - }); - bytes.extend(entry.bytes); - } - let footer = serde_json::to_vec(&rows)?; - let len = u32::try_from(footer.len()) - .map_err(|_| CrabError::Limit(crate::LimitKind::BundleFooter))?; - bytes.extend(footer); - bytes.extend(len.to_le_bytes()); - bytes.extend(b"CRB1"); - Self::decode(bytes, limits) - } - - /// Verifies the complete envelope, every extent, and every segment digest. - pub fn decode(bytes: Vec, limits: Limits) -> Result { - Self::decode_bytes(bytes.into(), limits) - } - - /// Verifies a shared envelope without copying its complete body. - pub fn decode_bytes(bytes: Bytes, limits: Limits) -> Result { - let limits = limits.validate()?; - if bytes.len() as u64 > limits.max_plan_bytes { - return Err(CrabError::Limit(crate::LimitKind::BundleBytes)); - } - let (rows, payload_end) = decode_footer(&bytes, limits)?; - validate_row_layout(&rows, payload_end, limits)?; - for row in &rows { - let start = usize::try_from(row.offset).map_err(|_| CrabError::LTXCorrupted)?; - let end = start - .checked_add( - usize::try_from(row.info.size_bytes).map_err(|_| CrabError::LTXCorrupted)?, - ) - .ok_or(CrabError::LTXCorrupted)?; - crate::recovery::verify_segment(&bytes[start..end], &row.info, limits)?; - } - Ok(Self { - digest: *blake3::hash(&bytes).as_bytes(), - length: bytes.len() as u64, - body: BundleBody::Memory(bytes), - rows, - }) - } - - /// Verifies an envelope from a caller-owned file without retaining its body in memory. - pub fn decode_file(path: &Path, limits: Limits) -> Result { - let (rows, length, digest) = decode_file_contents(path, limits)?; - Ok(Self { - body: BundleBody::File(Arc::new(BundleFile { - path: path.to_owned(), - remove_on_drop: false, - })), - rows, - digest, - length, - }) - } - - /// Verifies a temporary envelope and removes its file when the bundle is dropped. - pub fn decode_temp_file(path: TempPath, limits: Limits) -> Result { - let source = path.to_path_buf(); - let (rows, length, digest) = decode_file_contents(&source, limits)?; - let source = path - .keep() - .map_err(|error| CrabError::Other(Box::new(error)))?; - Ok(Self { - body: BundleBody::File(Arc::new(BundleFile { - path: source, - remove_on_drop: true, - })), - rows, - digest, - length, - }) - } - - /// Verifies the expected outer digest before parsing a temporary envelope. - pub fn decode_temp_file_with_digest( - path: TempPath, - expected: [u8; 32], - limits: Limits, - ) -> Result { - let source = path.to_path_buf(); - let length = std::fs::metadata(&source)?.len(); - let limits = limits.validate()?; - if length > limits.max_plan_bytes { - return Err(CrabError::Limit(crate::LimitKind::BundleBytes)); - } - let digest = hash_file(&source)?; - if digest != expected { - return Err(CrabError::ChecksumMismatch); - } - let (rows, verified_length) = decode_file_rows(&source, limits)?; - if verified_length != length { - return Err(CrabError::LTXCorrupted); - } - let source = path - .keep() - .map_err(|error| CrabError::Other(Box::new(error)))?; - Ok(Self { - body: BundleBody::File(Arc::new(BundleFile { - path: source, - remove_on_drop: true, - })), - rows, - digest, - length, - }) - } - - /// Detaches a uniquely owned temporary bundle file from this bundle. - /// - /// The caller assumes responsibility for the returned path. Memory-backed - /// bundles and bundles still referenced by an upload source cannot detach. - pub fn detach_file(self) -> Result { - let Bundle { body, .. } = self; - let BundleBody::File(file) = body else { - return Err(CrabError::InvalidState( - "memory-backed bundle has no detachable file", - )); - }; - let mut file = Arc::try_unwrap(file) - .map_err(|_| CrabError::InvalidState("bundle file is still referenced"))?; - file.remove_on_drop = false; - Ok(std::mem::take(&mut file.path)) - } - - /// Returns the verified segment rows in bundle order. - #[must_use] - pub fn rows(&self) -> &[BundleRow] { - &self.rows - } - - /// Returns the total verified bundle length in bytes. - #[must_use] - pub const fn len(&self) -> u64 { - self.length - } - - /// Reports whether the verified bundle holds no bytes. - #[must_use] - pub fn is_empty(&self) -> bool { - self.length == 0 - } - - /// Returns the BLAKE3 digest of the verified envelope. - #[must_use] - pub const fn digest(&self) -> [u8; 32] { - self.digest - } - - /// Returns the complete verified body, allocating only for file-backed bundles. - pub fn read_all(&self) -> Result { - match &self.body { - BundleBody::Memory(bytes) => Ok(bytes.clone()), - BundleBody::File(file) => { - let bytes = std::fs::read(&file.path)?; - if bytes.len() as u64 != self.length - || *blake3::hash(&bytes).as_bytes() != self.digest - { - return Err(CrabError::ChecksumMismatch); - } - Ok(Bytes::from(bytes)) - } - } - } - - /// Returns the verified bundle body, re-checking a file-backed digest. - #[must_use = "use or handle the verified bundle bytes"] - pub fn bytes(&self) -> Result { - self.read_all() - } - - /// Returns a re-openable multipart source for the verified envelope. - pub fn upload_source(&self) -> Arc { - Arc::new(BundleUploadSource { - body: self.body.clone(), - length: self.length, - }) - } - - /// Returns the verified bytes for a row index, never a caller-supplied extent. - pub fn read_segment(&self, index: usize) -> Result { - let row = self.rows.get(index).ok_or(CrabError::TxNotAvailable)?; - let start = usize::try_from(row.offset).map_err(|_| CrabError::LTXCorrupted)?; - let length = usize::try_from(row.info.size_bytes).map_err(|_| CrabError::LTXCorrupted)?; - match &self.body { - BundleBody::Memory(bytes) => Ok(bytes.slice(start..start + length)), - BundleBody::File(file) => { - let mut source = File::open(&file.path)?; - let mut bytes = vec![0; length]; - read_exact_at(&mut source, row.offset, &mut bytes)?; - if *blake3::hash(&bytes).as_bytes() != row.info.blake3 { - return Err(CrabError::ChecksumMismatch); - } - Ok(Bytes::from(bytes)) - } - } - } -} - -struct BundleUploadSource { - body: BundleBody, - length: u64, -} - -#[async_trait::async_trait] -impl MultipartUploadSource for BundleUploadSource { - async fn byte_len(&self) -> crab_storage::Result { - match &self.body { - BundleBody::Memory(_) => Ok(self.length), - BundleBody::File(file) => Ok(tokio::fs::metadata(&file.path).await?.len()), - } - } - - async fn read_exact(&self, offset: u64, length: usize) -> crab_storage::Result { - let end = offset - .checked_add(length as u64) - .ok_or_else(|| StorageError::ReadRejected { - source: Box::new(std::io::Error::new( - std::io::ErrorKind::InvalidInput, - "bundle upload range overflows", - )), - })?; - if end > self.length { - return Err(StorageError::ReadRejected { - source: Box::new(std::io::Error::new( - std::io::ErrorKind::UnexpectedEof, - "bundle upload range exceeds verified body", - )), - }); - } - match &self.body { - BundleBody::Memory(bytes) => { - let start = usize::try_from(offset).map_err(|_| StorageError::ReadRejected { - source: Box::new(std::io::Error::other("bundle upload offset overflows")), - })?; - Ok(bytes.slice(start..start + length)) - } - BundleBody::File(file) => { - let path = file.path.clone(); - tokio::task::spawn_blocking(move || { - let mut source = File::open(path)?; - let mut bytes = vec![0; length]; - read_exact_at(&mut source, offset, &mut bytes)?; - Ok::<_, std::io::Error>(Bytes::from(bytes)) - }) - .await - .map_err(|error| StorageError::ReadRejected { - source: Box::new(error), - })? - .map_err(Into::into) - } - } - } -} - -fn decode_footer(bytes: &[u8], limits: Limits) -> Result<(Vec, u64)> { - let trailer = bytes.len().checked_sub(8).ok_or(CrabError::LTXCorrupted)?; - if &bytes[trailer + 4..] != b"CRB1" { - return Err(CrabError::LTXCorrupted); - } - let length = u32::from_le_bytes( - bytes[trailer..trailer + 4] - .try_into() - .map_err(|_| CrabError::LTXCorrupted)?, - ) as usize; - let start = trailer.checked_sub(length).ok_or(CrabError::LTXCorrupted)?; - let rows: Vec = serde_json::from_slice(&bytes[start..trailer])?; - if rows.is_empty() || rows.len() > limits.max_segments { - return Err(CrabError::Limit(crate::LimitKind::BundleEntries)); - } - Ok((rows, start as u64)) -} - -fn decode_file_contents(path: &Path, limits: Limits) -> Result<(Vec, u64, [u8; 32])> { - let limits = limits.validate()?; - let (rows, length) = decode_file_rows(path, limits)?; - let digest = hash_file(path)?; - Ok((rows, length, digest)) -} - -fn decode_file_rows(path: &Path, limits: Limits) -> Result<(Vec, u64)> { - let mut source = File::open(path)?; - let length = source.metadata()?.len(); - if length > limits.max_plan_bytes { - return Err(CrabError::Limit(crate::LimitKind::BundleBytes)); - } - let trailer_offset = length.checked_sub(8).ok_or(CrabError::LTXCorrupted)?; - let mut trailer = [0; 8]; - read_exact_at(&mut source, trailer_offset, &mut trailer)?; - if &trailer[4..] != b"CRB1" { - return Err(CrabError::LTXCorrupted); - } - let footer_length = u32::from_le_bytes( - trailer[..4] - .try_into() - .map_err(|_| CrabError::LTXCorrupted)?, - ) as u64; - let footer_start = trailer_offset - .checked_sub(footer_length) - .ok_or(CrabError::LTXCorrupted)?; - let footer_size = usize::try_from(footer_length) - .map_err(|_| CrabError::Limit(crate::LimitKind::BundleFooter))?; - let mut footer = vec![0; footer_size]; - read_exact_at(&mut source, footer_start, &mut footer)?; - let rows: Vec = serde_json::from_slice(&footer)?; - validate_row_layout(&rows, footer_start, limits)?; - for row in &rows { - let size = usize::try_from(row.info.size_bytes).map_err(|_| CrabError::LTXCorrupted)?; - let mut bytes = vec![0; size]; - read_exact_at(&mut source, row.offset, &mut bytes)?; - crate::recovery::verify_segment(&bytes, &row.info, limits)?; - } - Ok((rows, length)) -} - -fn hash_file(path: &Path) -> Result<[u8; 32]> { - let mut source = File::open(path)?; - let mut digest = blake3::Hasher::new(); - let mut buffer = vec![0; 1 << 20]; - loop { - let read = source.read(&mut buffer)?; - if read == 0 { - break; - } - digest.update(&buffer[..read]); - } - Ok(*digest.finalize().as_bytes()) -} - -fn validate_row_layout(rows: &[BundleRow], payload_end: u64, limits: Limits) -> Result<()> { - if rows.is_empty() || rows.len() > limits.max_segments { - return Err(CrabError::Limit(crate::LimitKind::BundleEntries)); - } - let mut end = 0u64; - let mut identities = BTreeSet::new(); - for row in rows { - if row.repository.is_empty() - || row.repository.len() > 4096 - || !valid_epoch(&row.epoch) - || row.offset != end - || !identities.insert(( - &row.repository, - &row.epoch, - row.info.min_txid, - row.info.max_txid, - )) - { - return Err(CrabError::LTXCorrupted); - } - end = end - .checked_add(row.info.size_bytes) - .ok_or(CrabError::LTXCorrupted)?; - if end > payload_end { - return Err(CrabError::LTXCorrupted); - } - } - if end != payload_end { - return Err(CrabError::LTXCorrupted); - } - Ok(()) -} - -fn read_exact_at(file: &mut File, offset: u64, bytes: &mut [u8]) -> std::io::Result<()> { - file.seek(SeekFrom::Start(offset))?; - file.read_exact(bytes) -} - -pub(crate) fn cell_identity(cell: &[u8; 32], incarnation: &[u8; 16]) -> (String, String) { - ( - crate::hex::encode_hex(cell), - crate::hex::encode_hex(incarnation), - ) -} - -fn valid_epoch(epoch: &str) -> bool { - !epoch.is_empty() - && epoch.len() <= 128 - && epoch - .bytes() - .all(|byte| byte.is_ascii_alphanumeric() || byte == b'-' || byte == b'_') -} diff --git a/crates/crab-ltx/src/capture.rs b/crates/crab-ltx/src/capture.rs deleted file mode 100644 index d8d2142cc..000000000 --- a/crates/crab-ltx/src/capture.rs +++ /dev/null @@ -1,719 +0,0 @@ -// Derived from denoland/celld, commit 10cb1303dac710dcb3b557e318e08c855261f68b. -// Apache-2.0; see LICENSE and UPSTREAM.md. Modified by Crab contributors. -//! Synchronous capture of sealed SQLite cuts into LTX frames. - -use crate::error::{CrabError, Result}; -use crate::ltx::{self, lock_pgno}; -use crate::wal::WalReader; -use crate::{ - CHECKPOINT_MODE_PASSIVE, CHECKPOINT_MODE_TRUNCATE, META_DIR_SUFFIX, Pos, Txid, - WAL_FRAME_HEADER_SIZE, WAL_HEADER_SIZE, ltx_file_path, -}; -use rusqlite::Connection; - -mod checkpoint; -mod timing; -mod verify; -mod wal; - -use std::collections::HashMap; -use std::path::{Path, PathBuf}; -use std::time::{Duration, Instant}; -use timing::*; - -/// SQLite checkpoint mode a capture requests when its WAL bound is exceeded. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum CheckpointMode { - /// Checkpoints without waiting for readers or writers. - Passive, - /// Waits for readers and writers so every frame can move. - Full, - /// Like `Full`, and restarts readers still reading from the WAL. - Restart, - /// Like `Restart`, and truncates the WAL to zero frames afterwards. - Truncate, -} - -#[derive(Clone, Copy, Debug)] -struct CheckpointPragma { - busy: bool, - wal_frames: i64, - backfilled: i64, -} - -impl std::fmt::Display for CheckpointMode { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - let s = match self { - CheckpointMode::Passive => CHECKPOINT_MODE_PASSIVE, - CheckpointMode::Full => "FULL", - CheckpointMode::Restart => "RESTART", - CheckpointMode::Truncate => CHECKPOINT_MODE_TRUNCATE, - }; - f.write_str(s) - } -} - -#[derive(Debug, Clone, Default)] -struct SyncInfo { - offset: i64, - salt1: u32, - salt2: u32, - prev_commit: u32, - snapshotting: bool, -} - -const RELATIVE_TRUNCATE_PAGES: u32 = 1024; - -#[derive(Clone, Debug)] -struct LastL0Header { - wal_offset: i64, - wal_size: i64, - wal_salts: Option<(u32, u32)>, - commit: u32, - final_pgno: u32, - final_page: Vec, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(crate) enum TimingPhase { - Preparation, - SchemaCheck, - WalExistence, - PositionResolution, - WalRead, - PageCollection, - Verification, - Encode, - LocalWrite, - Fsync, - ParentSync, - Checkpoint, -} - -pub(crate) struct CaptureEngine { - checksums: crate::pages::PageChecksums, - host: crate::LtxHost, - path: PathBuf, - meta_path: PathBuf, - conn: Connection, - rtx_conn: Connection, - page_size: u32, - - read_lock_held: bool, - - pub min_checkpoint_page_n: u32, - pub truncate_page_n: u32, - pub checkpoint_interval: Duration, - - synced_since_checkpoint: bool, - synced_to_wal_end: bool, - last_synced_wal_offset: i64, - last_db_pages: u32, - checkpointed_wal_offset: i64, - verified_schema_version: Option, - last_l0_header: Option<(Txid, LastL0Header)>, - sealed_l0_segments: HashMap, - /// Largest one incremental LTX cut may be. - /// - /// A commit whose delta cannot fit this bound is captured as a full - /// database image instead, which is bounded by the host's `max_file_bytes`. - /// The commit-time admission in `Db::transaction_with` normally refuses such - /// a commit before SQLite publishes it. - max_incremental_bytes: u64, - #[cfg(feature = "replica")] - sealed_l0_captured_indexes: HashMap>, - - position: Pos, - l0_dir_ready: bool, - l0_ancestors_durable: bool, - wal_file: Option, - timing: Option, - defer_durability: bool, -} - -const CONTROL_TABLES_DDL: &str = "CREATE TABLE IF NOT EXISTS _litestream_seq (id INTEGER PRIMARY KEY, seq INTEGER);\ - CREATE TABLE IF NOT EXISTS _litestream_lock (id INTEGER);"; -#[cfg(feature = "replica")] -const RETAINED_CAPTURE_INDEX_BYTES: usize = 1 << 20; - -impl CaptureEngine { - pub const DEFAULT_MIN_CHECKPOINT_PAGE_N: u32 = 1000; - pub const DEFAULT_TRUNCATE_PAGE_N: u32 = 121_359; - - pub const DEFAULT_CHECKPOINT_INTERVAL: Duration = Duration::from_secs(60); - pub const DEFAULT_BUSY_TIMEOUT: Duration = Duration::from_secs(1); - - pub fn open_with_host( - path: impl AsRef, - host: crate::LtxHost, - vfs: Option<&str>, - max_incremental_bytes: u64, - ) -> Result { - let path = path.as_ref().to_path_buf(); - let meta_path = Self::meta_path_for(&path); - - let open = |path: &Path| crate::db::open_connection(path, vfs); - let conn = open(&path).map_err(CrabError::Sqlite)?; - - // All managed writers disable autocheckpoint. The separate long-lived - // read mark protects the WAL until the capture/checkpoint barrier runs. - conn.busy_timeout(Self::DEFAULT_BUSY_TIMEOUT) - .map_err(CrabError::Sqlite)?; - conn.pragma_update(None, "wal_autocheckpoint", 0) - .map_err(CrabError::Sqlite)?; - conn.pragma_update(None, "synchronous", "FULL") - .map_err(CrabError::Sqlite)?; - conn.pragma_update(None, "foreign_keys", true) - .map_err(CrabError::Sqlite)?; - - // Enable WAL; SQLite returns the new mode on success (db.go:849-853). - let mode: String = conn - .query_row("PRAGMA journal_mode=WAL", [], |r| r.get(0)) - .map_err(CrabError::Sqlite)?; - if mode != "wal" { - return Err(CrabError::Other( - format!("enable wal failed, mode={mode:?}").into(), - )); - } - - conn.execute_batch(CONTROL_TABLES_DDL) - .map_err(CrabError::Sqlite)?; - - // Dedicated read-lock connection (mirrors a second pooled connection). - let rtx_conn = open(&path).map_err(CrabError::Sqlite)?; - rtx_conn - .busy_timeout(Self::DEFAULT_BUSY_TIMEOUT) - .map_err(CrabError::Sqlite)?; - rtx_conn - .pragma_update(None, "wal_autocheckpoint", 0) - .map_err(CrabError::Sqlite)?; - rtx_conn - .pragma_update(None, "synchronous", "FULL") - .map_err(CrabError::Sqlite)?; - rtx_conn - .pragma_update(None, "foreign_keys", true) - .map_err(CrabError::Sqlite)?; - - let mut db = Self { - checksums: crate::pages::PageChecksums::default(), - host, - path, - meta_path, - conn, - rtx_conn, - page_size: 0, - read_lock_held: false, - min_checkpoint_page_n: Self::DEFAULT_MIN_CHECKPOINT_PAGE_N, - truncate_page_n: Self::DEFAULT_TRUNCATE_PAGE_N, - checkpoint_interval: Self::DEFAULT_CHECKPOINT_INTERVAL, - synced_since_checkpoint: false, - synced_to_wal_end: false, - last_synced_wal_offset: 0, - last_db_pages: 0, - checkpointed_wal_offset: 0, - verified_schema_version: None, - last_l0_header: None, - sealed_l0_segments: HashMap::new(), - max_incremental_bytes, - #[cfg(feature = "replica")] - sealed_l0_captured_indexes: HashMap::new(), - position: Pos::ZERO, - l0_dir_ready: false, - l0_ancestors_durable: false, - wal_file: None, - timing: None, - defer_durability: false, - }; - - // Start the long-running read transaction (db.go:867-871). - db.acquire_read_lock()?; - - // Read page size (db.go:874-878). - let page_size: i64 = db - .conn - .query_row("PRAGMA page_size", [], |r| r.get(0)) - .map_err(CrabError::Sqlite)?; - if !ltx::is_valid_page_size(page_size as u32) { - return Err(CrabError::Other( - format!("invalid db page size: {page_size}").into(), - )); - } - // Validated > 0 above; SQLite page sizes are <= 65536. - db.page_size = page_size as u32; - let pages: u32 = db - .conn - .query_row("PRAGMA page_count", [], |r| r.get(0)) - .map_err(CrabError::Sqlite)?; - db.host - .check_database_size(u64::from(pages) * u64::from(db.page_size))?; - - // Ensure the meta directory exists (db.go:880-883). - db.host.create_dir_all(&db.meta_path)?; - - // Capture creates a WAL frame when needed. Writing it during open would - // change a restored root even if the application only reads then closes. - Ok(db) - } - - pub(crate) fn meta_path_for(path: &Path) -> PathBuf { - // Go: filepath.Join(dir, "."+file+MetaDirSuffix) (db.go:206). - let dir = path.parent(); - let file = path.file_name().map(|s| s.to_owned()).unwrap_or_default(); - let mut name = std::ffi::OsString::from("."); - name.push(&file); - name.push(META_DIR_SUFFIX); - match dir { - Some(d) if !d.as_os_str().is_empty() => d.join(name), - _ => PathBuf::from(name), - } - } - - pub fn wal_path(&self) -> PathBuf { - let mut s = self.path.clone().into_os_string(); - s.push("-wal"); - PathBuf::from(s) - } - - pub fn ltx_path(&self, level: u32, min_txid: Txid, max_txid: Txid) -> String { - ltx_file_path(&self.meta_path.to_string_lossy(), level, min_txid, max_txid) - } - - /// Seals the directory chain that contains the LTX file's synced name. - /// - /// Syncing `ltx/0` alone cannot preserve a newly created `0`, `ltx`, or - /// session-directory entry after power loss. Later cuts reuse these names. - pub(crate) fn sync_l0_ancestors(&mut self) -> Result<()> { - if self.l0_ancestors_durable { - return Ok(()); - } - let ltx = self.meta_path.join("ltx"); - let l0 = ltx.join("0"); - self.timing_begin(TimingPhase::ParentSync); - let result = [&l0, <x, &self.meta_path] - .into_iter() - .try_for_each(|path| self.host.facilities.filesystem.sync_parent(path)); - self.timing_end(TimingPhase::ParentSync); - result?; - self.l0_ancestors_durable = true; - Ok(()) - } - - fn acquire_read_lock(&mut self) -> Result<()> { - if self.read_lock_held { - return Ok(()); - } - self.rtx_conn - .prepare_cached("BEGIN") - .and_then(|mut statement| statement.execute([])) - .map_err(CrabError::Sqlite)?; - // Execute a read query to obtain the read lock. On failure, roll back. - if let Err(e) = self - .rtx_conn - .query_row("SELECT COUNT(1) FROM _litestream_seq", [], |r| { - r.get::<_, i64>(0) - }) - { - let _ = self.rtx_conn.execute_batch("ROLLBACK"); - return Err(CrabError::Sqlite(e)); - } - self.read_lock_held = true; - Ok(()) - } - - fn release_read_lock(&mut self) -> Result<()> { - if !self.read_lock_held { - return Ok(()); - } - self.read_lock_held = false; - rollback(&self.rtx_conn) - } - - fn ensure_control_tables(&mut self) -> Result<()> { - // A swept control table is a schema change, and every schema change - // bumps SQLite's schema version — so an unchanged version proves - // the last verification still holds and the DDL (a full parse and - // execute per statement) can be skipped on the hot capture path. - // Through the statement cache: this guard runs on every sync, and a - // fresh `PRAGMA` per sync was one SQL compilation per capture on the - // fleet profile. - let version: i64 = self - .conn - .prepare_cached("PRAGMA schema_version") - .and_then(|mut statement| statement.query_row([], |row| row.get(0))) - .map_err(CrabError::Sqlite)?; - if self.verified_schema_version == Some(version) { - return Ok(()); - } - self.conn - .execute_batch(CONTROL_TABLES_DDL) - .map_err(CrabError::Sqlite)?; - let verified: i64 = self - .conn - .prepare_cached("PRAGMA schema_version") - .and_then(|mut statement| statement.query_row([], |row| row.get(0))) - .map_err(CrabError::Sqlite)?; - self.verified_schema_version = Some(verified); - Ok(()) - } - - fn with_wal_file( - &mut self, - op: impl Fn(&mut crate::HostFile) -> std::io::Result, - ) -> std::io::Result { - let mut attempt = 0; - loop { - if self.wal_file.is_none() { - self.wal_file = Some(self.host.open(&self.wal_path())?); - } - let file = self - .wal_file - .as_mut() - .ok_or_else(|| std::io::Error::other("WAL handle missing"))?; - match op(file) { - Ok(value) => return Ok(value), - Err(error) => { - self.wal_file = None; - let retry = attempt == 0 - && matches!( - error.kind(), - std::io::ErrorKind::NotFound | std::io::ErrorKind::UnexpectedEof - ); - if !retry { - return Err(error); - } - attempt += 1; - } - } - } - } - - fn wal_header_bytes(&mut self) -> Result<[u8; WAL_HEADER_SIZE]> { - let bytes = self.with_wal_file(|file| file.read_exact_at(0, WAL_HEADER_SIZE))?; - self.timing_observe_wal_transfer(0, bytes.len() as u64); - bytes.try_into().map_err(|_| CrabError::LTXCorrupted) - } - - fn wal_bytes_at(&mut self, offset: i64, n: i64) -> Result> { - let bytes = self.with_wal_file(|file| file.read_exact_at(offset as u64, n as usize))?; - self.timing_observe_wal_transfer(0, bytes.len() as u64); - Ok(bytes) - } - - fn read_whole_wal(&mut self) -> Result> { - let bytes = self.host.read(&self.wal_path())?; - self.timing_observe_wal_transfer(bytes.len() as u64, bytes.len() as u64); - Ok(bytes) - } - - fn ensure_wal_exists(&mut self) -> Result<()> { - if self.wal_file_size()? >= WAL_HEADER_SIZE as i64 { - return Ok(()); - } - self.conn - .execute_batch( - "INSERT INTO _litestream_seq (id, seq) VALUES (1, 1) \ - ON CONFLICT (id) DO UPDATE SET seq = seq + 1", - ) - .map_err(CrabError::Sqlite)?; - Ok(()) - } - - fn wal_file_size(&mut self) -> Result { - match self.with_wal_file(|file| file.file_len()) { - Ok(len) => { - self.timing_observe_wal_transfer(len, 0); - Ok(len as i64) - } - Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(0), - Err(e) => Err(e.into()), - } - } - - fn db_file_size(&self) -> Result { - match self.host.metadata(&self.path) { - Ok(md) => Ok(md.len as i64), - Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(0), - Err(e) => Err(e.into()), - } - } - - pub fn pos(&self) -> Pos { - self.position - } - - #[cfg(feature = "replica")] - pub(crate) fn page_size(&self) -> u32 { - self.page_size - } - - #[cfg(feature = "replica")] - pub(crate) fn checksums(&self) -> &crate::pages::PageChecksums { - &self.checksums - } - - pub(crate) fn take_sealed_l0_segment(&mut self, txid: Txid) -> Option { - self.sealed_l0_segments.remove(&txid.0) - } - - #[cfg(feature = "replica")] - pub(crate) fn take_sealed_l0_captured_index(&mut self, txid: Txid) -> Option> { - self.sealed_l0_captured_indexes.remove(&txid.0) - } - - pub(crate) fn seed_continuation( - &mut self, - position: crate::Position, - checksums: crate::pages::PageChecksums, - page_size: u32, - count: u32, - ) -> Result<()> { - if self.position != Pos::ZERO - || page_size != self.page_size - || checksums.checksum() != position.checksum - || position.txid == 0 - { - return Err(CrabError::ChecksumMismatch); - } - self.sealed_l0_segments.clear(); - #[cfg(feature = "replica")] - { - self.sealed_l0_captured_indexes.clear(); - } - self.last_l0_header = Some(( - Txid(position.txid), - LastL0Header { - wal_offset: WAL_HEADER_SIZE as i64, - wal_size: 0, - wal_salts: None, - commit: count, - final_pgno: 0, - final_page: Vec::new(), - }, - )); - self.position = Pos::new(Txid(position.txid), position.checksum); - self.checksums = checksums; - self.last_db_pages = count; - Ok(()) - } - - pub fn sync(&mut self, required: Option) -> Result<()> { - self.sync_impl(required) - } - - pub(crate) fn sync_deferred(&mut self, required: Option) -> Result<()> { - let previous = self.defer_durability; - self.defer_durability = true; - let result = self.sync(required); - self.defer_durability = previous; - result - } - - fn sync_impl(&mut self, required: Option) -> Result<()> { - // Self-heal: recreate the control tables if something swept them out - // of `sqlite_schema` from under the replicator — without them every - // capture fails until the database is reopened. A no-op when the - // tables exist (no schema change, no WAL write). - self.timing_begin(TimingPhase::SchemaCheck); - let schema_result = self.ensure_control_tables(); - self.timing_end(TimingPhase::SchemaCheck); - schema_result?; - - // Ensure the WAL has at least one frame (db.go:1017-1020). - self.timing_begin(TimingPhase::WalExistence); - let wal_result = self.ensure_wal_exists(); - self.timing_end(TimingPhase::WalExistence); - wal_result?; - - let (orig_wal_size, new_wal_size, synced) = self.verify_and_sync()?; - - // Validate the application's WAL hook boundary BEFORE any checkpoint - // can replace its salts or write control frames. A valid earlier cut - // alone does not prove the last committed transaction was captured. - if let Some(required) = required { - let (_, header) = self - .last_l0_header - .as_ref() - .ok_or(CrabError::LTXCorrupted)?; - let end = WAL_HEADER_SIZE as i64 - + i64::from(required.frames) - * (i64::from(self.page_size) + WAL_FRAME_HEADER_SIZE as i64); - if header.wal_salts != Some((required.salt1, required.salt2)) - || self.last_synced_wal_offset < end - { - return Err(CrabError::LTXCorrupted); - } - } - - // Track that data was synced for time-based checkpoint decisions. - if synced { - self.synced_since_checkpoint = true; - } - - self.checkpoint_if_needed(orig_wal_size, new_wal_size)?; - - Ok(()) - } - - fn verify_and_sync(&mut self) -> Result<(i64, i64, bool)> { - // Use the last synced WAL offset as the logical size for checkpoint - // decisions; on the first sync fall back to file size (db.go:1062-1069). - let mut orig_wal_size = self.last_synced_wal_offset; - if orig_wal_size == 0 { - orig_wal_size = self.wal_file_size()?; - } - - self.timing_begin(TimingPhase::PositionResolution); - let info_result = self.verify(); - self.timing_end(TimingPhase::PositionResolution); - let info = info_result?; - - let synced = self.sync_inner(info)?; - - let new_wal_size = self.last_synced_wal_offset; - Ok((orig_wal_size, new_wal_size, synced)) - } - - pub fn snapshot_to_writer(&mut self, w: &mut W) -> Result { - if self.page_size == 0 { - return Err(CrabError::Other( - "db not ready: page size not initialized".into(), - )); - } - - let pos = self.position; - - let db_size = self.db_file_size()?; - let mut commit = (db_size / self.page_size as i64) as u32; - - let wal = WalImage::whole(self.read_whole_wal()?); - let mut rd = WalReader::new(&wal.bytes).map_err(CrabError::from)?; - let (page_map, max_offset, wal_commit) = rd.page_map().map_err(CrabError::from)?; - if wal_commit > 0 { - commit = wal_commit; - } - let wal_offset = rd.offset(); - let sz = if max_offset > 0 { - max_offset - wal_offset - } else { - 0 - }; - let (salt1, salt2) = rd.salt(); - - self.host - .check_database_size(u64::from(commit) * u64::from(self.page_size))?; - let header = ltx::Header { - version: ltx::VERSION, - flags: 0, - page_size: self.page_size, - commit, - min_txid: Txid(1), - max_txid: pos.txid, - timestamp: self.host.now_unix_millis(), - pre_apply_checksum: 0, - wal_offset, - wal_size: sz, - wal_salt1: salt1, - wal_salt2: salt2, - node_id: 0, - }; - - // A snapshot tracks the rolling post-apply checksum (MinTXID==1, no - // NoChecksum flag). Encode each page as it is read so callers can back - // the writer with a bounded scratch file instead of a database-sized - // resident buffer. - let lock = lock_pgno(self.page_size); - let mut rolling: crate::Checksum = crate::CHECKSUM_FLAG; - let mut encoder = crate::codec::Encoder::new_block(w); - encoder.encode_header(header)?; - for pgno in (1..=commit).filter(|page| *page != lock) { - let data = self.capture_page(&wal, &page_map, pgno)?; - rolling = crate::CHECKSUM_FLAG | (rolling ^ ltx::checksum_page(pgno, &data)); - encoder.encode_page(ltx::PageHeader { pgno, flags: 0 }, &data)?; - } - encoder.close(rolling)?; - - Ok(Pos::new(pos.txid, rolling)) - } - - pub fn close(mut self) -> Result<()> { - self.release_read_lock()?; - // The connection is dropped here, closing it. - Ok(()) - } -} - -fn calc_wal_size(page_size: u32, page_n: u32) -> i64 { - WAL_HEADER_SIZE as i64 + (WAL_FRAME_HEADER_SIZE as i64 + page_size as i64) * page_n as i64 -} - -fn rollback(conn: &Connection) -> Result<()> { - // SQLite can auto-rollback on I/O failure. Check native state instead of - // swallowing arbitrary errors whose text happens to mention rollback. - if conn.is_autocommit() { - return Ok(()); - } - conn.execute_batch("ROLLBACK").map_err(CrabError::Sqlite) -} - -struct WalImage { - bytes: Vec, - tail_base: usize, -} - -impl WalImage { - fn whole(bytes: Vec) -> Self { - Self { - bytes, - tail_base: 0, - } - } - - fn reader_at( - &self, - offset: i64, - salt1: u32, - salt2: u32, - ) -> std::result::Result, crate::wal::WalError> { - if self.tail_base == 0 { - WalReader::new_with_offset(&self.bytes, offset, salt1, salt2) - } else { - WalReader::new_with_offset_over_tail( - &self.bytes, - self.tail_base as i64, - offset, - salt1, - salt2, - ) - } - } - - fn slice(&self, offset: i64, n: usize) -> Option<&[u8]> { - if offset < 0 { - return None; - } - let offset = offset as usize; - let start = if self.tail_base == 0 || offset < WAL_HEADER_SIZE { - offset - } else { - WAL_HEADER_SIZE + offset.checked_sub(self.tail_base)? - }; - let end = start.checked_add(n)?; - (end <= self.bytes.len()).then(|| &self.bytes[start..end]) - } - - fn page(&self, offset: i64, page_size: u32) -> Result> { - self.slice(offset + WAL_FRAME_HEADER_SIZE as i64, page_size as usize) - .map(<[u8]>::to_vec) - .ok_or_else(|| { - CrabError::Io(std::io::Error::new( - std::io::ErrorKind::UnexpectedEof, - format!("short read wal page @ {offset}"), - )) - }) - } -} - -#[inline] -fn be_u32(b: &[u8]) -> u32 { - u32::from_be_bytes([b[0], b[1], b[2], b[3]]) -} diff --git a/crates/crab-ltx/src/capture/checkpoint.rs b/crates/crab-ltx/src/capture/checkpoint.rs deleted file mode 100644 index 1491717d3..000000000 --- a/crates/crab-ltx/src/capture/checkpoint.rs +++ /dev/null @@ -1,288 +0,0 @@ -// Derived from denoland/celld, commit 10cb1303dac710dcb3b557e318e08c855261f68b. -// Apache-2.0; see LICENSE and UPSTREAM.md. Modified by Crab contributors. -// Split from upstream db.rs; see UPSTREAM.md for Crab's changes. - -use super::*; - -impl CaptureEngine { - pub(super) fn checkpoint_if_needed( - &mut self, - orig_wal_size: i64, - new_wal_size: i64, - ) -> Result<()> { - if self.page_size == 0 { - return Ok(()); - } - - // Priority 1: emergency TRUNCATE (blocking) on the *original* logical - // size. A truncate ends in a boundary image of the whole database - // (see `checkpoint`). For a small database that image is the cheap - // price the threshold was tuned for, so below `RELATIVE_TRUNCATE_PAGES` - // the threshold is absolute, as upstream's. Above it the WAL must also - // have grown past the database before a truncate: the image then costs - // at most what the WAL it replaces did, and the chain stays within 2x - // of the writes. A fixed threshold made a 1MB-row whale pay a - // database-sized capture every write, and its chain grew as the - // square of its size. - if self.truncate_page_n > 0 { - let relative = if self.last_db_pages > RELATIVE_TRUNCATE_PAGES { - self.last_db_pages - } else { - 0 - }; - let threshold = self.truncate_page_n.max(relative); - if orig_wal_size >= calc_wal_size(self.page_size, threshold) { - return self.checkpoint(CheckpointMode::Truncate); - } - } - - // Priority 2: PASSIVE once the frames appended since the last - // backfill reach the threshold. See `checkpointed_wal_offset` for why - // this is not the whole logical size: a checkpoint whose sealing - // write could not restart the WAL must cost one retry at the next - // threshold, not a checkpoint per sync. - let backfilled_through = self.checkpointed_wal_offset.clamp( - WAL_HEADER_SIZE as i64, - new_wal_size.max(WAL_HEADER_SIZE as i64), - ); - let threshold = - calc_wal_size(self.page_size, self.min_checkpoint_page_n) - WAL_HEADER_SIZE as i64; - if new_wal_size - backfilled_through >= threshold { - return self.checkpoint(CheckpointMode::Passive); - } - - // Priority 3: time-based PASSIVE, gated on data synced since last - // checkpoint (#896). Uses the DB-file mtime and a logical-size guard so an - // idle DB does not spin LTX files (db.go:1133-1153). - if self.checkpoint_interval > Duration::ZERO && self.synced_since_checkpoint { - let elapsed = self.host.file_age(&self.path)?; - if elapsed > self.checkpoint_interval && new_wal_size > calc_wal_size(self.page_size, 1) - { - return self.checkpoint(CheckpointMode::Passive); - } - } - - Ok(()) - } - - pub(crate) fn checkpoint(&mut self, mode: CheckpointMode) -> Result<()> { - // Self-heal, as in `sync`: `checkpoint` writes to both control tables - // and re-acquires the read lock through `_litestream_seq`, and the - // invariant is that the tables exist before any control-table - // statement — not only before a capture. - self.timing_begin(TimingPhase::SchemaCheck); - let schema_result = self.ensure_control_tables(); - self.timing_end(TimingPhase::SchemaCheck); - schema_result?; - - // Read the WAL header before the checkpoint to detect a restart. - let hdr = self.wal_header_bytes()?; - - // Copy the end of the WAL before the checkpoint to capture as much as - // possible (db.go:1823-1826). - self.verify_and_sync()?; - - let frame_size = self.page_size as i64 + WAL_FRAME_HEADER_SIZE as i64; - let pre_checkpoint_frame_n = if self.last_synced_wal_offset > WAL_HEADER_SIZE as i64 { - (self.last_synced_wal_offset - WAL_HEADER_SIZE as i64) / frame_size - } else { - 0 - }; - - // A passive checkpoint does not acquire SQLite's writer lock. Hold a - // short write transaction on the dedicated read-lock connection, then - // sync again to seal every commit before running the checkpoint on the - // main connection. Keep the barrier until the checkpoint completes. - let pragma = if mode == CheckpointMode::Passive { - self.exec_passive_checkpoint_with_barrier(hdr)? - } else { - self.exec_checkpoint(mode)? - }; - // The backfilled boundary in this WAL's coordinates. A short backfill - // (a reader pinned the WAL) leaves the remainder counting toward the - // next threshold, so a pinned WAL retries at the threshold and an - // unpinned one does not retry at all. - self.checkpointed_wal_offset = - WAL_HEADER_SIZE as i64 + pragma.backfilled.max(0) * frame_size; - - // Force a write so a restarted WAL has a new header and at least one - // frame that verify can read. - self.conn - .execute_batch( - "INSERT INTO _litestream_seq (id, seq) VALUES (1, 1) \ - ON CONFLICT (id) DO UPDATE SET seq = seq + 1", - ) - .map_err(CrabError::Sqlite)?; - - // If the WAL header is unchanged, the WAL did not restart — done. - let other = self.wal_header_bytes()?; - if hdr == other { - self.synced_since_checkpoint = false; - return Ok(()); - } - if let Some(timing) = &mut self.timing { - timing.checkpoint_restart(); - } - - // The WAL restarted. Grab the write lock, then either copy the new WAL - // tail or take a complete boundary image. TRUNCATE always needs the - // boundary image because SQLite reports zero frames after resetting the - // WAL. A forced checkpoint also needs one if it covered more frames than - // the sealed pre-checkpoint sync observed. - self.conn - .prepare_cached("BEGIN") - .and_then(|mut statement| statement.execute([])) - .map_err(CrabError::Sqlite)?; - let post = (|| -> Result<()> { - self.conn - .prepare_cached("INSERT INTO _litestream_lock (id) VALUES (1)") - .and_then(|mut statement| statement.execute([])) - .map_err(CrabError::Sqlite)?; - if mode == CheckpointMode::Truncate - || (mode != CheckpointMode::Passive && pragma.wal_frames > pre_checkpoint_frame_n) - { - let info = SyncInfo { - offset: WAL_HEADER_SIZE as i64, - salt1: be_u32(&other[16..]), - salt2: be_u32(&other[20..]), - snapshotting: true, - - ..Default::default() - }; - self.sync_inner(info)?; - } else { - self.verify_and_sync()?; - } - Ok(()) - })(); - // Always roll back the write transaction (db.go:1849,1867). - let rb = rollback(&self.conn); - post?; - rb?; - - self.synced_since_checkpoint = false; - Ok(()) - } - - pub(super) fn exec_passive_checkpoint_with_barrier( - &mut self, - pre_checkpoint_header: [u8; WAL_HEADER_SIZE], - ) -> Result { - self.release_read_lock()?; - - let result = (|| -> Result { - self.rtx_conn - .prepare_cached("BEGIN") - .and_then(|mut statement| statement.execute([])) - .map_err(CrabError::Sqlite)?; - self.rtx_conn - .prepare_cached("INSERT INTO _litestream_lock (id) VALUES (1)") - .and_then(|mut statement| statement.execute([])) - .map_err(CrabError::Sqlite)?; - - // Writers can cross their own autocheckpoint threshold after the - // read lock is released but before this barrier wins SQLite's - // writer lock. A changed WAL header means some of those commits - // can already live only in the database file. The normal - // synced-to-end shortcut cannot distinguish that race from our own - // completed checkpoint, so an incremental LTX can omit the - // checkpointed pages and produce a malformed restore. Seal the - // complete boundary while the writer lock makes the database file - // and the new WAL tail a stable pair. The cost is one database-size - // LTX file only when another checkpoint wins this narrow gap. - let barrier_header = self.wal_header_bytes()?; - if barrier_header != pre_checkpoint_header { - let info = SyncInfo { - offset: WAL_HEADER_SIZE as i64, - salt1: be_u32(&barrier_header[16..]), - salt2: be_u32(&barrier_header[20..]), - snapshotting: true, - - ..Default::default() - }; - self.sync_inner(info)?; - } else { - // Commits can land between the earlier sync and acquisition of - // the barrier. This second sync seals them before the - // checkpoint. - self.verify_and_sync()?; - } - self.run_checkpoint_pragma(CheckpointMode::Passive) - })(); - - // Release the writer barrier before restoring the long-lived read lock. - // Preserve the operation error if both the operation and cleanup fail. - let rollback_result = rollback(&self.rtx_conn); - let reacquire_result = self.acquire_read_lock(); - match result { - Err(error) => Err(error), - Ok(pragma) => { - rollback_result?; - reacquire_result?; - Ok(pragma) - } - } - } - - pub(super) fn exec_checkpoint(&mut self, mode: CheckpointMode) -> Result { - // Ensure the read lock is removed before the checkpoint; defer the - // re-acquire so it runs even on early return. - self.release_read_lock()?; - - let result = self.run_checkpoint_pragma(mode); - - // Re-acquire the read lock immediately after the checkpoint (the deferred - // re-acquire in Go). If the pragma succeeded, propagate any re-acquire - // error; otherwise surface the original pragma error. - let reacquire = self.acquire_read_lock(); - match (result, reacquire) { - (Ok(pragma), Ok(())) => Ok(pragma), - (Ok(_), Err(e)) => Err(e), - (Err(e), _) => Err(e), - } - } - - pub(super) fn run_checkpoint_pragma( - &mut self, - mode: CheckpointMode, - ) -> Result { - let sql = format!("PRAGMA wal_checkpoint({mode})"); - if let Some(timing) = &mut self.timing { - timing.checkpoint_run(); - } - self.timing_begin(TimingPhase::Checkpoint); - let result = (|| -> Result { - self.conn - .prepare_cached(&sql) - .map_err(CrabError::Sqlite)? - .query_row([], |row| { - Ok(CheckpointPragma { - busy: row.get::<_, i64>(0)? != 0, - wal_frames: row.get::<_, i64>(1)?, - backfilled: row.get::<_, i64>(2)?, - }) - }) - .map_err(CrabError::Sqlite) - })(); - self.timing_end(TimingPhase::Checkpoint); - match &result { - Ok(pragma) => { - if let Some(timing) = &mut self.timing { - timing.checkpoint_result(pragma.busy, pragma.wal_frames, pragma.backfilled); - } - } - Err(CrabError::Sqlite(error)) - if matches!( - error.sqlite_error_code(), - Some(rusqlite::ErrorCode::DatabaseBusy | rusqlite::ErrorCode::DatabaseLocked) - ) => - { - if let Some(timing) = &mut self.timing { - timing.checkpoint_busy_error(); - } - } - Err(_) => {} - } - result - } -} diff --git a/crates/crab-ltx/src/capture/timing.rs b/crates/crab-ltx/src/capture/timing.rs deleted file mode 100644 index 9354eaa8a..000000000 --- a/crates/crab-ltx/src/capture/timing.rs +++ /dev/null @@ -1,253 +0,0 @@ -//! Capture timing recorder and the engine's telemetry hooks. - -use super::*; - -pub(crate) struct TimingRecorder { - started: Instant, - active: Option<(TimingPhase, Instant)>, - timing: crate::CaptureTiming, -} - -impl TimingRecorder { - pub(crate) fn new(started: Instant) -> Self { - Self { - started, - active: None, - timing: crate::CaptureTiming::default(), - } - } - - pub(crate) fn begin(&mut self, phase: TimingPhase, now: Instant) { - if let Some((active, started)) = self.active.take() { - self.add_phase(active, now.saturating_duration_since(started)); - } - self.active = Some((phase, now)); - } - - pub(crate) fn end(&mut self, phase: TimingPhase, now: Instant) { - if let Some((active, started)) = self.active - && active == phase - { - self.add_phase(active, now.saturating_duration_since(started)); - self.active = None; - } - } - - pub(crate) fn add_wal_bytes(&mut self, bytes: u64) { - self.timing.wal_bytes = self.timing.wal_bytes.saturating_add(bytes); - } - - pub(crate) fn add_database_bytes(&mut self, bytes: u64) { - self.timing.database_bytes = self.timing.database_bytes.saturating_add(bytes); - } - - pub(crate) fn add_ltx_bytes(&mut self, bytes: u64) { - self.timing.ltx_bytes = self.timing.ltx_bytes.saturating_add(bytes); - } - - pub(crate) fn add_segment(&mut self) { - self.timing.segment_count = self.timing.segment_count.saturating_add(1); - } - - pub(crate) fn add_phase_nanos(&mut self, phase: TimingPhase, elapsed: u64) { - self.add_phase(phase, Duration::from_nanos(elapsed)); - } - - pub(crate) fn observe_wal_image(&mut self, sparse: bool, fallback: bool, bytes: usize) { - if sparse { - self.timing.wal_sparse_reads = self.timing.wal_sparse_reads.saturating_add(1); - } else { - self.timing.wal_full_reads = self.timing.wal_full_reads.saturating_add(1); - } - if fallback { - self.timing.wal_fallback_reads = self.timing.wal_fallback_reads.saturating_add(1); - } - self.timing.wal_image_bytes = self.timing.wal_image_bytes.max(bytes as u64); - } - - pub(crate) fn observe_wal_transfer(&mut self, file_bytes: u64, read_bytes: u64) { - self.timing.wal_file_bytes = self.timing.wal_file_bytes.max(file_bytes); - self.timing.wal_read_bytes = self.timing.wal_read_bytes.saturating_add(read_bytes); - } - - pub(crate) fn observe_wal_snapshot(&mut self) { - self.timing.wal_snapshot_reads = self.timing.wal_snapshot_reads.saturating_add(1); - } - - pub(crate) fn checkpoint_run(&mut self) { - self.timing.checkpoint_runs = self.timing.checkpoint_runs.saturating_add(1); - } - - pub(crate) fn checkpoint_result(&mut self, busy: bool, frames: i64, backfilled: i64) { - self.timing.checkpoint_busy = self.timing.checkpoint_busy.saturating_add(u32::from(busy)); - self.timing.checkpoint_frames = self - .timing - .checkpoint_frames - .saturating_add(u64::try_from(frames.max(0)).unwrap_or_default()); - self.timing.checkpoint_backfilled = self - .timing - .checkpoint_backfilled - .saturating_add(u64::try_from(backfilled.max(0)).unwrap_or_default()); - } - - pub(crate) fn checkpoint_busy_error(&mut self) { - self.timing.checkpoint_busy_errors = self.timing.checkpoint_busy_errors.saturating_add(1); - } - - pub(crate) fn checkpoint_restart(&mut self) { - self.timing.checkpoint_restarts = self.timing.checkpoint_restarts.saturating_add(1); - } - - pub(crate) fn finish(mut self, now: Instant) -> crate::CaptureTiming { - if let Some((active, started)) = self.active.take() { - self.add_phase(active, now.saturating_duration_since(started)); - } - self.timing.total_nanos = nanos(now.saturating_duration_since(self.started)); - self.timing - } - - fn add_phase(&mut self, phase: TimingPhase, elapsed: Duration) { - let target = match phase { - TimingPhase::Preparation => &mut self.timing.preparation_nanos, - TimingPhase::SchemaCheck => &mut self.timing.schema_check_nanos, - TimingPhase::WalExistence => &mut self.timing.wal_existence_nanos, - TimingPhase::PositionResolution => &mut self.timing.position_resolution_nanos, - TimingPhase::WalRead => &mut self.timing.wal_read_nanos, - TimingPhase::PageCollection => &mut self.timing.page_collection_nanos, - TimingPhase::Verification => &mut self.timing.verification_nanos, - TimingPhase::Encode => &mut self.timing.encode_nanos, - TimingPhase::LocalWrite => &mut self.timing.local_write_nanos, - TimingPhase::Fsync => &mut self.timing.fsync_nanos, - TimingPhase::ParentSync => &mut self.timing.parent_sync_nanos, - TimingPhase::Checkpoint => &mut self.timing.checkpoint_nanos, - }; - *target = target.saturating_add(nanos(elapsed)); - } -} - -pub(super) fn nanos(duration: Duration) -> u64 { - u64::try_from(duration.as_nanos()).unwrap_or(u64::MAX) -} - -impl CaptureEngine { - pub(crate) fn start_timing(&mut self, now: Instant) { - self.timing = Some(TimingRecorder::new(now)); - } - - pub(crate) fn finish_timing(&mut self, now: Instant) -> crate::CaptureTiming { - self.timing - .take() - .map(|recorder| recorder.finish(now)) - .unwrap_or_default() - } - - pub(crate) fn timing_begin(&mut self, phase: TimingPhase) { - if self.timing.is_some() { - let now = self.host.now_monotonic(); - if let Some(recorder) = &mut self.timing { - recorder.begin(phase, now); - } - } - } - - pub(crate) fn timing_end(&mut self, phase: TimingPhase) { - if self.timing.is_some() { - let now = self.host.now_monotonic(); - if let Some(recorder) = &mut self.timing { - recorder.end(phase, now); - } - } - } - - pub(crate) fn timing_add_wal_bytes(&mut self, bytes: u64) { - if let Some(recorder) = &mut self.timing { - recorder.add_wal_bytes(bytes); - } - } - - pub(crate) fn timing_add_database_bytes(&mut self, bytes: u64) { - if let Some(recorder) = &mut self.timing { - recorder.add_database_bytes(bytes); - } - } - - pub(crate) fn timing_add_ltx_bytes(&mut self, bytes: u64) { - if let Some(recorder) = &mut self.timing { - recorder.add_ltx_bytes(bytes); - } - } - - pub(crate) fn timing_add_segment(&mut self) { - if let Some(recorder) = &mut self.timing { - recorder.add_segment(); - } - } - - pub(crate) fn timing_add_phase_nanos(&mut self, phase: TimingPhase, elapsed: u64) { - if let Some(recorder) = &mut self.timing { - recorder.add_phase_nanos(phase, elapsed); - } - } - - pub(crate) fn timing_observe_wal_image(&mut self, sparse: bool, fallback: bool, bytes: usize) { - if let Some(recorder) = &mut self.timing { - recorder.observe_wal_image(sparse, fallback, bytes); - } - } - - pub(crate) fn timing_observe_wal_transfer(&mut self, file_bytes: u64, read_bytes: u64) { - if let Some(recorder) = &mut self.timing { - recorder.observe_wal_transfer(file_bytes, read_bytes); - } - } - - pub(crate) fn timing_observe_wal_snapshot(&mut self) { - if let Some(recorder) = &mut self.timing { - recorder.observe_wal_snapshot(); - } - } -} - -#[cfg(test)] -mod timing_tests { - use super::*; - - #[test] - fn recorder_uses_monotonic_instants_without_affecting_capture_state() { - let start = Instant::now(); - let mut recorder = TimingRecorder::new(start); - recorder.begin(TimingPhase::Preparation, start + Duration::from_millis(1)); - recorder.end(TimingPhase::Preparation, start + Duration::from_millis(3)); - recorder.begin(TimingPhase::WalRead, start + Duration::from_millis(4)); - recorder.end(TimingPhase::WalRead, start + Duration::from_millis(9)); - recorder.add_wal_bytes(11); - recorder.add_database_bytes(22); - recorder.add_ltx_bytes(33); - recorder.add_segment(); - - let timing = recorder.finish(start + Duration::from_millis(10)); - assert_eq!(timing.total_nanos, 10_000_000); - assert_eq!(timing.preparation_nanos, 2_000_000); - assert_eq!(timing.wal_read_nanos, 5_000_000); - assert_eq!(timing.wal_bytes, 11); - assert_eq!(timing.database_bytes, 22); - assert_eq!(timing.ltx_bytes, 33); - assert_eq!(timing.segment_count, 1); - } - - #[test] - fn recorder_closes_an_incomplete_phase_and_saturates_counters() { - let start = Instant::now(); - let mut recorder = TimingRecorder::new(start); - recorder.begin(TimingPhase::Encode, start); - recorder.add_wal_bytes(u64::MAX); - recorder.add_wal_bytes(1); - recorder.add_segment(); - recorder.add_segment(); - - let timing = recorder.finish(start + Duration::from_nanos(7)); - assert_eq!(timing.encode_nanos, 7); - assert_eq!(timing.wal_bytes, u64::MAX); - assert_eq!(timing.segment_count, 2); - } -} diff --git a/crates/crab-ltx/src/capture/verify.rs b/crates/crab-ltx/src/capture/verify.rs deleted file mode 100644 index b52940bbe..000000000 --- a/crates/crab-ltx/src/capture/verify.rs +++ /dev/null @@ -1,153 +0,0 @@ -// Derived from denoland/celld, commit 10cb1303dac710dcb3b557e318e08c855261f68b. -// Apache-2.0; see LICENSE and UPSTREAM.md. Modified by Crab contributors. -// Split from upstream db.rs; see UPSTREAM.md for Crab's changes. - -use super::*; - -impl CaptureEngine { - pub(super) fn verify(&mut self) -> Result { - let frame_size = self.page_size as i64 + WAL_FRAME_HEADER_SIZE as i64; - let mut info = SyncInfo { - snapshotting: true, - ..Default::default() - }; - - let pos = self.position; - if pos.txid == Txid(0) { - info.offset = WAL_HEADER_SIZE as i64; - return Ok(info); // first sync - } - - // Only this session's successfully written cut can seed continuity. - // Local directory listing must never promote unpublished state. - let (txid, hdr) = self - .last_l0_header - .as_ref() - .ok_or(CrabError::LTXCorrupted)?; - if *txid != pos.txid { - return Err(CrabError::LTXCorrupted); - } - let hdr = hdr.clone(); - info.offset = hdr.wal_offset + hdr.wal_size; - info.prev_commit = hdr.commit; - let Some((prior_salt1, prior_salt2)) = hdr.wal_salts else { - // An exact inherited root has no local WAL boundary yet. Capture - // creates/observes the first WAL under the session's read lock; - // its complete committed prefix continues the verified checksums. - let wal = self.wal_header_bytes()?; - info.salt1 = be_u32(&wal[16..]); - info.salt2 = be_u32(&wal[20..]); - info.snapshotting = false; - return Ok(info); - }; - info.salt1 = prior_salt1; - info.salt2 = prior_salt2; - - // If the LTX WAL offset exceeds the real WAL size, the WAL was truncated. - let wal_size = self.wal_file_size()?; - if info.offset > wal_size { - // If we previously synced to the exact WAL end, this truncation is an - // expected checkpoint: reset to the header and continue incrementally - // rather than snapshotting (issue #927, db.go:1335-1355). - if self.synced_to_wal_end { - self.synced_to_wal_end = false; - - let wal_hdr = self.wal_header_bytes()?; - info.offset = WAL_HEADER_SIZE as i64; - info.salt1 = be_u32(&wal_hdr[16..]); - info.salt2 = be_u32(&wal_hdr[20..]); - info.snapshotting = false; - return Ok(info); - } - - return Ok(info); - } - - // Compare WAL headers; restart from the beginning of the WAL if different. - let wal_hdr = self.wal_header_bytes()?; - let salt1 = be_u32(&wal_hdr[16..]); - let salt2 = be_u32(&wal_hdr[20..]); - let salt_match = salt1 == prior_salt1 && salt2 == prior_salt2; - - // Edge case: LTX represents the start of the WAL (WALOffset=32, WALSize=0). - // Handle this before computing prev_wal_offset to avoid underflow - // (32 - 4120 = -4088). See issue #900 (db.go:1375-1383). - if info.offset == WAL_HEADER_SIZE as i64 { - if salt_match { - info.snapshotting = false; - return Ok(info); - } - return Ok(info); - } - - // If the offset is at the start of the first page, we can't check the - // previous page (db.go:1386-1399). - let prev_wal_offset = info.offset - frame_size; - if prev_wal_offset == WAL_HEADER_SIZE as i64 { - if salt_match { - info.snapshotting = false; - return Ok(info); - } - return Ok(info); - } else if prev_wal_offset < WAL_HEADER_SIZE as i64 { - return Err(CrabError::Other( - format!("prev WAL offset is less than the header size: {prev_wal_offset}").into(), - )); - } - - // If we can't verify the last page is in the last LTX file, snapshot. - let last_page_match = self.last_page_match_cached(&hdr, prev_wal_offset, frame_size)?; - if !last_page_match { - return Ok(info); - } - - // Salt changed (possible FULL/RESTART checkpoint). With a last-page match - // we assume the WAL was not overwritten (db.go:1412-1431). - if !salt_match { - info.offset = WAL_HEADER_SIZE as i64; - info.salt1 = salt1; - info.salt2 = salt2; - - let detected = - self.detect_full_checkpoint(&[(salt1, salt2), (prior_salt1, prior_salt2)])?; - if detected { - } else { - info.snapshotting = false; - } - return Ok(info); - } - - info.snapshotting = false; - Ok(info) - } - - pub(super) fn last_page_match_cached( - &mut self, - hdr: &LastL0Header, - prev_wal_offset: i64, - frame_size: i64, - ) -> Result { - if prev_wal_offset <= WAL_HEADER_SIZE as i64 || hdr.final_page.is_empty() { - return Ok(false); - } - let frame = self.wal_bytes_at(prev_wal_offset, frame_size)?; - let pgno = be_u32(&frame[0..]); - let fsalt1 = be_u32(&frame[8..]); - let fsalt2 = be_u32(&frame[12..]); - let data = &frame[WAL_FRAME_HEADER_SIZE..]; - Ok(Some((fsalt1, fsalt2)) == hdr.wal_salts - && pgno == hdr.final_pgno - && data == hdr.final_page.as_slice()) - } - - pub(super) fn detect_full_checkpoint(&mut self, known_salts: &[(u32, u32)]) -> Result { - let wal_bytes = self.read_whole_wal()?; - let rd = WalReader::new(&wal_bytes).map_err(CrabError::from)?; - let last_known = known_salts.last().copied().unwrap_or((0, 0)); - let mut m = rd.frame_salts_until(last_known); - for s in known_salts { - m.remove(s); - } - Ok(!m.is_empty()) - } -} diff --git a/crates/crab-ltx/src/capture/wal.rs b/crates/crab-ltx/src/capture/wal.rs deleted file mode 100644 index 66383b596..000000000 --- a/crates/crab-ltx/src/capture/wal.rs +++ /dev/null @@ -1,729 +0,0 @@ -// Derived from denoland/celld, commit 10cb1303dac710dcb3b557e318e08c855261f68b. -// Apache-2.0; see LICENSE and UPSTREAM.md. Modified by Crab contributors. -// Split from upstream db.rs; see UPSTREAM.md for Crab's changes. - -use super::*; -use std::cell::Cell; -use std::sync::{ - Arc, - atomic::{AtomicU64, Ordering}, -}; - -const IN_MEMORY_INDEX_PAGE_LIMIT: usize = 64 << 10; - -// The captured index only carries bytes when the replica feature is on; keeping -// the type uniform lets the cut path stay single-sourced without binding unit. -type CapturedIndex = Option>; - -struct TimedWriter { - inner: W, - host: crate::Host, - write_nanos: Arc, - bytes_written: u64, - digest: blake3::Hasher, -} - -impl std::io::Write for TimedWriter { - fn write(&mut self, bytes: &[u8]) -> std::io::Result { - let started = self.host.now_monotonic(); - let result = self.inner.write(bytes); - add_elapsed(&self.write_nanos, started, self.host.now_monotonic()); - if let Ok(written) = result { - let written = written.min(bytes.len()); - self.bytes_written = self.bytes_written.saturating_add(written as u64); - self.digest.update(&bytes[..written]); - } - result - } - - fn flush(&mut self) -> std::io::Result<()> { - self.inner.flush() - } -} - -impl TimedWriter { - fn finish(self) -> (W, u64, [u8; 32]) { - ( - self.inner, - self.bytes_written, - *self.digest.finalize().as_bytes(), - ) - } -} - -struct TimedFileIo { - inner: Box, - host: crate::Host, - write_nanos: Arc, -} - -impl crate::environment::FileIo for TimedFileIo { - fn write_all(&mut self, bytes: &[u8]) -> std::io::Result<()> { - let started = self.host.now_monotonic(); - let result = self.inner.write_all(bytes); - add_elapsed(&self.write_nanos, started, self.host.now_monotonic()); - result - } - - fn write_all_at(&mut self, offset: u64, bytes: &[u8]) -> std::io::Result<()> { - let started = self.host.now_monotonic(); - let result = self.inner.write_all_at(offset, bytes); - add_elapsed(&self.write_nanos, started, self.host.now_monotonic()); - result - } - - fn read_exact_at(&mut self, offset: u64, len: usize) -> std::io::Result> { - self.inner.read_exact_at(offset, len) - } - - fn sync_all(&mut self) -> std::io::Result<()> { - self.inner.sync_all() - } - - fn file_len(&self) -> std::io::Result { - self.inner.file_len() - } - - fn set_len(&mut self, len: u64) -> std::io::Result<()> { - self.inner.set_len(len) - } -} - -fn add_elapsed(total: &AtomicU64, started: Instant, finished: Instant) { - let elapsed = nanos(finished.saturating_duration_since(started)); - let _ = total.fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { - Some(current.saturating_add(elapsed)) - }); -} - -impl CaptureEngine { - pub(super) fn read_valid_wal_image( - &mut self, - info: &SyncInfo, - start: usize, - ) -> Result { - let frame_size = self.page_size as usize + WAL_FRAME_HEADER_SIZE; - let offset = info.offset; - let salt1 = info.salt1; - let salt2 = info.salt2; - let read_bytes = Cell::new(0_u64); - let file_bytes = Cell::new(0_u64); - let result = self.with_wal_file(|file| { - let file_len = file.file_len()? as usize; - file_bytes.set(file_len as u64); - if file_len < WAL_HEADER_SIZE || start < WAL_HEADER_SIZE || start >= file_len { - return Err(std::io::Error::from(std::io::ErrorKind::UnexpectedEof)); - } - let complete_end = - WAL_HEADER_SIZE + ((file_len - WAL_HEADER_SIZE) / frame_size) * frame_size; - if start >= complete_end || !(start - WAL_HEADER_SIZE).is_multiple_of(frame_size) { - return Err(std::io::Error::from(std::io::ErrorKind::UnexpectedEof)); - } - - let tail_base = if start == WAL_HEADER_SIZE { 0 } else { start }; - let mut bytes = file.read_exact_at(0, WAL_HEADER_SIZE)?; - read_bytes.set(read_bytes.get().saturating_add(bytes.len() as u64)); - let mut cursor = start; - let mut target_frames = 1_usize; - loop { - let target_end = start - .saturating_add(target_frames.saturating_mul(frame_size)) - .min(complete_end); - if target_end > cursor { - let chunk = file.read_exact_at(cursor as u64, target_end - cursor)?; - read_bytes.set(read_bytes.get().saturating_add(chunk.len() as u64)); - bytes.extend_from_slice(&chunk); - cursor = target_end; - } - - let valid_end = { - let parsed = if offset == WAL_HEADER_SIZE as i64 { - WalReader::new(&bytes) - } else if tail_base == 0 { - WalReader::new_with_offset(&bytes, offset, salt1, salt2) - } else { - WalReader::new_with_offset_over_tail( - &bytes, - tail_base as i64, - offset, - salt1, - salt2, - ) - }; - let mut reader = parsed.map_err(|error| { - std::io::Error::new(std::io::ErrorKind::InvalidData, error) - })?; - reader.page_map().map_err(|error| { - std::io::Error::new(std::io::ErrorKind::InvalidData, error) - })?; - if reader.offset() == 0 { - WAL_HEADER_SIZE - } else { - reader.offset() as usize + frame_size - } - }; - - if valid_end < cursor { - let keep = if tail_base == 0 { - valid_end - } else { - WAL_HEADER_SIZE + valid_end.saturating_sub(tail_base) - }; - bytes.truncate(keep); - return Ok(WalImage { bytes, tail_base }); - } - if cursor == complete_end { - return Ok(WalImage { bytes, tail_base }); - } - target_frames = target_frames.saturating_mul(2); - } - }); - self.timing_observe_wal_transfer(file_bytes.get(), read_bytes.get()); - Ok(result?) - } - - pub(super) fn sync_inner(&mut self, mut info: SyncInfo) -> Result { - let frame_size_bytes = self.page_size as i64 + WAL_FRAME_HEADER_SIZE as i64; - // Decide the representation before reading the WAL. A delta whose - // worst-case encoded size exceeds the incremental bound is captured as - // a full database image instead of fencing the session, and a full - // image needs the whole WAL: pages that only an earlier, already - // captured WAL segment holds are not in the database file yet. The - // bound uses the uncaptured frame count, so it never understates the - // delta the encoder would produce. - let uncaptured_frames = if info.snapshotting { - 0 - } else { - let uncaptured_bytes = self.wal_file_size()?.saturating_sub(info.offset).max(0) as u64; - uncaptured_bytes.div_ceil(frame_size_bytes.max(1) as u64) - }; - let full_image = info.snapshotting - || ltx::cut_upper_bound(self.page_size, uncaptured_frames)? - > self.max_incremental_bytes; - if full_image && !info.snapshotting { - // A full image is anchored at the WAL header so every frame the - // current database state still depends on is in the page map. - info.offset = WAL_HEADER_SIZE as i64; - } - // A capture that starts at the WAL header reads a logical WAL with no - // backfilled prefix: the first sync, a restart, or a boundary image. - // The checkpoint trigger counts from the backfilled boundary, so it - // must not carry an offset from the WAL that just ended, or the new - // WAL would grow past that offset before its first checkpoint. - if info.offset == WAL_HEADER_SIZE as i64 { - self.checkpointed_wal_offset = WAL_HEADER_SIZE as i64; - } - self.timing_begin(crate::capture::TimingPhase::WalRead); - let pos = self.position; - let tx_id = Txid(pos.txid.0.checked_add(1).ok_or(CrabError::TxNotAvailable)?); - let filename = self.ltx_path(0, tx_id, tx_id); - - let db_size = self.db_file_size()?; - let mut commit = (db_size / self.page_size as i64) as u32; - self.last_db_pages = commit; - - // The incremental path reads only the valid checksum chain: the - // 32-byte WAL header, the previous frame when there is one, and the - // frames from `info.offset` on. SQLite can retain a large stale physical - // suffix after a logical restart. Stop at the valid prefix instead of - // repeatedly reading that suffix; a sparse-tail mismatch needs a full - // re-read so a zero-filled prefix cannot hide uncaptured commits. - let mut sparse_tail = false; - let mut fallback = false; - if info.snapshotting { - self.timing_observe_wal_snapshot(); - } - let mut wal = if info.snapshotting || full_image { - let bytes = self.read_whole_wal()?; - WalImage::whole(bytes) - } else { - let start = if info.offset <= WAL_HEADER_SIZE as i64 + frame_size_bytes { - WAL_HEADER_SIZE - } else { - (info.offset - frame_size_bytes) as usize - }; - match self.read_valid_wal_image(&info, start) { - Ok(image) => { - sparse_tail = start != WAL_HEADER_SIZE; - image - } - Err(_) => { - fallback = true; - let bytes = self.read_whole_wal()?; - WalImage::whole(bytes) - } - } - }; - - // Choose the WAL reader start: from the header, or seek to info.offset. - // A previous-frame mismatch falls back to a full read (snapshot), - // mirroring NewWALReaderWithOffset's PrevFrameMismatchError handling - // (db.go:1565-1581). - // A previous-frame mismatch restarts the read from the header. The - // sparse tail image is zero-filled below `start`, so a from-header - // reader over it would see no valid frame and report "nothing to - // capture" — a silent miss the ship loop would then credit. The - // mismatch path therefore re-reads the complete WAL first, which is - // exactly the port's former full-read behavior on this branch. - let mismatch = !(info.offset == WAL_HEADER_SIZE as i64) - && matches!( - wal.reader_at(info.offset, info.salt1, info.salt2), - Err(crate::wal::WalError::PrevFrameMismatch) - ); - if mismatch { - info.offset = WAL_HEADER_SIZE as i64; - if sparse_tail { - fallback = true; - let bytes = self.read_whole_wal()?; - wal = WalImage::whole(bytes); - } - } - self.timing_observe_wal_image(sparse_tail && !fallback, fallback, wal.bytes.capacity()); - let mut rd = if info.offset == WAL_HEADER_SIZE as i64 { - WalReader::new(&wal.bytes).map_err(CrabError::from)? - } else { - wal.reader_at(info.offset, info.salt1, info.salt2) - .map_err(CrabError::from)? - }; - - self.timing_end(crate::capture::TimingPhase::WalRead); - self.timing_begin(crate::capture::TimingPhase::PageCollection); - let page_map_result = rd.page_map().map_err(CrabError::from); - self.timing_end(crate::capture::TimingPhase::PageCollection); - let (page_map, max_offset, wal_commit) = page_map_result?; - if wal_commit > 0 { - commit = wal_commit; - } - - let sz = if max_offset > 0 { - max_offset - info.offset - } else { - 0 - }; - if sz < 0 { - return Err(CrabError::Other( - format!( - "wal size must be positive: sz={sz}, maxOffset={max_offset}, info.offset={}", - info.offset - ) - .into(), - )); - } - - // Exit if there are no new WAL pages and we are not snapshotting - // (db.go:1603-1607). - if !info.snapshotting && sz == 0 { - return Ok(false); - } - - self.timing_add_wal_bytes(u64::try_from(sz).unwrap_or_default()); - self.timing_add_database_bytes(u64::from(commit).saturating_mul(u64::from(self.page_size))); - - let (rd_salt1, rd_salt2) = rd.salt(); - - // Build the page stream for the encoder. - self.host - .check_database_size(u64::from(commit) * u64::from(self.page_size))?; - // Admit the selected representation against its bound. The full image - // keeps the current TXID, pre-apply checksum, and chain position, so it - // stays a valid successor cut of the same lineage. - let encoded_pages = if full_image { - commit as usize - } else { - let lock = lock_pgno(self.page_size); - let growth = commit.saturating_sub(info.prev_commit) as usize; - let written_new_pages = page_map - .keys() - .filter(|&&pgno| pgno > info.prev_commit && pgno <= commit) - .count(); - let missing_lock_page = usize::from( - lock > info.prev_commit && lock <= commit && !page_map.contains_key(&lock), - ); - // WAL pages in the growth range are already in the map. - let missing_new_pages = growth - .saturating_sub(written_new_pages) - .saturating_sub(missing_lock_page); - page_map.len().saturating_add(missing_new_pages) - }; - let cut_limit = if full_image { - self.host.max_file_bytes - } else { - self.max_incremental_bytes - }; - if ltx::cut_upper_bound(self.page_size, encoded_pages as u64)? > cut_limit { - return Err(CrabError::Limit(crate::LimitKind::LtxFileBytes)); - } - let header = ltx::Header { - version: ltx::VERSION, - flags: 0, - page_size: self.page_size, - commit, - min_txid: tx_id, - max_txid: tx_id, - timestamp: self.host.now_unix_millis(), - pre_apply_checksum: pos.post_apply_checksum, - wal_offset: info.offset, - wal_size: sz, - wal_salt1: rd_salt1, - wal_salt2: rd_salt2, - node_id: 0, - }; - - // Atomic tmp → fsync → rename (db.go:1609-1685). - let tmp_filename = format!("{filename}.tmp"); - let index_filename = format!("{filename}.index.tmp"); - let parent = Path::new(&tmp_filename).parent().map(Path::to_path_buf); - if !self.l0_dir_ready { - if let Some(parent) = &parent { - self.host.create_dir_all(parent)?; - } - self.l0_dir_ready = true; - self.l0_ancestors_durable = false; - } - // A directory that vanished under a ready flag is recreated once and - // the complete cut is retried. The candidate checksum index remains - // isolated until the output has been synced and renamed. - let write_result = self.write_streamed_cut( - &tmp_filename, - &index_filename, - &filename, - header, - &wal, - &page_map, - full_image, - cut_limit, - encoded_pages, - info.prev_commit, - commit, - !self.defer_durability, - ); - let (checksums, size_bytes, digest, captured_index) = match write_result { - Err(CrabError::Io(error)) if error.kind() == std::io::ErrorKind::NotFound => { - if let Some(parent) = &parent { - self.host.create_dir_all(parent)?; - } - self.l0_ancestors_durable = false; - self.write_streamed_cut( - &tmp_filename, - &index_filename, - &filename, - header, - &wal, - &page_map, - full_image, - cut_limit, - encoded_pages, - info.prev_commit, - commit, - !self.defer_durability, - )? - } - other => other?, - }; - if !self.defer_durability { - // The first acknowledged cut also needs the newly created path to survive. - self.sync_l0_ancestors()?; - } - let post_checksum = checksums.checksum(); - // The candidate stays isolated until the cut is sealed. Retiring the - // predecessor lets the owner reuse its memory base; a failed sidecar - // update fences Db before partially updated state can be used again. - self.checksums.commit(checksums)?; - // The next verify reads exactly these fields back; caching them — - // plus the final consumed WAL frame for the page check — is what - // spares it re-reading the file it just watched being written. - let frame_size = self.page_size as i64 + WAL_FRAME_HEADER_SIZE as i64; - let final_frame_offset = max_offset - frame_size; - let (final_pgno, final_page) = match (final_frame_offset >= WAL_HEADER_SIZE as i64) - .then(|| wal.slice(final_frame_offset, frame_size as usize)) - .flatten() - { - Some(frame) => (be_u32(&frame[0..]), frame[WAL_FRAME_HEADER_SIZE..].to_vec()), - None => (0, Vec::new()), - }; - self.last_l0_header = Some(( - tx_id, - LastL0Header { - wal_offset: info.offset, - wal_size: sz, - wal_salts: Some((rd_salt1, rd_salt2)), - commit, - final_pgno, - final_page, - }, - )); - // Checkpointing can seal another cut before Db collects this one. - // Retain each writer-produced digest so collection need not reread it. - self.sealed_l0_segments.insert( - tx_id.0, - crate::SegmentInfo { - min_txid: tx_id.0, - max_txid: tx_id.0, - page_size: self.page_size, - database_pages: commit, - pre_checksum: pos.post_apply_checksum, - post_checksum, - size_bytes, - blake3: digest, - }, - ); - #[cfg(feature = "replica")] - if let Some(index) = captured_index { - self.sealed_l0_captured_indexes.insert(tx_id.0, index); - } - #[cfg(not(feature = "replica"))] - let _ = captured_index; - - // Advance only after both the sealed cut and checksum merge succeed. - self.position = Pos::new(tx_id, post_checksum); - - // Track the logical end of WAL content for checkpoint decisions - // (db.go:1704-1718, issues #997/#927). - let final_offset = info.offset + sz; - self.last_synced_wal_offset = final_offset; - self.synced_to_wal_end = match self.wal_file_size() { - Ok(wal_size) => final_offset == wal_size, - Err(_) => false, - }; - - Ok(true) - } - - #[expect(clippy::too_many_arguments)] - fn write_streamed_cut( - &mut self, - tmp_filename: &str, - index_filename: &str, - filename: &str, - header: ltx::Header, - wal: &WalImage, - page_map: &HashMap, - full_image: bool, - limit: u64, - estimated_pages: usize, - prev_commit: u32, - commit: u32, - durable: bool, - ) -> Result<(crate::pages::PageChecksums, u64, [u8; 32], CapturedIndex)> { - #[cfg(feature = "replica")] - let captured_index_budget = RETAINED_CAPTURE_INDEX_BYTES.saturating_sub( - self.sealed_l0_captured_indexes - .values() - .fold(0_usize, |total, index| total.saturating_add(index.len())), - ); - let result = (|| -> Result<(crate::pages::PageChecksums, u64, [u8; 32], CapturedIndex)> { - // The cut is written through a host limited by the representation's - // bound, so an encoder that ever exceeded its admitted bound fails - // instead of publishing an oversized artifact. - let output_host = crate::LtxHost { - facilities: self.host.facilities.clone(), - max_database_bytes: self.host.max_database_bytes, - max_file_bytes: limit, - }; - let output = output_host.create(Path::new(tmp_filename))?; - let spool_index = estimated_pages > IN_MEMORY_INDEX_PAGE_LIMIT; - let index = if spool_index { - let index = self - .host - .facilities - .filesystem - .create(Path::new(index_filename))?; - drop(index); - Some( - self.host - .facilities - .filesystem - .open_rw(Path::new(index_filename))?, - ) - } else { - None - }; - let write_nanos = Arc::new(AtomicU64::new(0)); - let output = TimedWriter { - inner: output, - host: self.host.facilities.clone(), - write_nanos: Arc::clone(&write_nanos), - bytes_written: 0, - digest: blake3::Hasher::new(), - }; - let output = std::io::BufWriter::with_capacity(64 << 10, output); - let index = index.map(|index| { - Box::new(TimedFileIo { - inner: index, - host: self.host.facilities.clone(), - write_nanos: Arc::clone(&write_nanos), - }) as Box - }); - let mut encoder = crate::codec::Encoder::new_block_with_index(output, index); - #[cfg(feature = "replica")] - let mut captured_index = Some(Vec::with_capacity( - estimated_pages - .saturating_mul(crate::paged::ENTRY_BYTES) - .min(captured_index_budget), - )); - let encode_started = self.host.now_monotonic(); - encoder.encode_header(header)?; - - let mut checksums = self.checksums.clone(); - if full_image { - let lock = lock_pgno(self.page_size); - let pages = (1..=commit).filter(|page| *page != lock).map(|pgno| { - let data = self.capture_page(wal, page_map, pgno)?; - let encoded = encoder.encode_page(ltx::PageHeader { pgno, flags: 0 }, &data)?; - #[cfg(feature = "replica")] - retain_encoded_page(&mut captured_index, &encoded, captured_index_budget)?; - #[cfg(not(feature = "replica"))] - let _ = encoded; - Ok((pgno, data)) - }); - checksums.apply_iter( - self.page_size, - commit, - pages, - self.host.max_database_bytes, - )?; - } else { - let pgnos = self.wal_page_numbers(page_map, prev_commit, commit); - let pages = pgnos.into_iter().map(|pgno| { - let data = self.capture_page(wal, page_map, pgno)?; - let encoded = encoder.encode_page(ltx::PageHeader { pgno, flags: 0 }, &data)?; - #[cfg(feature = "replica")] - retain_encoded_page(&mut captured_index, &encoded, captured_index_budget)?; - #[cfg(not(feature = "replica"))] - let _ = encoded; - Ok((pgno, data)) - }); - checksums.apply_iter( - self.page_size, - commit, - pages, - self.host.max_database_bytes, - )?; - } - encoder.close(checksums.checksum())?; - #[cfg(not(feature = "replica"))] - let captured_index = None; - let output = encoder - .into_writer() - .into_inner() - .map_err(|error| error.into_error())?; - let encode_elapsed = nanos( - self.host - .now_monotonic() - .saturating_duration_since(encode_started), - ); - let local_write_nanos = write_nanos.load(Ordering::Relaxed); - self.timing_add_phase_nanos( - crate::capture::TimingPhase::Encode, - encode_elapsed.saturating_sub(local_write_nanos), - ); - self.timing_add_phase_nanos(crate::capture::TimingPhase::LocalWrite, local_write_nanos); - let (mut output, size_bytes, digest) = output.finish(); - if durable { - self.timing_begin(crate::capture::TimingPhase::Fsync); - output.sync_all()?; - self.timing_end(crate::capture::TimingPhase::Fsync); - } - drop(output); - if spool_index { - self.host.remove_file(Path::new(index_filename))?; - } - self.timing_begin(crate::capture::TimingPhase::ParentSync); - if durable { - self.host - .rename(Path::new(tmp_filename), Path::new(filename))?; - } else { - self.host - .rename_uncommitted(Path::new(tmp_filename), Path::new(filename))?; - } - self.timing_end(crate::capture::TimingPhase::ParentSync); - Ok((checksums, size_bytes, digest, captured_index)) - })(); - if result.is_err() { - let _ = self.host.remove_file(Path::new(tmp_filename)); - let _ = self.host.remove_file(Path::new(index_filename)); - } - result - } - - fn wal_page_numbers( - &self, - page_map: &HashMap, - prev_commit: u32, - commit: u32, - ) -> Vec { - let mut pgnos: Vec = page_map.keys().copied().collect(); - let lock = lock_pgno(self.page_size); - if commit > prev_commit { - for pgno in (prev_commit + 1)..=commit { - if pgno != lock && !page_map.contains_key(&pgno) { - pgnos.push(pgno); - } - } - } - pgnos.sort_unstable(); - pgnos - } - - pub(super) fn capture_page( - &self, - wal: &WalImage, - page_map: &HashMap, - pgno: u32, - ) -> Result> { - match page_map.get(&pgno) { - Some(&offset) => wal.page(offset, self.page_size), - None => self.read_db_page(pgno), - } - } - - pub(super) fn read_db_page(&self, pgno: u32) -> Result> { - let offset = u64::from(pgno.checked_sub(1).ok_or(CrabError::LTXCorrupted)?) - * u64::from(self.page_size); - // Read through SQLite's file, below its WAL-aware pager. A sparse VFS - // must hydrate holes here too, not only on application SQL reads. - crate::db::read_main(&self.conn, offset, self.page_size as usize) - } -} - -#[cfg(feature = "replica")] -fn retain_encoded_page( - index: &mut Option>, - page: &crate::codec::EncodedPage, - budget: usize, -) -> Result<()> { - let Some(bytes) = index else { - return Ok(()); - }; - if crate::paged::ENTRY_BYTES > budget.saturating_sub(bytes.len()) { - *index = None; - return Ok(()); - } - crate::paged::append_index_page(bytes, page) -} - -#[cfg(all(test, feature = "replica"))] -mod tests { - use super::*; - - fn encoded_page(page: u32) -> crate::codec::EncodedPage { - crate::codec::EncodedPage { - page, - offset: u64::from(page) * 4096, - size: 4096, - frame_hash: [page as u8; 32], - checksum: u64::from(page), - } - } - - #[test] - fn captured_index_falls_back_when_page_would_exceed_budget() { - let mut index = Some(Vec::new()); - - retain_encoded_page(&mut index, &encoded_page(1), crate::paged::ENTRY_BYTES).unwrap(); - assert_eq!(index.as_ref().unwrap().len(), crate::paged::ENTRY_BYTES); - - retain_encoded_page(&mut index, &encoded_page(2), crate::paged::ENTRY_BYTES).unwrap(); - assert!(index.is_none()); - } -} diff --git a/crates/crab-ltx/src/cell_layout.rs b/crates/crab-ltx/src/cell_layout.rs deleted file mode 100644 index 68021ad89..000000000 --- a/crates/crab-ltx/src/cell_layout.rs +++ /dev/null @@ -1,358 +0,0 @@ -use object_store::path::Path; - -use crab_storage::Store; - -use crate::hex::encode_hex; - -/// Typed physical paths for one application's SQLite Cell objects. -#[derive(Clone)] -pub struct CellStorageLayout { - store: Store, - root: Path, - application: [u8; 16], -} - -/// Immutable object kinds accepted below one Cell incarnation. -#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] -pub enum CellObjectKind { - /// Immutable LTX segment. - Ltx, - /// Capture index beside one segment. - Index, - /// Root directory page. - Directory, - /// Published root record. - Root, - /// Recovery bundle. - Bundle, -} - -impl CellObjectKind { - const fn extension(self) -> &'static str { - match self { - Self::Ltx => "ltx", - Self::Index => "index", - Self::Directory => "dir", - Self::Root => "root", - Self::Bundle => "bundle", - } - } -} - -impl CellStorageLayout { - /// Binds Cell paths to an already validated authoritative storage prefix. - #[must_use] - pub fn new(store: Store, root: Path, application: [u8; 16]) -> Self { - Self { - store, - root, - application, - } - } - - /// Returns the transport every path in this layout writes through. - #[must_use] - pub fn store(&self) -> &Store { - &self.store - } - - /// Returns the application ID every application-scoped path is bound to. - #[must_use] - pub const fn application_id(&self) -> &[u8; 16] { - &self.application - } - - /// Returns the process-local identity used to isolate immutable read caches. - #[must_use] - pub fn immutable_cache_identity(&self) -> u64 { - self.store.immutable_cache_identity() - } - - /// Returns the application identity path. - #[must_use] - pub fn identity_path(&self) -> Path { - Self::root_identity_path(&self.root) - } - - /// Returns the application identity path before an application ID is known. - #[must_use] - pub fn root_identity_path(root: &Path) -> Path { - Path::from(format!("{root}/cells/v1/identity.json")) - } - - /// Returns the current release pointer path. - #[must_use] - pub fn release_path(&self) -> Path { - self.application_path("release.json") - } - - /// Returns the application-scoped prefix every Cell path shares. - #[must_use] - pub fn application_prefix(&self) -> Path { - self.application_path("") - } - - /// Returns the catalog pin prefix. - #[must_use] - pub fn pin_prefix(&self) -> Path { - self.application_path("pins") - } - - /// Returns the release descriptor path for one digest. - #[must_use] - pub fn release_descriptor_path(&self, digest: &[u8; 32]) -> Path { - self.application_path(&format!("releases/{}.json", encode_hex(digest))) - } - - /// Returns the control record path for one Cell. - #[must_use] - pub fn control_path(&self, cell: &[u8; 32]) -> Path { - self.application_path(&format!("cells/{}/control.json", encode_hex(cell))) - } - - /// Returns the advisory desired read-replica count for one Cell. - #[must_use] - pub fn read_policy_path(&self, cell: &[u8; 32]) -> Path { - self.application_path(&format!("cells/{}/read-policy.json", encode_hex(cell))) - } - - /// Returns the immutable object path for one Cell incarnation. - #[must_use] - pub fn incarnation_object_path( - &self, - cell: &[u8; 32], - incarnation: &[u8; 16], - digest: &[u8; 32], - kind: CellObjectKind, - ) -> Path { - self.application_path(&format!( - "cells/{}/inc/{}/objects/{}.{}", - encode_hex(cell), - encode_hex(incarnation), - encode_hex(digest), - kind.extension() - )) - } - - /// Returns a private, unreferenced staging key for an immutable Cell object. - /// - /// Staging keys are never part of a root or manifest. The digest makes - /// retries and failover converge on one unreferenced target for the same - /// immutable bytes; callers must promote the object and delete this key - /// before returning. - #[must_use] - pub fn incarnation_staging_path( - &self, - cell: &[u8; 32], - incarnation: &[u8; 16], - digest: &[u8; 32], - kind: CellObjectKind, - ) -> Path { - self.application_path(&format!( - "cells/{}/inc/{}/objects/.staging/{}.{}", - encode_hex(cell), - encode_hex(incarnation), - encode_hex(digest), - kind.extension() - )) - } - - /// Returns the prefix containing all tenant catalog heads. - #[must_use] - pub fn catalog_tenants_prefix(&self) -> Path { - self.application_path("catalog/tenants") - } - - /// Returns the catalog head path for one tenant and shard. - #[must_use] - pub fn catalog_head_path(&self, tenant: &[u8; 16], shard: u8) -> Path { - Path::from(format!( - "{}/{}/{shard:02x}/head.json", - self.catalog_tenants_prefix(), - encode_hex(tenant) - )) - } - - /// Returns the due-hint prefix for one minute bucket. - /// - /// A hint is an accelerator, never authority: the Cell's own control and - /// SQLite state decide what is due, and the full catalog scan remains the - /// backstop when a hint is missing. - #[must_use] - pub fn due_hint_prefix(&self, bucket: u64) -> Path { - self.application_path(&format!("due/{bucket:016x}/")) - } - - /// Returns the due-hint path for one Cell in one minute bucket. - #[must_use] - pub fn due_hint_path(&self, bucket: u64, cell: &[u8; 32]) -> Path { - self.application_path(&format!("due/{bucket:016x}/{}.json", encode_hex(cell))) - } - - /// Returns the catalog page object path for one digest. - #[must_use] - pub fn catalog_object_path(&self, digest: &[u8; 32]) -> Path { - self.application_path(&format!("catalog/objects/{}.json", encode_hex(digest))) - } - - /// Returns the catalog pin record path for one pin. - #[must_use] - pub fn pin_path(&self, pin: &[u8; 16]) -> Path { - self.application_path(&format!("pins/{}.json", encode_hex(pin))) - } - - /// Returns the catalog pin object path for one digest. - #[must_use] - pub fn pin_object_path(&self, digest: &[u8; 32]) -> Path { - self.application_path(&format!("pins/objects/{}.json", encode_hex(digest))) - } - - /// Returns the release-migration progress path for one Cell operation. - #[must_use] - pub fn migration_path(&self, cell: &[u8; 32], operation: &[u8; 16], suffix: &str) -> Path { - self.application_path(&format!( - "cells/{}/migration/{}/{}", - encode_hex(cell), - encode_hex(operation), - suffix - )) - } - - /// Returns the node advertisement path for one session. - #[must_use] - pub fn node_path(&self, session: &[u8; 16]) -> Path { - Path::from(format!( - "{}/cells/v1/nodes/{}.json", - self.root, - encode_hex(session) - )) - } - - /// Returns the node directory prefix. - #[must_use] - pub fn node_directory_path(&self) -> Path { - Path::from(format!("{}/cells/v1/nodes", self.root)) - } - - /// Immutable recovered follower bundle outside any one application prefix. - #[must_use] - pub fn node_log_bundle_path(&self, leader: &[u8; 16], epoch: u64, digest: &[u8; 32]) -> Path { - Path::from(format!( - "{}/cells/v1/node-logs/{}/{epoch}/bundles/{}.bundle", - self.root, - encode_hex(leader), - encode_hex(digest) - )) - } - - /// Content-addressed manifest that pins every Cell tail recovered together. - #[must_use] - pub fn node_log_recovery_path(&self, leader: &[u8; 16], epoch: u64, digest: &[u8; 32]) -> Path { - Path::from(format!( - "{}/cells/v1/node-logs/{}/{epoch}/recovery/{}.json", - self.root, - encode_hex(leader), - encode_hex(digest) - )) - } - - fn application_path(&self, suffix: &str) -> Path { - Path::from(format!( - "{}/cells/v1/apps/{}/{}", - self.root, - encode_hex(&self.application), - suffix - )) - } -} - -#[cfg(test)] -mod tests { - use std::sync::Arc; - - use object_store::memory::InMemory; - - use super::*; - - #[test] - fn cell_paths_are_fixed_width_and_scoped_to_application() { - let layout = CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("tenant-root"), - [0xab; 16], - ); - assert_eq!( - layout.control_path(&[0xcd; 32]).as_ref(), - "tenant-root/cells/v1/apps/abababababababababababababababab/cells/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd/control.json" - ); - assert_eq!( - layout.node_directory_path().as_ref(), - "tenant-root/cells/v1/nodes" - ); - assert_eq!( - layout - .node_log_bundle_path(&[0xdd; 16], 7, &[0xef; 32]) - .as_ref(), - "tenant-root/cells/v1/node-logs/dddddddddddddddddddddddddddddddd/7/bundles/efefefefefefefefefefefefefefefefefefefefefefefefefefefefefefefef.bundle" - ); - assert_eq!( - layout - .node_log_recovery_path(&[0xdd; 16], 7, &[0xef; 32]) - .as_ref(), - "tenant-root/cells/v1/node-logs/dddddddddddddddddddddddddddddddd/7/recovery/efefefefefefefefefefefefefefefefefefefefefefefefefefefefefefefef.json" - ); - assert_eq!( - layout - .migration_path(&[0xcd; 32], &[0xee; 16], "source.json") - .as_ref(), - "tenant-root/cells/v1/apps/abababababababababababababababab/cells/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd/migration/eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee/source.json" - ); - assert_eq!( - layout - .incarnation_object_path(&[0xcd; 32], &[0xef; 16], &[1; 32], CellObjectKind::Root,) - .as_ref(), - "tenant-root/cells/v1/apps/abababababababababababababababab/cells/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd/inc/efefefefefefefefefefefefefefefef/objects/0101010101010101010101010101010101010101010101010101010101010101.root" - ); - assert_eq!( - layout - .incarnation_staging_path( - &[0xcd; 32], - &[0xef; 16], - &[7; 32], - CellObjectKind::Bundle, - ) - .as_ref(), - "tenant-root/cells/v1/apps/abababababababababababababababab/cells/cdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcdcd/inc/efefefefefefefefefefefefefefefef/objects/.staging/0707070707070707070707070707070707070707070707070707070707070707.bundle" - ); - assert_eq!( - layout.pin_object_path(&[2; 32]).as_ref(), - "tenant-root/cells/v1/apps/abababababababababababababababab/pins/objects/0202020202020202020202020202020202020202020202020202020202020202.json" - ); - assert_eq!( - layout.application_prefix().as_ref(), - "tenant-root/cells/v1/apps/abababababababababababababababab" - ); - assert_eq!( - layout.pin_prefix().as_ref(), - "tenant-root/cells/v1/apps/abababababababababababababababab/pins" - ); - } - - #[test] - fn immutable_cache_identity_is_shared_only_by_store_clones() { - let inner = Arc::new(InMemory::new()); - let first = - CellStorageLayout::new(Store::new(inner.clone()), Path::from("same-root"), [1; 16]); - let clone = first.clone(); - let independent = - CellStorageLayout::new(Store::new(inner), Path::from("same-root"), [1; 16]); - assert_eq!( - first.immutable_cache_identity(), - clone.immutable_cache_identity() - ); - assert_ne!( - first.immutable_cache_identity(), - independent.immutable_cache_identity() - ); - } -} diff --git a/crates/crab-ltx/src/codec.rs b/crates/crab-ltx/src/codec.rs deleted file mode 100644 index 83bf089de..000000000 --- a/crates/crab-ltx/src/codec.rs +++ /dev/null @@ -1,697 +0,0 @@ -// Derived from denoland/celld, commit 10cb1303dac710dcb3b557e318e08c855261f68b. -// Apache-2.0; see LICENSE and UPSTREAM.md. Modified by Crab contributors. - -use crate::CHECKSUM_FLAG; -use crate::environment::FileIo; -use crate::error::{CrabError, Result}; -use crate::ltx::{ - CHECKSUM_SIZE, Crc64, HEADER_SIZE, Header, PAGE_HEADER_FLAG_SIZE, PAGE_HEADER_SIZE, PageHeader, - TRAILER_SIZE, Trailer, checksum_page, lock_pgno, -}; -use std::io::{BufReader, Read, Write}; - -const INDEX_COPY_BYTES: usize = 64 << 10; - -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -enum DecoderState { - Header, - Pages, - Close, - Closed, -} - -pub(crate) struct Decoder { - reader: CountingReader, - index: Vec<(u32, u64, u64)>, - compressed: Vec, - state: DecoderState, - pub(crate) header: Header, - pub(crate) trailer: Trailer, - hash: Crc64, - rolling_checksum: u64, - #[cfg(feature = "replica")] - replica_index: Option>, -} - -impl Decoder { - pub(crate) fn new(reader: R) -> Self { - Self { - reader: CountingReader { - inner: reader, - bytes: 0, - digest: blake3::Hasher::new(), - }, - index: Vec::new(), - compressed: Vec::new(), - state: DecoderState::Header, - header: Header::default(), - trailer: Trailer::default(), - hash: Crc64::new(), - rolling_checksum: 0, - #[cfg(feature = "replica")] - replica_index: None, - } - } - - #[cfg(feature = "replica")] - pub(crate) fn new_with_index(reader: R) -> Self { - Self { - replica_index: Some(Vec::new()), - ..Self::new(reader) - } - } - - pub(crate) fn decode_header(&mut self) -> Result<()> { - if self.state != DecoderState::Header { - return Err(CrabError::LTXCorrupted); - } - let mut bytes = [0; HEADER_SIZE]; - self.reader.read_exact(&mut bytes)?; - self.header = Header::parse(&bytes)?; - self.header.validate()?; - self.hash.update(&bytes); - if !self.header.no_checksum() { - self.rolling_checksum = CHECKSUM_FLAG; - } - self.state = DecoderState::Pages; - Ok(()) - } - - pub(crate) fn decode_page(&mut self, data: &mut [u8]) -> Result> { - if self.state == DecoderState::Close { - return Ok(None); - } - if self.state != DecoderState::Pages || data.len() != self.header.page_size as usize { - return Err(CrabError::LTXCorrupted); - } - - let offset = self.reader.bytes; - let mut header_bytes = [0; PAGE_HEADER_SIZE]; - self.reader.read_exact(&mut header_bytes)?; - let page = PageHeader::parse(&header_bytes)?; - self.hash.update(&header_bytes); - if page.is_zero() { - self.state = DecoderState::Close; - return Ok(None); - } - page.validate()?; - let previous = self.index.last().map(|entry| entry.0).unwrap_or(0); - let lock = lock_pgno(self.header.page_size); - let expected = previous + if previous == lock - 1 { 2 } else { 1 }; - if page.pgno <= previous - || page.pgno > self.header.commit - || page.pgno == lock - || (self.header.is_snapshot() && page.pgno != expected) - { - return Err(CrabError::LTXCorrupted); - } - - if page.flags & PAGE_HEADER_FLAG_SIZE != 0 { - let mut size_bytes = [0; 4]; - self.reader.read_exact(&mut size_bytes)?; - self.hash.update(&size_bytes); - let compressed_size = u32::from_be_bytes(size_bytes) as usize; - if compressed_size > crate::lz4_block::compress_bound(data.len()) { - return Err(CrabError::LTXCorrupted); - } - self.compressed.resize(compressed_size, 0); - self.reader.read_exact(&mut self.compressed)?; - let n = lz4_flex::block::decompress_into(&self.compressed, data) - .map_err(|error| CrabError::Other(Box::new(error)))?; - if n != data.len() { - return Err(CrabError::LTXCorrupted); - } - #[cfg(feature = "replica")] - if let Some(index) = &mut self.replica_index { - let mut frame_hash = blake3::Hasher::new(); - frame_hash.update(&header_bytes); - frame_hash.update(&size_bytes); - frame_hash.update(&self.compressed); - index.push(EncodedPage { - page: page.pgno, - offset, - size: self.reader.bytes - offset, - frame_hash: *frame_hash.finalize().as_bytes(), - checksum: checksum_page(page.pgno, data), - }); - } - } else { - let mut decoder = lz4_flex::frame::FrameDecoder::new(&mut self.reader); - decoder - .read_exact(data) - .map_err(|error| CrabError::Other(Box::new(error)))?; - let mut extra = [0; 1]; - if decoder - .read(&mut extra) - .map_err(|error| CrabError::Other(Box::new(error)))? - != 0 - { - return Err(CrabError::LTXCorrupted); - } - #[cfg(feature = "replica")] - if let Some(index) = &mut self.replica_index { - index.push(EncodedPage { - page: page.pgno, - offset, - size: self.reader.bytes - offset, - frame_hash: [0; 32], - checksum: checksum_page(page.pgno, data), - }); - } - } - - self.index - .push((page.pgno, offset, self.reader.bytes - offset)); - self.hash.update(data); - if self.header.is_snapshot() - && !self.header.no_checksum() - && page.pgno != lock_pgno(self.header.page_size) - { - self.rolling_checksum = - CHECKSUM_FLAG | (self.rolling_checksum ^ checksum_page(page.pgno, data)); - } - Ok(Some(page)) - } - - pub(crate) fn close(&mut self) -> Result<()> { - if self.state == DecoderState::Closed { - return Ok(()); - } - if self.state != DecoderState::Close { - return Err(CrabError::LTXCorrupted); - } - - // Buffer footer reads so varints do not cause one filesystem call per - // byte. Count consumed encodings separately from the reader's readahead. - let mut footer = BufReader::with_capacity(INDEX_COPY_BYTES, &mut self.reader); - let mut index_size = 0; - for &(page, offset, size) in &self.index { - for expected in [u64::from(page), offset, size] { - let (actual, bytes) = read_uvarint(&mut footer, &mut self.hash)?; - if actual != expected { - return Err(CrabError::LTXCorrupted); - } - index_size += bytes; - } - } - let (end, bytes) = read_uvarint(&mut footer, &mut self.hash)?; - if end != 0 { - return Err(CrabError::LTXCorrupted); - } - index_size += bytes; - let mut size_bytes = [0; 8]; - read_footer_exact(&mut footer, &mut size_bytes)?; - if u64::from_be_bytes(size_bytes) != index_size { - return Err(CrabError::LTXCorrupted); - } - self.hash.update(&size_bytes); - if self.header.is_snapshot() { - let last = if self.header.commit == lock_pgno(self.header.page_size) { - self.header.commit - 1 - } else { - self.header.commit - }; - if self.index.last().map(|entry| entry.0).unwrap_or(0) != last { - return Err(CrabError::LTXCorrupted); - } - } - - let mut trailer_bytes = [0; TRAILER_SIZE]; - read_footer_exact(&mut footer, &mut trailer_bytes)?; - self.trailer = Trailer::parse(&trailer_bytes)?; - self.trailer.validate(self.header)?; - self.hash - .update(&trailer_bytes[..TRAILER_SIZE - CHECKSUM_SIZE]); - if footer.read(&mut [0; 1])? != 0 { - return Err(CrabError::LTXCorrupted); - } - if CHECKSUM_FLAG | self.hash.sum64() != self.trailer.file_checksum { - return Err(CrabError::ChecksumMismatch); - } - if self.header.is_snapshot() - && !self.header.no_checksum() - && self.rolling_checksum != self.trailer.post_apply_checksum - { - return Err(CrabError::ChecksumMismatch); - } - - self.state = DecoderState::Closed; - Ok(()) - } - - pub(crate) fn artifact(&self) -> Result<(u64, [u8; 32])> { - if self.state != DecoderState::Closed { - return Err(CrabError::InvalidState("LTX decoder is not closed")); - } - Ok(( - self.reader.bytes, - *self.reader.digest.clone().finalize().as_bytes(), - )) - } - - #[cfg(feature = "replica")] - pub(crate) fn into_replica_index(self) -> Result> { - self.replica_index - .ok_or(CrabError::InvalidState("LTX page index was not requested")) - } -} - -pub(crate) struct Encoder { - pub(crate) writer: W, - pub(crate) header: Header, - pub(crate) trailer: Trailer, - hash: Crc64, - index: EncoderIndex, - compressor: crate::lz4_block::Compressor, - bytes_written: u64, - previous_page_number: u32, - header_written: bool, - closed: bool, -} - -#[derive(Clone)] -#[cfg_attr(not(feature = "replica"), expect(dead_code))] -pub(crate) struct EncodedPage { - pub(crate) page: u32, - pub(crate) offset: u64, - pub(crate) size: u64, - pub(crate) frame_hash: [u8; 32], - pub(crate) checksum: u64, -} - -enum EncoderIndex { - Memory(Vec), - File(Box), -} - -impl EncoderIndex { - fn write_entry(&mut self, page: u32, offset: u64, size: u64) -> Result<()> { - let mut bytes = Vec::with_capacity(30); - write_uvarint(&mut bytes, u64::from(page)); - write_uvarint(&mut bytes, offset); - write_uvarint(&mut bytes, size); - self.write_all(&bytes) - } - - fn finish(&mut self) -> Result<()> { - self.write_all(&[0]) - } - - fn write_all(&mut self, bytes: &[u8]) -> Result<()> { - match self { - Self::Memory(index) => index.extend_from_slice(bytes), - Self::File(index) => index.write_all(bytes)?, - } - Ok(()) - } - - fn len(&self) -> Result { - match self { - Self::Memory(index) => Ok(index.len() as u64), - Self::File(index) => Ok(index.file_len()?), - } - } - - fn read_exact_at(&mut self, offset: u64, length: usize) -> Result> { - match self { - Self::Memory(index) => { - let start = usize::try_from(offset).map_err(|_| CrabError::LTXCorrupted)?; - let end = start.checked_add(length).ok_or(CrabError::LTXCorrupted)?; - Ok(index - .get(start..end) - .ok_or(CrabError::LTXCorrupted)? - .to_vec()) - } - Self::File(index) => Ok(index.read_exact_at(offset, length)?), - } - } -} - -impl Encoder { - pub(crate) fn new_block(writer: W) -> Self { - Self::new(writer, EncoderIndex::Memory(Vec::new())) - } - - pub(crate) fn new_block_with_index(writer: W, index: Option>) -> Self { - Self::new( - writer, - index.map_or_else(|| EncoderIndex::Memory(Vec::new()), EncoderIndex::File), - ) - } - - fn new(writer: W, index: EncoderIndex) -> Self { - Self { - writer, - header: Header::default(), - trailer: Trailer::default(), - hash: Crc64::new(), - index, - compressor: crate::lz4_block::Compressor::default(), - bytes_written: 0, - previous_page_number: 0, - header_written: false, - closed: false, - } - } - - pub(crate) fn into_writer(self) -> W { - self.writer - } - - pub(crate) fn encode_header(&mut self, header: Header) -> Result<()> { - if self.header_written || self.closed { - return Err(CrabError::LTXCorrupted); - } - header.validate()?; - self.header = header; - let bytes = header.marshal(); - self.write_hashed(&bytes)?; - self.header_written = true; - Ok(()) - } - - pub(crate) fn encode_page(&mut self, mut page: PageHeader, data: &[u8]) -> Result { - if !self.header_written - || self.closed - || page.pgno > self.header.commit - || data.len() != self.header.page_size as usize - { - return Err(CrabError::LTXCorrupted); - } - page.validate()?; - let lock_page = lock_pgno(self.header.page_size); - if page.pgno == lock_page { - return Err(CrabError::LTXCorrupted); - } - - if self.header.is_snapshot() { - if self.previous_page_number == 0 && page.pgno != 1 { - return Err(CrabError::LTXCorrupted); - } - let expected = if self.previous_page_number == lock_page - 1 { - self.previous_page_number + 2 - } else { - self.previous_page_number + 1 - }; - if self.previous_page_number != 0 && page.pgno != expected { - return Err(CrabError::LTXCorrupted); - } - } else if self.previous_page_number >= page.pgno { - return Err(CrabError::LTXCorrupted); - } - - let offset = self.bytes_written; - let compressed = self.compressor.compress(data)?; - page.flags |= PAGE_HEADER_FLAG_SIZE; - let header = page.marshal(); - self.write_hashed(&header)?; - let size = u32::try_from(compressed.len()) - .map_err(|error| CrabError::Other(Box::new(error)))? - .to_be_bytes(); - self.write_hashed(&size)?; - self.writer.write_all(&compressed)?; - self.bytes_written += compressed.len() as u64; - self.hash.update(data); - - self.previous_page_number = page.pgno; - let frame_size = self.bytes_written - offset; - self.index.write_entry(page.pgno, offset, frame_size)?; - let mut frame_hash = blake3::Hasher::new(); - frame_hash.update(&header); - frame_hash.update(&size); - frame_hash.update(&compressed); - Ok(EncodedPage { - page: page.pgno, - offset, - size: frame_size, - frame_hash: *frame_hash.finalize().as_bytes(), - checksum: checksum_page(page.pgno, data), - }) - } - - pub(crate) fn close(&mut self, post_apply_checksum: u64) -> Result<()> { - if !self.header_written || self.closed { - return Err(CrabError::LTXCorrupted); - } - - self.write_hashed(&[0; PAGE_HEADER_SIZE])?; - let index_offset = self.bytes_written; - self.index.finish()?; - let index_length = self.index.len()?; - let mut copied = 0_u64; - while copied < index_length { - let length = usize::try_from((index_length - copied).min(INDEX_COPY_BYTES as u64)) - .map_err(|_| CrabError::LTXCorrupted)?; - let bytes = self.index.read_exact_at(copied, length)?; - self.write_hashed(&bytes)?; - copied += length as u64; - } - self.write_hashed(&(self.bytes_written - index_offset).to_be_bytes())?; - - self.trailer.post_apply_checksum = post_apply_checksum; - self.hash.update(&post_apply_checksum.to_be_bytes()); - self.trailer.file_checksum = CHECKSUM_FLAG | self.hash.sum64(); - self.trailer.validate(self.header)?; - if self.header.commit == 0 && post_apply_checksum != CHECKSUM_FLAG { - return Err(CrabError::LTXCorrupted); - } - self.writer.write_all(&self.trailer.marshal())?; - self.bytes_written += TRAILER_SIZE as u64; - self.closed = true; - Ok(()) - } - - fn write_hashed(&mut self, bytes: &[u8]) -> Result<()> { - self.writer.write_all(bytes)?; - self.hash.update(bytes); - self.bytes_written += bytes.len() as u64; - Ok(()) - } -} - -fn read_uvarint(reader: &mut impl Read, hash: &mut Crc64) -> Result<(u64, u64)> { - let mut value = 0u64; - let mut bytes = [0; 10]; - for index in 0..bytes.len() { - read_footer_exact(reader, &mut bytes[index..=index])?; - let byte = bytes[index]; - if byte < 0x80 { - if index == 9 && byte > 1 { - return Err(CrabError::LTXCorrupted); - } - hash.update(&bytes[..=index]); - return Ok((value | (u64::from(byte) << (index * 7)), index as u64 + 1)); - } - value |= u64::from(byte & 0x7f) << (index * 7); - } - Err(CrabError::LTXCorrupted) -} - -fn read_footer_exact(reader: &mut impl Read, mut bytes: &mut [u8]) -> Result<()> { - // A clean EOF inside the footer is malformed input. Preserve actual I/O - // failures so callers retain their retry/capacity classification and cause. - while !bytes.is_empty() { - let read = reader.read(bytes)?; - if read == 0 { - return Err(CrabError::LTXCorrupted); - } - bytes = &mut bytes[read..]; - } - Ok(()) -} - -fn write_uvarint(bytes: &mut Vec, mut value: u64) { - while value >= 0x80 { - bytes.push(value as u8 | 0x80); - value >>= 7; - } - bytes.push(value as u8); -} - -struct CountingReader { - inner: R, - bytes: u64, - digest: blake3::Hasher, -} -impl Read for CountingReader { - fn read(&mut self, bytes: &mut [u8]) -> std::io::Result { - let n = self.inner.read(bytes)?; - self.bytes += n as u64; - self.digest.update(&bytes[..n]); - Ok(n) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::{Txid, ltx}; - - #[test] - #[cfg(feature = "replica")] - fn ordinary_verification_does_not_collect_replica_entries() { - for bytes in [ - include_bytes!("../tests/vectors/celld-10cb130-snapshot-block-512.ltx").as_slice(), - include_bytes!("../tests/vectors/celld-10cb130-snapshot-frame-512.ltx").as_slice(), - ] { - let mut decoder = Decoder::new(bytes); - decoder.decode_header().unwrap(); - let mut scratch = vec![0; decoder.header.page_size as usize]; - while decoder.decode_page(&mut scratch).unwrap().is_some() {} - decoder.close().unwrap(); - assert!(decoder.replica_index.is_none()); - } - } - - #[test] - fn footer_io_failures_preserve_their_source_and_classification() { - struct Fault; - impl Read for Fault { - fn read(&mut self, _: &mut [u8]) -> std::io::Result { - Err(std::io::Error::from(std::io::ErrorKind::TimedOut)) - } - } - let error = read_uvarint(&mut Fault, &mut Crc64::new()).unwrap_err(); - assert!(matches!( - error, - CrabError::Io(ref source) if source.kind() == std::io::ErrorKind::TimedOut - )); - assert_eq!( - error.classify(), - crate::FailureClass::Retryable { after: None } - ); - } - - #[test] - fn malformed_footer_rejection_does_not_drain_an_unbounded_tail() { - let mut encoder = Encoder::new_block(Vec::new()); - encoder - .encode_header(ltx::Header { - version: ltx::VERSION, - page_size: 4_096, - commit: 1, - min_txid: Txid(1), - max_txid: Txid(1), - ..ltx::Header::default() - }) - .unwrap(); - let page = vec![1; 4_096]; - encoder - .encode_page(ltx::PageHeader { pgno: 1, flags: 0 }, &page) - .unwrap(); - encoder.close(checksum_page(1, &page)).unwrap(); - let bytes = encoder.into_writer(); - - for early_end in [false, true] { - let mut prefix = bytes.clone(); - if early_end { - prefix[HEADER_SIZE..HEADER_SIZE + PAGE_HEADER_SIZE].fill(0); - } - let mut reader = CountingReader { - inner: std::io::Cursor::new(prefix).chain(std::io::repeat(0).take(8 << 20)), - bytes: 0, - digest: blake3::Hasher::new(), - }; - let mut decoder = Decoder::new(&mut reader); - decoder.decode_header().unwrap(); - let mut scratch = vec![0; 4_096]; - while decoder.decode_page(&mut scratch).unwrap().is_some() {} - assert!(matches!(decoder.close(), Err(CrabError::LTXCorrupted))); - drop(decoder); - assert!(reader.bytes < 128 << 10, "early end: {early_end}"); - } - } - - #[test] - fn file_spooled_index_matches_in_memory_encoder_bytes() { - let header = ltx::Header { - version: ltx::VERSION, - page_size: 4_096, - commit: 2, - min_txid: Txid(1), - max_txid: Txid(1), - ..ltx::Header::default() - }; - let pages = [(1, vec![1; 4_096]), (2, vec![2; 4_096])]; - let checksum = pages.iter().fold(CHECKSUM_FLAG, |checksum, (page, data)| { - CHECKSUM_FLAG | (checksum ^ ltx::checksum_page(*page, data)) - }); - let encode = |mut encoder: Encoder>| { - encoder.encode_header(header).unwrap(); - for (page, data) in &pages { - encoder - .encode_page( - ltx::PageHeader { - pgno: *page, - flags: 0, - }, - data, - ) - .unwrap(); - } - encoder.close(checksum).unwrap(); - encoder.into_writer() - }; - let expected = encode(Encoder::new_block(Vec::new())); - let directory = tempfile::TempDir::new().unwrap(); - let index_path = directory.path().join("index"); - let index = std::fs::OpenOptions::new() - .create_new(true) - .read(true) - .write(true) - .open(index_path) - .unwrap(); - let actual = encode(Encoder::new_block_with_index( - Vec::new(), - Some(Box::new(index)), - )); - - assert_eq!(actual, expected); - crate::ltx::inspect_reader(std::io::Cursor::new(&actual)).unwrap(); - } -} - -#[test] -fn uvarint_roundtrips_every_boundary_value() { - for value in [0_u64, 1, 127, 128, 16_383, 16_384, u64::MAX - 1, u64::MAX] { - let mut bytes = Vec::new(); - write_uvarint(&mut bytes, value); - let (decoded, position) = read_uvarint(&mut bytes.as_slice(), &mut Crc64::new()).unwrap(); - assert_eq!(decoded, value); - assert_eq!( - position, - bytes.len() as u64, - "varint {value} must consume exactly its encoding" - ); - } -} - -#[test] -fn uvarint_rejects_a_truncated_encoding() { - assert!(matches!( - read_uvarint(&mut [0x80].as_slice(), &mut Crc64::new()), - Err(CrabError::LTXCorrupted) - )); -} - -#[test] -fn uvarint_rejects_encodings_that_overflow_the_value() { - assert!(matches!( - read_uvarint(&mut [0xff; 10].as_slice(), &mut Crc64::new()), - Err(CrabError::LTXCorrupted) - )); - - let mut overflows = vec![0xff; 9]; - overflows.push(0x02); - assert!(matches!( - read_uvarint(&mut overflows.as_slice(), &mut Crc64::new()), - Err(CrabError::LTXCorrupted) - )); - - let mut maximal = vec![0xff; 9]; - maximal.push(0x01); - assert_eq!( - read_uvarint(&mut maximal.as_slice(), &mut Crc64::new()).unwrap(), - (u64::MAX, 10) - ); -} diff --git a/crates/crab-ltx/src/commit.rs b/crates/crab-ltx/src/commit.rs deleted file mode 100644 index 70a13d715..000000000 --- a/crates/crab-ltx/src/commit.rs +++ /dev/null @@ -1,82 +0,0 @@ -//! Records SQLite's committed WAL boundary so a torn/corrupt WAL cannot silently -//! turn a successful application commit into an older successful capture. - -use std::ffi::{c_char, c_int, c_void}; -use std::path::Path; -use std::sync::atomic::{AtomicU32, Ordering}; - -use crate::Result; -use rusqlite::{Connection, ffi}; - -#[derive(Clone, Copy)] -pub(crate) struct WalCut { - pub salt1: u32, - pub salt2: u32, - pub frames: u32, -} - -pub(crate) struct CommitObserver { - frames: Box, -} - -impl CommitObserver { - pub fn install(connection: &Connection) -> Self { - let mut observer = Self { - frames: Box::new(AtomicU32::new(0)), - }; - // SAFETY: the boxed atomic has a stable address and outlives the writer. - // The callback neither unwinds nor calls SQLite. Db drops its - // writer before this observer, including on implicit drop. - unsafe { - ffi::sqlite3_wal_hook( - connection.handle(), - Some(committed), - (&mut *observer.frames as *mut AtomicU32).cast(), - ); - } - observer - } - - pub fn reset(&self) { - self.frames.store(0, Ordering::Relaxed); - } - pub fn frames(&self) -> u32 { - self.frames.load(Ordering::Relaxed) - } - - pub fn cut(&self, path: &Path, host: &crate::Host) -> Result> { - let frames = self.frames(); - if frames == 0 { - return Ok(None); - } - let mut wal_path = path.as_os_str().to_owned(); - wal_path.push("-wal"); - let header = host - .filesystem - .open(Path::new(&wal_path))? - .read_exact_at(0, 32)?; - if header.len() != 32 { - return Err(std::io::Error::from(std::io::ErrorKind::UnexpectedEof).into()); - } - let salt1 = u32::from_be_bytes([header[16], header[17], header[18], header[19]]); - let salt2 = u32::from_be_bytes([header[20], header[21], header[22], header[23]]); - Ok(Some(WalCut { - salt1, - salt2, - frames, - })) - } -} - -unsafe extern "C" fn committed( - data: *mut c_void, - _db: *mut ffi::sqlite3, - _name: *const c_char, - frames: c_int, -) -> c_int { - // SAFETY: only CommitObserver::install supplies this pointer; its allocation - // remains live until after the connection closes. SQLite calls synchronously. - let counter = unsafe { &*data.cast::() }; - counter.store(frames as u32, Ordering::Relaxed); - ffi::SQLITE_OK -} diff --git a/crates/crab-ltx/src/db.rs b/crates/crab-ltx/src/db.rs deleted file mode 100644 index 4fc70ff88..000000000 --- a/crates/crab-ltx/src/db.rs +++ /dev/null @@ -1,1069 +0,0 @@ -//! Managed database over the local replica: admissions, checkpoints, and cut publishing. -use std::collections::BTreeMap; -use std::path::{Path, PathBuf}; - -use rusqlite::{Connection, Transaction}; - -use crate::{CaptureBatch, CrabError, Limits, LocalSegment, Position, Result, SegmentInfo}; -use crate::{capture::CaptureEngine, host::LtxHost, ltx, types::Txid}; - -#[cfg(feature = "replica")] -pub use crate::writable_vfs::{HydrationBatch, HydrationRead}; - -/// Number of SQLite connections retained by one open managed database. -pub const MANAGED_SQLITE_CONNECTIONS: u64 = 3; - -/// Page-cache byte target budgeted for each retained SQLite connection. -pub const MANAGED_CONNECTION_PAGE_CACHE_BYTES: u64 = 64 * 1024; - -const MANAGED_CONNECTION_PAGE_CACHE_KIB: i64 = 64; - -/// One exclusive local capture session with a serialized SQLite writer. -/// -/// The caller owns the database and its directory: no external writers, direct -/// checkpoints, control-table edits, or deletion of retained LTX files. Opening -/// claims a fresh metadata directory; reactivation requires exact restore into -/// a fresh directory, not reusing potentially unpublished local state. -pub struct Db { - capture: CaptureEngine, - writer: Connection, - observer: crate::commit::CommitObserver, - required_cut: Option, - limits: Limits, - fenced: bool, - retained_bytes: u64, - retained_segments: usize, - local_disk: crate::DiskReservation, - sparse: bool, - #[cfg(feature = "replica")] - retained: Vec, - path: PathBuf, - host: crate::Host, - pending_durability: Vec, - #[cfg(feature = "replica")] - paged: Option, -} - -impl Db { - /// Returns a thread-safe handle for interrupting the current SQLite operation. - /// - /// The handle becomes inert after the database closes. Calling it does not - /// prove rollback or cancellation; the owner must still await the operation. - #[must_use] - pub fn interrupt_handle(&self) -> rusqlite::InterruptHandle { - self.writer.get_interrupt_handle() - } - - #[cfg(feature = "replica")] - pub(crate) fn open_cell_paged( - database: crate::CellWritableDatabase, - destination: &Path, - ) -> Result { - Self::open_sparse(crate::paged_io::Database::Cell(database), destination) - } - - #[cfg(feature = "replica")] - fn open_sparse(database: crate::paged_io::Database, destination: &Path) -> Result { - let limits = database.limits(); - let position = database.position(); - let page_size = database.page_size(); - let count = database.page_count(); - let checksums = database.checksums()?; - let host = database.host(); - let registration = crate::writable_vfs::Registration::new(database, destination)?; - let mut db = Self::open_inner( - destination, - limits, - Some(registration.vfs()), - host, - true, - None, - ) - .map_err(|error| registration.take_error().unwrap_or(error))?; - db.capture - .seed_continuation(position, checksums, page_size, count)?; - db.paged = Some(registration); - Ok(db) - } - - /// Reports resolved pages of the pinned cut for a sparse activation. - #[cfg(feature = "replica")] - pub fn hydration(&self) -> Result> { - self.paged.as_ref().map(|p| p.hydration()).transpose() - } - - /// Hydrates at most `pages` unresolved cut pages through the same VFS as SQL. - /// - /// Call periodically on the database worker; scheduling and cancellation - /// belong to the owner. This operation never publishes or checkpoints WAL. - #[cfg(feature = "replica")] - pub fn hydrate_step(&mut self, pages: u32) -> Result { - self.ensure_active()?; - let paged = self - .paged - .as_mut() - .ok_or(CrabError::InvalidState("not a sparse activation"))?; - crate::paged_io::with_paged_io_origin(crate::LtxReadOrigin::Hydrating, || { - paged.step(&self.writer, pages) - }) - } - - /// Selects up to 64 missing pages for asynchronous background hydration. - /// - /// The caller owns fetch admission and must install only on this activation. - #[cfg(feature = "replica")] - pub fn prepare_hydration(&self, pages: u32) -> Result> { - self.ensure_active()?; - self.paged - .as_ref() - .map(|paged| paged.prepare_hydration(pages)) - .transpose() - } - - /// Installs authenticated pages without provider I/O on the SQLite worker. - /// - /// Superseded pages are skipped. An installation failure fences this Db. - #[cfg(feature = "replica")] - pub fn install_hydration(&mut self, batch: HydrationBatch) -> Result { - self.ensure_active()?; - let result = self - .paged - .as_mut() - .ok_or(CrabError::InvalidState("not a sparse activation"))? - .install_hydration(&self.writer, batch); - if result.is_err() { - self.fenced = true; - } - result - } - - /// Takes the provider/checksum source behind a sparse SQLite I/O error. - #[cfg(feature = "replica")] - pub fn take_io_error(&self) -> Option { - self.paged.as_ref().and_then(|p| p.take_error()) - } - - /// Deletes this session's exact captured artifacts after their root publishes. - /// - /// Publication is the durability proof for selected deferred captures, so - /// they are reverified and unlinked without a local durability barrier. - /// Every session owns a fresh metadata directory, so a crash-resurrected - /// local name remains quarantined rather than becoming acknowledged state. - /// An error retains unfinished accounting so the owner can retry or discard - /// the complete local session. - #[cfg(feature = "replica")] - pub fn prune_captured(&mut self, batch: &crate::CaptureBatch) -> Result { - self.prune_retained(|segment| { - batch.segments.iter().any(|published| { - published.path() == segment.path() && published.info() == segment.info() - }) - }) - } - - #[cfg(feature = "replica")] - fn prune_retained( - &mut self, - mut selected: impl FnMut(&crate::LocalSegment) -> bool, - ) -> Result { - let mut removed = 0; - let mut index = 0; - while index < self.retained.len() { - let segment = &self.retained[index]; - if !selected(segment) { - index += 1; - continue; - } - let mut file = LtxHost { - facilities: self.host.clone(), - max_database_bytes: self.limits.max_database_bytes, - max_file_bytes: segment.info().size_bytes, - } - .open(segment.path())?; - if file.file_len()? != segment.info().size_bytes { - return Err(CrabError::ChecksumMismatch); - } - crate::recovery::verify_segment_reader( - std::io::BufReader::with_capacity(64 << 10, file), - segment.info(), - self.limits, - )?; - // The published immutable root, not durable local deletion, releases - // the result. A fresh session never adopts crash-resurrected residue. - self.host.filesystem.remove_file(segment.path())?; - self.pending_durability - .retain(|pending| pending != segment.path()); - let segment = self.retained.remove(index); - self.retained_bytes -= segment.info().size_bytes; - self.retained_segments -= 1; - removed += 1; - } - self.reconcile_local_disk()?; - Ok(removed) - } - - /// Restores a pinned plan into a fresh session, preserving the parent position. - /// - /// Local leftovers are never used to infer acknowledged state. Any prior - /// unacknowledged session remains quarantined for explicit reconciliation. - pub fn resume(plan: &crate::VerifiedPlan, destination: &Path, limits: Limits) -> Result { - Self::resume_with_host(plan, destination, limits, crate::Host::default()) - } - - /// Resumes an exact verified plan using the same host for installation and capture. - pub fn resume_with_host( - plan: &crate::VerifiedPlan, - destination: &Path, - limits: Limits, - host: crate::Host, - ) -> Result { - let materialized = plan.materialize(); - let limits = limits.validate()?; - if u64::from(materialized.database_pages) * u64::from(materialized.page_size) - > limits.max_database_bytes - { - return Err(CrabError::Limit(crate::LimitKind::DatabaseBytes)); - } - let database_bytes = - u64::from(materialized.database_pages) * u64::from(materialized.page_size); - let local_disk = host.reserve_local_disk(database_bytes)?; - host.restore_materialized(materialized, destination)?; - let vfs = host.sqlite_vfs.clone(); - let mut db = Self::open_inner( - destination, - limits, - vfs.as_deref(), - host, - false, - Some(local_disk), - )?; - db.capture.seed_continuation( - plan.position(), - materialized.checksums.clone(), - materialized.page_size, - materialized.database_pages, - )?; - Ok(db) - } - - /// Opens a new capture session on a new or exactly restored SQLite file. - /// - /// The parent must exist and the local path must be UTF-8. An existing - /// capture directory is refused, even after a clean close. Failed opens - /// leave that directory quarantined for caller-owned cleanup. - pub fn open(path: &Path, limits: Limits) -> Result { - Self::open_with_host(path, limits, crate::Host::default()) - } - - /// Opens a fresh session using the host's filesystem, SQLite VFS and clock. - pub fn open_with_host(path: &Path, limits: Limits, host: crate::Host) -> Result { - let vfs = host.sqlite_vfs.clone(); - Self::open_inner(path, limits, vfs.as_deref(), host, false, None) - } - - fn open_inner( - path: &Path, - limits: Limits, - vfs: Option<&str>, - facilities: crate::Host, - sparse: bool, - local_disk: Option, - ) -> Result { - let limits = limits.validate()?; - if path.to_str().is_none() || path.file_name().is_none() { - return Err(CrabError::InvalidState( - "database path must be a UTF-8 file path", - )); - } - // Resolve aliases before claiming the session directory, and keep all - // later WAL reads independent of the process working directory. - let resolved = match facilities.filesystem.canonicalize(path) { - Ok(path) => path, - Err(error) if error.kind() == std::io::ErrorKind::NotFound => { - let parent = path - .parent() - .filter(|p| !p.as_os_str().is_empty()) - .unwrap_or(Path::new(".")); - facilities.filesystem.canonicalize(parent)?.join( - path.file_name() - .ok_or(CrabError::InvalidState("missing database filename"))?, - ) - } - Err(error) => return Err(error.into()), - }; - let path = resolved.as_path(); - if path.to_str().is_none() { - return Err(CrabError::InvalidState( - "resolved database path must be UTF-8", - )); - } - let host = LtxHost { - facilities: facilities.clone(), - max_database_bytes: limits.max_database_bytes, - max_file_bytes: limits.max_file_bytes, - }; - let database_bytes = match facilities.filesystem.file_len(path) { - Ok(len) => { - host.check_database_size(len)?; - len - } - Err(error) if error.kind() == std::io::ErrorKind::NotFound => 0, - Err(error) => return Err(error.into()), - }; - let local_disk = match local_disk { - Some(local_disk) => { - local_disk.resize(if sparse { 0 } else { database_bytes })?; - local_disk - } - None => facilities.reserve_local_disk(if sparse { 0 } else { database_bytes })?, - }; - // Atomic directory creation fences concurrent handles and stale sessions. - // Never unlink it on close: an old open file must not acquire a new epoch. - facilities - .filesystem - .create_dir(&CaptureEngine::meta_path_for(path))?; - let capture = CaptureEngine::open_with_host(path, host, vfs, limits.max_capture_bytes)?; - let writer = open_connection(path, vfs)?; - writer.busy_timeout(std::time::Duration::from_secs(1))?; - writer.pragma_update(None, "wal_autocheckpoint", 0)?; - writer.pragma_update(None, "synchronous", "FULL")?; - writer.pragma_update(None, "foreign_keys", true)?; - let page_size: u32 = writer.query_row("PRAGMA page_size", [], |row| row.get(0))?; - let max_pages = limits.max_database_bytes / u64::from(page_size); - if max_pages == 0 { - return Err(CrabError::Limit(crate::LimitKind::DatabasePageSize)); - } - writer.pragma_update(None, "max_page_count", max_pages)?; - let observer = crate::commit::CommitObserver::install(&writer); - Ok(Self { - capture, - writer, - observer, - required_cut: None, - limits, - fenced: false, - retained_bytes: 0, - retained_segments: 0, - local_disk, - sparse, - #[cfg(feature = "replica")] - retained: Vec::new(), - path: path.to_owned(), - host: facilities, - pending_durability: Vec::new(), - #[cfg(feature = "replica")] - paged: None, - }) - } - - /// Commits one local SQL transaction; call `capture` before publishing it. - /// - /// SQL is trusted: do not issue transaction-control statements or change - /// pager pragmas through the callback. Success is NOT remote durability. - pub fn transaction( - &mut self, - operation: impl FnOnce(&Transaction<'_>) -> rusqlite::Result, - ) -> Result { - self.transaction_with(operation) - .map_err(|error| match error { - crate::TransactionError::Admission(error) => error, - crate::TransactionError::Operation(error) - | crate::TransactionError::Sqlite(error) => error.into(), - crate::TransactionError::Capture(error) => error, - }) - } - - /// Commits one transaction while preserving application-domain failures. - /// - /// An `Operation` result guarantees the transaction was rolled back and the - /// writer remains reusable. SQLite commit/rollback ambiguity fences the - /// writer. A successful return is still local-only until capture and remote - /// publication complete. - /// - /// A commit larger than `Limits::max_capture_bytes` is not refused: the - /// later capture represents it as a full database image, which is bounded - /// by `Limits::max_file_bytes`, so a large write can never leave a local - /// commit that the session cannot capture. - pub fn transaction_with( - &mut self, - operation: impl FnOnce(&Transaction<'_>) -> std::result::Result, - ) -> std::result::Result> - where - E: std::error::Error + 'static, - { - self.ensure_active() - .map_err(crate::TransactionError::Capture)?; - self.ensure_capacity() - .map_err(crate::TransactionError::Admission)?; - let disk_before = self.local_disk.bytes(); - let write_bytes = self.limits.max_capture_bytes.checked_mul(2).ok_or( - crate::TransactionError::Admission(CrabError::Limit(crate::LimitKind::LocalDiskBytes)), - )?; - self.local_disk - .try_grow(write_bytes) - .map_err(crate::TransactionError::Admission)?; - self.observer.reset(); - let tx = match self - .writer - .transaction_with_behavior(rusqlite::TransactionBehavior::Immediate) - { - Ok(tx) => tx, - Err(error) => { - let _ = self.local_disk.resize(disk_before); - return Err(crate::TransactionError::Sqlite(error)); - } - }; - let value = match operation(&tx) { - Ok(value) => value, - Err(error) => { - // FULL, interrupt, and ROLLBACK constraints can end the whole - // transaction. A second rollback would falsely fence the writer; - // autocommit proves rollback only if no WAL commit was observed. - let rollback = if tx.is_autocommit() && self.observer.frames() == 0 { - drop(tx); - Ok(()) - } else { - tx.rollback() - }; - if let Err(rollback) = rollback { - self.fenced = true; - return Err(crate::TransactionError::Sqlite(rollback)); - } - let _ = self.local_disk.resize(disk_before); - return Err(crate::TransactionError::Operation(error)); - } - }; - if let Err(error) = tx.commit() { - // Commit failure is potentially ambiguous even if SQLite did not - // invoke the WAL hook. Never accept another mutation here. - self.fenced = true; - return Err(crate::TransactionError::Sqlite(error)); - } - match self.observer.cut(&self.path, &self.host) { - Ok(Some(cut)) => self.required_cut = Some(cut), - Ok(None) => { - let _ = self.local_disk.resize(disk_before); - } - Err(error) => { - self.fenced = true; - return Err(crate::TransactionError::Capture(error)); - } - } - Ok(value) - } - - /// Runs one synchronous callback with SQLite writes disabled. - /// - /// The callback must not change connection pragmas or retain borrowed SQLite - /// values. Establishing or removing the read-only boundary failure fences - /// this capture session; an application error leaves it reusable. - pub fn query_with( - &mut self, - operation: impl FnOnce(&Connection) -> std::result::Result, - ) -> std::result::Result> - where - E: std::error::Error + 'static, - { - self.ensure_active().map_err(crate::QueryError::State)?; - if let Err(error) = self.writer.pragma_update(None, "query_only", true) { - self.fenced = true; - return Err(crate::QueryError::Sqlite(error)); - } - let result = operation(&self.writer); - if let Err(error) = self.writer.pragma_update(None, "query_only", false) { - self.fenced = true; - return Err(crate::QueryError::Sqlite(error)); - } - result.map_err(crate::QueryError::Operation) - } - - /// Captures committed WAL pages and all cuts made by checkpoint maintenance. - /// - /// A failure that can leave partial local state fences further use. A - /// declared capacity refusal happens before the cut is written, so the - /// session stays active with [`Self::has_pending_capture`] reporting the - /// uncaptured commit; the host may clear the obstruction and retry, or - /// discard the session and restore authoritative state. Retain returned - /// files until the canonical Cell root publishes; `prune_captured` can then - /// release one exact acknowledged batch. - pub fn capture(&mut self) -> Result { - self.ensure_active()?; - self.flush_pending_durability()?; - let (result, timing, wrote_cut) = self.capture_inner(false); - #[cfg(feature = "replica")] - self.host.observe_ltx_capture(&timing, result.is_ok()); - let result = result.map(|mut batch| { - batch.timing = timing; - batch - }); - // A failure after the cut writer started can leave partial local state, - // so the session fences. A refusal raised before the writer starts only - // declined work, and the pending WAL cut stays recoverable. - if result.is_err() && wrote_cut { - self.fenced = true; - } - result - } - - /// Captures committed WAL pages while deferring their durability barrier. - /// - /// LTX files are complete and readable when this returns, but their contents - /// and names are not durable until [`Self::durability_barrier`] succeeds. - /// This permits a host to group several captures behind one storage flush; - /// callers must complete the barrier before acknowledging a locally durable - /// batch. A higher-level protocol may instead publish the exact bytes to its - /// own durability boundary, then pass that published batch to - /// `prune_captured`. A failed local barrier fences the session. - /// - /// As with [`Self::capture`], a declared capacity refusal that happens - /// before any cut is written leaves the session active with - /// [`Self::has_pending_capture`] set. - pub fn capture_deferred(&mut self) -> Result { - self.ensure_active()?; - let (result, timing, wrote_cut) = self.capture_inner(true); - #[cfg(feature = "replica")] - self.host.observe_ltx_capture(&timing, result.is_ok()); - let result = result.map(|mut batch| { - batch.timing = timing; - self.pending_durability.extend( - batch - .segments - .iter() - .map(|segment| segment.path().to_owned()), - ); - batch - }); - if result.is_err() && wrote_cut { - self.fenced = true; - } - result - } - - /// Makes all files published by deferred captures durable as one barrier. - /// - /// The barrier syncs each completed file, each destination directory once, - /// and the new LTX directory chain once per session. If any step fails, the - /// session is fenced; no caller may acknowledge the pending paths. - pub fn durability_barrier(&mut self) -> Result<()> { - self.ensure_active()?; - self.flush_pending_durability() - } - - fn flush_pending_durability(&mut self) -> Result<()> { - if self.pending_durability.is_empty() { - return Ok(()); - } - let mut parents = BTreeMap::new(); - for path in &self.pending_durability { - let parent = path - .parent() - .filter(|parent| !parent.as_os_str().is_empty()) - .unwrap_or(Path::new(".")) - .to_owned(); - parents.entry(parent).or_insert_with(|| path.clone()); - } - let result = (|| { - self.host.filesystem.sync_files(&self.pending_durability)?; - parents - .values() - .try_for_each(|path| self.host.filesystem.sync_parent(path))?; - Ok::<(), std::io::Error>(()) - })() - .map_err(CrabError::from) - // A new LTX parent name cannot survive merely because its own contents did. - .and_then(|()| self.capture.sync_l0_ancestors()); - if result.is_ok() { - self.pending_durability.clear(); - } else { - self.fenced = true; - } - result - } - - /// Runs one capture attempt. - /// - /// The returned flag reports whether the attempt could have written local - /// cut state: a refusal raised before the writer starts leaves the session - /// intact, while any later failure requires fencing. - fn capture_inner( - &mut self, - defer_durability: bool, - ) -> (Result, crate::CaptureTiming, bool) { - self.capture.start_timing(self.host.now_monotonic()); - self.capture - .timing_begin(crate::capture::TimingPhase::Preparation); - if let Err(error) = self.ensure_capacity() { - self.capture - .timing_end(crate::capture::TimingPhase::Preparation); - let timing = self.capture.finish_timing(self.host.now_monotonic()); - return (Err(error), timing, false); - } - let result = (|| { - self.capture - .timing_end(crate::capture::TimingPhase::Preparation); - let before = self.capture.pos(); - if defer_durability { - self.capture.sync_deferred(self.required_cut)?; - } else { - self.capture.sync(self.required_cut)?; - } - self.required_cut = None; - let batch = self.collect_cuts(before)?; - self.reconcile_local_disk()?; - Ok(batch) - })(); - let timing = self.capture.finish_timing(self.host.now_monotonic()); - (result, timing, true) - } - - fn collect_cuts(&mut self, before: crate::Pos) -> Result { - let after = self.capture.pos(); - let mut segments = Vec::new(); - if after.txid.0 > before.txid.0 { - for txid in before.txid.0 + 1..=after.txid.0 { - let path = PathBuf::from(self.capture.ltx_path(0, Txid(txid), Txid(txid))); - let info = if let Some(info) = self.capture.take_sealed_l0_segment(Txid(txid)) { - info - } else { - // The inspection limit is the full-file bound: a captured - // cut may be a full database image, which the writer already - // bounded by `max_file_bytes`. - let file = crate::LtxHost { - facilities: self.host.clone(), - max_database_bytes: self.limits.max_database_bytes, - max_file_bytes: self.limits.max_file_bytes, - } - .open(&path)?; - self.capture - .timing_begin(crate::capture::TimingPhase::Verification); - let inspected = ltx::inspect_reader(file); - self.capture - .timing_end(crate::capture::TimingPhase::Verification); - let (decoded, size, digest) = inspected?; - SegmentInfo::from_inspected(&decoded, size, digest) - }; - self.capture.timing_add_ltx_bytes(info.size_bytes); - self.capture.timing_add_segment(); - // The capture writer already enforced the tighter incremental - // bound for delta cuts, so retention accounting only has to - // honor the file bound shared by every cut. - self.account(&info)?; - let segment = LocalSegment::new(path, info); - #[cfg(feature = "replica")] - let segment = match self.capture.take_sealed_l0_captured_index(Txid(txid)) { - Some(index) => segment.with_captured_index(index), - None => segment, - }; - #[cfg(feature = "replica")] - self.retained.push(segment.clone()); - segments.push(segment); - } - } - Ok(CaptureBatch { - segments, - position: after.into(), - timing: crate::CaptureTiming::default(), - }) - } - - /// Captures pending writes before a requested checkpoint and returns every cut. - /// - /// Failure fences this session, including failed forced checkpoints. All - /// returned cuts must be published before acknowledging the operation. - pub fn checkpoint(&mut self, mode: crate::CheckpointMode) -> Result { - self.ensure_active()?; - self.flush_pending_durability()?; - let (initial, mut timing, _) = self.capture_inner(false); - let result = (|| { - let mut batch = initial?; - self.local_disk.try_grow( - self.limits - .max_capture_bytes - .checked_mul(2) - .ok_or(CrabError::Limit(crate::LimitKind::LocalDiskBytes))?, - )?; - let before = self.capture.pos(); - self.capture.start_timing(self.host.now_monotonic()); - let checkpoint_result = self.capture.checkpoint(mode); - let extra_result = checkpoint_result.and_then(|()| self.collect_cuts(before)); - let checkpoint_timing = self.capture.finish_timing(self.host.now_monotonic()); - timing.merge(checkpoint_timing); - let extra = extra_result?; - batch.timing = timing; - batch.segments.extend(extra.segments); - batch.position = extra.position; - self.reconcile_local_disk()?; - Ok(batch) - })(); - #[cfg(feature = "replica")] - self.host.observe_ltx_capture(&timing, result.is_ok()); - if result.is_err() { - self.fenced = true; - } - result - } - - /// Returns the last sealed local position, not a remote durability receipt. - #[must_use] - pub fn position(&self) -> Position { - self.capture.pos().into() - } - - /// Writes the continuation a later [`Db::open_resumed`] continues from. - /// - /// The session must be drained: a pending capture means a local commit has - /// no published cut, so the continuation would name a position no root - /// owns. A sparse activation must be fully materialized, because its - /// local file holds zeros where no page was ever faulted in. The record - /// authorizes nothing by itself — the caller still has to prove that the - /// file holds one authoritative root before opening it. - #[cfg(feature = "replica")] - pub fn persist_continuation(&self) -> Result<()> { - self.ensure_active()?; - if self.has_pending_capture() { - return Err(CrabError::InvalidState( - "capture continuation requires a drained database", - )); - } - if self.sparse && !self.hydration()?.is_some_and(crate::Hydration::complete) { - return Err(CrabError::InvalidState( - "capture continuation requires a fully materialized activation", - )); - } - let position = self.position(); - let pages = self.capture.checksums().count(); - let page_size = self.capture.page_size(); - if position.txid == 0 || pages == 0 || position.checksum & crate::CHECKSUM_FLAG == 0 { - return Err(CrabError::InvalidState( - "capture continuation requires a committed position", - )); - } - crate::resume::write_checksums(&self.host, &self.path, self.capture.checksums())?; - crate::resume::write_continuation( - &self.host, - &self.path, - &crate::resume::Continuation { - position, - page_size, - pages, - }, - ) - } - - /// Opens a database that a clean close left resumable at the same path. - /// - /// Reads the continuation and dense page checksums recorded beside the file, - /// requires the file to be exactly the recorded image, and seeds the capture - /// session from them. No origin object is read, so the caller must have - /// matched its own resume record against the authoritative control first. - #[cfg(feature = "replica")] - pub fn open_resumed(path: &Path, limits: Limits) -> Result { - Self::open_resumed_with_host(path, limits, crate::Host::default()) - } - - /// Opens a resumable database using the host's filesystem, SQLite VFS, and clock. - #[cfg(feature = "replica")] - pub fn open_resumed_with_host(path: &Path, limits: Limits, host: crate::Host) -> Result { - let limits = limits.validate()?; - let continuation = crate::resume::read_continuation(&host, path)?; - let database_bytes = u64::from(continuation.pages) * u64::from(continuation.page_size); - if database_bytes > limits.max_database_bytes { - return Err(CrabError::Limit(crate::LimitKind::DatabaseBytes)); - } - if host.filesystem.file_len(path)? != database_bytes { - return Err(CrabError::InvalidState( - "resumed database length does not match its continuation", - )); - } - let ltx_host = crate::LtxHost { - facilities: host.clone(), - max_database_bytes: limits.max_database_bytes, - max_file_bytes: limits.max_database_bytes, - }; - let checksums = crate::pages::PageChecksums::from_file( - crate::LtxHost { - facilities: host.clone(), - max_database_bytes: limits.max_database_bytes, - max_file_bytes: limits.max_database_bytes, - }, - &crate::resume::checksum_path(path), - continuation.page_size, - continuation.pages, - continuation.position.checksum, - )?; - // The continuation and sidecar name a published image; a same-length - // local corruption must fall back to the authoritative root. - checksums.verify_database(<x_host, path, continuation.page_size)?; - let vfs = host.sqlite_vfs.clone(); - let local_disk = host.reserve_local_disk(database_bytes)?; - let mut db = Self::open_inner(path, limits, vfs.as_deref(), host, false, Some(local_disk))?; - db.capture.seed_continuation( - continuation.position, - checksums, - continuation.page_size, - continuation.pages, - )?; - Ok(db) - } - - /// Captures pending commits, then writes a full checksum-bearing snapshot. - /// - /// Its range is `1..=position.txid`. The snapshot can replace all preceding - /// cuts in a new manifest. The second return value owns every newly captured - /// cut: publish it to continue an existing head. Neither output is published. - pub fn snapshot(&mut self, destination: &Path) -> Result<(LocalSegment, CaptureBatch)> { - self.ensure_active()?; - let result = self.snapshot_inner(destination); - if result.is_err() { - self.fenced = true; - } - result - } - - fn snapshot_inner(&mut self, destination: &Path) -> Result<(LocalSegment, CaptureBatch)> { - self.flush_pending_durability()?; - let (batch, timing, _) = self.capture_inner(false); - #[cfg(feature = "replica")] - self.host.observe_ltx_capture(&timing, batch.is_ok()); - let mut batch = batch?; - batch.timing = timing; - self.local_disk.try_grow(self.limits.max_file_bytes)?; - let (mut scratch, mut output) = - SnapshotScratch::create(&self.host, destination, self.limits.max_file_bytes)?; - let pos: Position = self.capture.snapshot_to_writer(&mut output)?.into(); - if pos != batch.position { - return Err(CrabError::ChecksumMismatch); - } - output.sync_all()?; - drop(output); - let file = crate::LtxHost { - facilities: self.host.clone(), - max_database_bytes: self.limits.max_database_bytes, - max_file_bytes: self.limits.max_file_bytes, - } - .open(&scratch.path)?; - let (decoded, size, digest) = ltx::inspect_reader(file)?; - let info = SegmentInfo::from_inspected(&decoded, size, digest); - self.account(&info)?; - self.host - .filesystem - .persist_file_new(&scratch.path, destination)?; - scratch.installed = true; - let segment = LocalSegment::new(destination.to_owned(), info); - #[cfg(feature = "replica")] - self.retained.push(segment.clone()); - self.reconcile_local_disk()?; - Ok((segment, batch)) - } - - /// Returns the local file path; it must not be independently mutated. - #[must_use] - pub fn path(&self) -> &Path { - &self.path - } - - /// Releases the writer and checkpoint read lock without claiming publication. - pub fn close(mut self) -> Result<()> { - self.flush_pending_durability()?; - drop(self.writer); - self.capture.close() - } - - fn ensure_active(&self) -> Result<()> { - if self.fenced { - return Err(CrabError::Fenced); - } - Ok(()) - } - - fn ensure_capacity(&self) -> Result<()> { - if self.retained_segments >= self.limits.max_segments - || self.retained_bytes >= self.limits.max_plan_bytes - { - return Err(CrabError::Limit(crate::LimitKind::RetainedCaptureArtifacts)); - } - Ok(()) - } - - /// Reports whether a committed WAL cut is waiting for capture. - /// - /// While this is true the local database holds a commit that no LTX file - /// covers, so the host must not acknowledge it, serve it, or reuse the - /// session's local state. Retry [`Db::capture`] after clearing the - /// obstruction, or discard the session and restore authoritative state. - #[must_use] - pub fn has_pending_capture(&self) -> bool { - self.required_cut.is_some() - } - - fn account(&mut self, info: &SegmentInfo) -> Result<()> { - if info.size_bytes > self.limits.max_file_bytes { - return Err(CrabError::Limit(crate::LimitKind::LtxFileBytes)); - } - self.retained_bytes = self - .retained_bytes - .checked_add(info.size_bytes) - .ok_or(CrabError::Limit(crate::LimitKind::RetainedBytes))?; - self.retained_segments += 1; - if self.retained_segments > self.limits.max_segments - || self.retained_bytes > self.limits.max_plan_bytes - { - return Err(CrabError::Limit(crate::LimitKind::RetainedCaptureArtifacts)); - } - Ok(()) - } - - fn reconcile_local_disk(&self) -> Result<()> { - let database_bytes = if self.sparse { - 0 - } else { - self.host.filesystem.file_len(&self.path)? - }; - let wal_bytes = match self.host.filesystem.file_len(&self.capture.wal_path()) { - Ok(bytes) => bytes, - Err(error) if error.kind() == std::io::ErrorKind::NotFound => 0, - Err(error) => return Err(error.into()), - }; - let live = database_bytes - .checked_add(self.retained_bytes) - .ok_or(CrabError::Limit(crate::LimitKind::LocalDiskBytes))? - .checked_add(wal_bytes) - .ok_or(CrabError::Limit(crate::LimitKind::LocalDiskBytes))?; - self.local_disk.resize(live) - } -} - -struct SnapshotScratch { - filesystem: std::sync::Arc, - path: PathBuf, - installed: bool, -} - -impl SnapshotScratch { - fn create( - host: &crate::Host, - destination: &Path, - max_file_bytes: u64, - ) -> Result<(Self, crate::HostFile)> { - let parent = destination - .parent() - .filter(|path| !path.as_os_str().is_empty()) - .unwrap_or(Path::new(".")); - let filename = destination - .file_name() - .ok_or(CrabError::InvalidState("missing snapshot filename"))?; - static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); - for _ in 0..16 { - let mut scratch_name = filename.to_owned(); - scratch_name.push(format!( - ".crab-snapshot-{}-{}", - std::process::id(), - NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed) - )); - let path = parent.join(scratch_name); - let ltx_host = crate::LtxHost { - facilities: host.clone(), - max_database_bytes: max_file_bytes, - max_file_bytes, - }; - match ltx_host.create(&path) { - Ok(file) => { - return Ok(( - Self { - filesystem: host.filesystem.clone(), - path, - installed: false, - }, - file, - )); - } - Err(error) if error.kind() == std::io::ErrorKind::AlreadyExists => continue, - Err(error) => return Err(error.into()), - } - } - Err(std::io::Error::new( - std::io::ErrorKind::AlreadyExists, - "snapshot scratch namespace exhausted", - ) - .into()) - } -} - -impl Drop for SnapshotScratch { - fn drop(&mut self) { - if !self.installed { - let _ = self.filesystem.remove_file(&self.path); - } - } -} - -pub(crate) fn open_connection(path: &Path, vfs: Option<&str>) -> rusqlite::Result { - let connection = match vfs { - Some(vfs) => Connection::open_with_flags_and_vfs(path, rusqlite::OpenFlags::default(), vfs), - None => Connection::open(path), - }?; - configure_managed_connection(&connection)?; - Ok(connection) -} - -pub(crate) fn configure_managed_connection(connection: &Connection) -> rusqlite::Result<()> { - disable_lookaside(connection)?; - connection.pragma_update(None, "cache_size", -MANAGED_CONNECTION_PAGE_CACHE_KIB) -} - -fn disable_lookaside(connection: &Connection) -> rusqlite::Result<()> { - use rusqlite::ffi; - - // SQLite's default lookaside arena reserves memory per connection. Managed - // LTX connections use a small, stable statement vocabulary, so keeping - // that arena only adds resident cost across a dense Cell fleet. - // SAFETY: callers configure a newly opened connection before any SQLite - // operation, so no lookaside slot can be in use. - let result = unsafe { - ffi::sqlite3_db_config( - connection.handle(), - ffi::SQLITE_DBCONFIG_LOOKASIDE, - std::ptr::null_mut::(), - 0, - 0, - ) - }; - if result != ffi::SQLITE_OK { - return Err(rusqlite::Error::SqliteFailure( - ffi::Error::new(result), - None, - )); - } - Ok(()) -} - -pub(crate) fn read_main(connection: &Connection, offset: u64, size: usize) -> Result> { - use rusqlite::ffi; - let mut file: *mut ffi::sqlite3_file = std::ptr::null_mut(); - let mut data = vec![0u8; size]; - let offset = i64::try_from(offset).map_err(|_| CrabError::LTXCorrupted)?; - let size = i32::try_from(size).map_err(|_| CrabError::LTXCorrupted)?; - // SAFETY: the connection is exclusively borrowed on its database worker. - // SQLite owns FILE_POINTER until this borrow ends; data holds size bytes. - let rc = unsafe { - let rc = ffi::sqlite3_file_control( - connection.handle(), - c"main".as_ptr(), - ffi::SQLITE_FCNTL_FILE_POINTER, - (&mut file as *mut *mut ffi::sqlite3_file).cast(), - ); - if rc != ffi::SQLITE_OK || file.is_null() || (*file).pMethods.is_null() { - return Err(CrabError::InvalidState("SQLite main file unavailable")); - } - let read = (*(*file).pMethods) - .xRead - .ok_or(CrabError::InvalidState("SQLite main file cannot read"))?; - read(file, data.as_mut_ptr().cast(), size, offset) - }; - if rc != ffi::SQLITE_OK { - return Err(rusqlite::Error::SqliteFailure(ffi::Error::new(rc), None).into()); - } - Ok(data) -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-ltx/src/db/tests.rs b/crates/crab-ltx/src/db/tests.rs deleted file mode 100644 index 890ba9305..000000000 --- a/crates/crab-ltx/src/db/tests.rs +++ /dev/null @@ -1,710 +0,0 @@ -use super::*; -use crate::{VerifiedPlan, restore_exact}; -use std::{ - path::Path, - sync::{ - Arc, - atomic::{AtomicU64, Ordering}, - }, - time::{Duration, Instant}, -}; - -#[derive(Debug, thiserror::Error)] -#[error("inventory rejected the command")] -struct Rejected; - -struct TimingClock { - origin: Instant, - ticks: AtomicU64, -} - -impl TimingClock { - fn new() -> Self { - Self { - origin: Instant::now(), - ticks: AtomicU64::new(0), - } - } -} - -impl crate::environment::Clock for TimingClock { - fn unix_millis(&self) -> i64 { - 123456789 - } - - fn file_age(&self, _: &Path) -> std::io::Result { - Ok(Duration::ZERO) - } - - fn monotonic(&self) -> Instant { - self.origin + Duration::from_micros(self.ticks.fetch_add(1, Ordering::Relaxed)) - } -} - -#[test] -fn managed_connections_set_the_budgeted_page_cache() { - let temp = tempfile::TempDir::new().unwrap(); - let connection = open_connection(&temp.path().join("cache.sqlite"), None).unwrap(); - let cache_kib: i64 = connection - .query_row("PRAGMA cache_size", [], |row| row.get(0)) - .unwrap(); - assert_eq!(cache_kib, -MANAGED_CONNECTION_PAGE_CACHE_KIB); -} - -#[test] -fn capture_reports_deterministic_bounded_timing_for_real_ltx_work() { - let temp = tempfile::TempDir::new().unwrap(); - let host = crate::Host::default().with_clock(Arc::new(TimingClock::new())); - let mut db = - Db::open_with_host(&temp.path().join("timed.sqlite"), Limits::default(), host).unwrap(); - db.transaction(|tx| { - tx.execute_batch("CREATE TABLE events(value TEXT); INSERT INTO events VALUES ('ok')") - }) - .unwrap(); - - let batch = db.capture().unwrap(); - let phase_nanos = batch.timing.preparation_nanos - + batch.timing.schema_check_nanos - + batch.timing.wal_existence_nanos - + batch.timing.position_resolution_nanos - + batch.timing.wal_read_nanos - + batch.timing.page_collection_nanos - + batch.timing.verification_nanos - + batch.timing.encode_nanos - + batch.timing.local_write_nanos - + batch.timing.fsync_nanos - + batch.timing.parent_sync_nanos - + batch.timing.checkpoint_nanos; - let ltx_bytes = batch - .segments - .iter() - .map(|segment| segment.info().size_bytes) - .sum::(); - let segment = &batch.segments[0]; - let file = crate::LtxHost { - facilities: crate::Host::default(), - max_database_bytes: Limits::default().max_database_bytes, - max_file_bytes: Limits::default().max_file_bytes, - } - .open(segment.path()) - .unwrap(); - let (decoded, size, digest) = crate::ltx::inspect_reader(file).unwrap(); - assert_eq!( - segment.info(), - &crate::SegmentInfo::from_inspected(&decoded, size, digest) - ); - assert!(batch.timing.total_nanos > 0); - assert!(phase_nanos <= batch.timing.total_nanos); - assert_eq!(batch.timing.segment_count as usize, batch.segments.len()); - assert_eq!(batch.timing.ltx_bytes, ltx_bytes); - assert!(batch.timing.wal_bytes > 0); - assert!(batch.timing.database_bytes > 0); - assert!(batch.timing.schema_check_nanos > 0); - assert!(batch.timing.wal_existence_nanos > 0); - assert!(batch.timing.position_resolution_nanos > 0); - assert!(batch.timing.page_collection_nanos > 0); - assert!(batch.timing.local_write_nanos > 0); - assert!(batch.timing.fsync_nanos > 0); - assert!(batch.timing.parent_sync_nanos > 0); - assert_eq!( - batch.timing.wal_sparse_reads + batch.timing.wal_full_reads, - 1 - ); - assert!(batch.timing.wal_image_bytes > 0); - assert!(batch.timing.wal_file_bytes >= batch.timing.wal_read_bytes); - assert!(batch.timing.wal_read_bytes > 0); - assert_eq!(batch.timing.wal_snapshot_reads, 1); -} - -#[cfg(feature = "replica")] -#[test] -fn failed_capture_emits_its_bounded_ledger() { - #[derive(Default)] - struct CaptureTelemetry(std::sync::Mutex>); - - impl crate::LtxTelemetry for CaptureTelemetry { - fn capture(&self, timing: &crate::CaptureTiming, succeeded: bool) { - self.0.lock().unwrap().push((*timing, succeeded)); - } - } - - let temp = tempfile::TempDir::new().unwrap(); - let telemetry = Arc::new(CaptureTelemetry::default()); - let host = crate::Host::default().with_ltx_telemetry(telemetry.clone()); - // One retained cut fills the session plan budget, so the next capture is - // refused before the writer starts and still reports its bounded ledger. - let limits = Limits { - max_segments: 1, - ..Limits::default() - }; - let mut db = Db::open_with_host(&temp.path().join("failed.sqlite"), limits, host).unwrap(); - db.transaction(|tx| { - tx.execute_batch( - "CREATE TABLE events(value BLOB); INSERT INTO events VALUES(randomblob(4096))", - ) - }) - .unwrap(); - db.capture().unwrap(); - - // A checkpoint captures first, so its refusal is raised before the writer - // starts: the failed attempt still emits its bounded ledger. - assert!(matches!( - db.checkpoint(crate::CheckpointMode::Passive), - Err(CrabError::Limit(crate::LimitKind::RetainedCaptureArtifacts)) - )); - let attempts = telemetry.0.lock().unwrap(); - assert_eq!(attempts.len(), 2); - assert!(attempts[0].1); - assert!(!attempts[1].1); - assert!(attempts[1].0.total_nanos > 0); - assert_eq!(attempts[1].0.wal_read_bytes, 0); - assert!(attempts[0].0.wal_read_bytes > 0); -} - -#[cfg(feature = "replica")] -#[test] -fn captured_indexes_match_independent_ltx_inspection() { - let temp = tempfile::TempDir::new().unwrap(); - let mut db = Db::open(&temp.path().join("indexed.sqlite"), Limits::default()).unwrap(); - db.transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB); \ - INSERT INTO payload VALUES(randomblob(200000))", - ) - }) - .unwrap(); - - let batch = db.capture().unwrap(); - - for segment in &batch.segments { - let file = std::fs::File::open(segment.path()).unwrap(); - let (decoded, size, digest, pages) = crate::ltx::inspect_reader_with_index(file).unwrap(); - assert_eq!( - crate::SegmentInfo::from_inspected(&decoded, size, digest), - *segment.info() - ); - assert_eq!( - segment.captured_index().unwrap().as_ref(), - crate::paged::encode_index_from_pages(&pages) - .unwrap() - .as_slice() - ); - let cloned = segment.clone(); - assert_eq!( - segment.captured_index().unwrap().as_ptr(), - cloned.captured_index().unwrap().as_ptr() - ); - } - db.close().unwrap(); -} - -#[test] -fn managed_connections_disable_sqlite_lookaside() { - use rusqlite::ffi; - - let temp = tempfile::TempDir::new().unwrap(); - for index in 0..MANAGED_SQLITE_CONNECTIONS { - let connection = - open_connection(&temp.path().join(format!("lookaside-{index}.sqlite")), None).unwrap(); - let _statement = connection.prepare("SELECT 1").unwrap(); - let mut current = 0; - let mut highwater = 0; - let result = unsafe { - // SAFETY: the connection remains alive and is exclusively - // borrowed for the duration of this status query. - ffi::sqlite3_db_status( - connection.handle(), - ffi::SQLITE_DBSTATUS_LOOKASIDE_USED, - &mut current, - &mut highwater, - 0, - ) - }; - assert_eq!(result, ffi::SQLITE_OK); - assert_eq!(current, 0); - assert_eq!(highwater, 0); - } -} - -#[test] -fn oversized_commit_captures_as_a_full_image_instead_of_fencing() { - let temp = tempfile::TempDir::new().unwrap(); - let limits = Limits { - max_capture_bytes: 8 * 1024, - ..Limits::default() - }; - let mut db = Db::open(&temp.path().join("oversized.sqlite"), limits).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE payload(value BLOB)")) - .unwrap(); - let first = db.capture().unwrap(); - - // A commit whose delta cannot fit the incremental bound must still be - // captured: the writer escalates it to a full database image bounded by - // the file bound instead of fencing the session. - db.transaction(|tx| tx.execute_batch("INSERT INTO payload VALUES(randomblob(65536))")) - .unwrap(); - let second = db.capture().unwrap(); - assert_eq!(second.segments.len(), 1); - let escalated = &second.segments[0]; - assert!( - escalated.info().size_bytes > limits.max_capture_bytes, - "the escalated cut must exceed the incremental bound" - ); - assert!(escalated.info().size_bytes <= limits.max_file_bytes); - - // The escalated cut stays a valid chain element: a plan over both cuts - // restores the exact database. - let mut segments = first.segments.clone(); - segments.extend(second.segments.iter().cloned()); - let plan = VerifiedPlan::new(&segments, second.position, limits).unwrap(); - let destination = temp.path().join("restored.sqlite"); - assert_eq!(restore_exact(&plan, &destination).unwrap(), second.position); - let connection = Connection::open(&destination).unwrap(); - let rows: i64 = connection - .query_row("SELECT count(*) FROM payload", [], |row| row.get(0)) - .unwrap(); - let bytes: i64 = connection - .query_row("SELECT length(value) FROM payload", [], |row| row.get(0)) - .unwrap(); - assert_eq!(rows, 1); - assert_eq!(bytes, 65536); -} - -#[test] -fn newly_written_pages_are_counted_once_for_incremental_admission() { - let temp = tempfile::TempDir::new().unwrap(); - let limits = Limits { - max_capture_bytes: 40 * 1024, - ..Limits::default() - }; - let mut db = Db::open(&temp.path().join("growth.sqlite"), limits).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE payload(value BLOB)")) - .unwrap(); - let schema = db.capture().unwrap(); - db.transaction(|tx| tx.execute_batch("INSERT INTO payload VALUES(randomblob(24576))")) - .unwrap(); - let growth = db.capture().unwrap(); - assert!(growth.segments[0].info().size_bytes <= limits.max_capture_bytes); - - let mut segments = schema.segments; - segments.extend(growth.segments); - let plan = VerifiedPlan::new(&segments, growth.position, limits).unwrap(); - let restored = temp.path().join("growth-restored.sqlite"); - restore_exact(&plan, &restored).unwrap(); - let connection = Connection::open(restored).unwrap(); - let size: i64 = connection - .query_row("SELECT length(value) FROM payload", [], |row| row.get(0)) - .unwrap(); - assert_eq!(size, 24576); -} - -#[test] -fn truncate_boundary_image_may_exceed_the_incremental_bound() { - let temp = tempfile::TempDir::new().unwrap(); - let limits = Limits { - max_capture_bytes: 64 * 1024, - ..Limits::default() - }; - let mut db = Db::open(&temp.path().join("boundary.sqlite"), limits).unwrap(); - db.transaction(|tx| { - tx.execute_batch( - "CREATE TABLE payload(value BLOB); CREATE INDEX payload_len ON payload(length(value))", - ) - }) - .unwrap(); - let mut segments = db.capture().unwrap().segments; - for _ in 0..20 { - db.transaction(|tx| tx.execute_batch("INSERT INTO payload VALUES(randomblob(8192))")) - .unwrap(); - segments.extend(db.capture().unwrap().segments); - } - - // A truncate checkpoint writes the whole database as one boundary image. - // That image legitimately exceeds the incremental bound and is bounded by - // the file bound instead of failing and fencing the session. - let batch = db.checkpoint(crate::CheckpointMode::Truncate).unwrap(); - let position = batch.position; - segments.extend(batch.segments); - let largest = segments - .iter() - .map(|segment| segment.info().size_bytes) - .max() - .unwrap(); - assert!( - largest > limits.max_capture_bytes, - "the boundary image must exceed the incremental bound" - ); - assert!(largest <= limits.max_file_bytes); - - let plan = VerifiedPlan::new(&segments, position, limits).unwrap(); - let destination = temp.path().join("boundary-restored.sqlite"); - assert_eq!(restore_exact(&plan, &destination).unwrap(), position); - let connection = Connection::open(&destination).unwrap(); - let rows: i64 = connection - .query_row("SELECT count(*) FROM payload", [], |row| row.get(0)) - .unwrap(); - assert_eq!(rows, 20); -} - -#[test] -fn local_disk_admission_rejects_before_running_the_transaction() { - let temp = tempfile::TempDir::new().unwrap(); - let budget = crate::DiskBudget::new(255); - let host = crate::Host::default().with_local_disk_budget(budget.clone()); - let limits = Limits { - max_capture_bytes: 128, - ..Limits::default() - }; - let mut db = Db::open_with_host(&temp.path().join("disk.sqlite"), limits, host).unwrap(); - let ran = std::cell::Cell::new(false); - - let result = db.transaction(|_| { - ran.set(true); - Ok(()) - }); - - assert!(matches!( - result, - Err(CrabError::Limit(crate::LimitKind::LocalDiskBytes)) - )); - assert!(!ran.get()); - assert_eq!(budget.used(), 0); -} - -#[test] -fn existing_database_disk_admission_precedes_session_claim() { - let temp = tempfile::TempDir::new().unwrap(); - let path = temp.path().join("existing.sqlite"); - let connection = Connection::open(&path).unwrap(); - connection.execute_batch("CREATE TABLE t(v)").unwrap(); - drop(connection); - let bytes = std::fs::metadata(&path).unwrap().len(); - let host = crate::Host::default() - .with_local_disk_budget(crate::DiskBudget::new(bytes.saturating_sub(1))); - - let result = Db::open_with_host(&path, Limits::default(), host); - - assert!(matches!( - result, - Err(CrabError::Limit(crate::LimitKind::LocalDiskBytes)) - )); - assert!(!CaptureEngine::meta_path_for(&path).exists()); -} - -#[test] -fn resume_disk_admission_precedes_database_installation() { - let temp = tempfile::TempDir::new().unwrap(); - let source_path = temp.path().join("source.sqlite"); - let limits = Limits::default(); - let mut source = Db::open(&source_path, limits).unwrap(); - source - .transaction(|transaction| { - transaction.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES (1)") - }) - .unwrap(); - let batch = source.capture().unwrap(); - let plan = crate::VerifiedPlan::new(&batch.segments, batch.position, limits).unwrap(); - source.close().unwrap(); - let database_bytes = u64::from(batch.segments[0].info().database_pages) - * u64::from(batch.segments[0].info().page_size); - let host = crate::Host::default() - .with_local_disk_budget(crate::DiskBudget::new(database_bytes.saturating_sub(1))); - let destination = temp.path().join("destination.sqlite"); - - let result = Db::resume_with_host(&plan, &destination, limits, host); - - assert!(matches!( - result, - Err(CrabError::Limit(crate::LimitKind::LocalDiskBytes)) - )); - assert!(!destination.exists()); -} - -#[test] -fn pending_wal_and_captured_segments_reconcile_and_release_disk_admission() { - let temp = tempfile::TempDir::new().unwrap(); - let path = temp.path().join("accounted.sqlite"); - let budget = crate::DiskBudget::new(4 * 1024 * 1024); - let host = crate::Host::default().with_local_disk_budget(budget.clone()); - let limits = Limits { - max_capture_bytes: 1024 * 1024, - ..Limits::default() - }; - let mut db = Db::open_with_host(&path, limits, host).unwrap(); - - db.transaction(|transaction| { - transaction.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES (1)") - }) - .unwrap(); - assert_eq!(budget.used(), 2 * limits.max_capture_bytes); - - let batch = db.capture().unwrap(); - let retained = batch - .segments - .iter() - .map(|segment| segment.info().size_bytes) - .sum::(); - let wal = std::fs::metadata(format!("{}-wal", path.display())) - .unwrap() - .len(); - let database = std::fs::metadata(&path).unwrap().len(); - assert_eq!(budget.used(), database + retained + wal); - - db.close().unwrap(); - assert_eq!(budget.used(), 0); -} - -#[test] -fn typed_operation_error_rolls_back_and_keeps_writer_usable() { - let temp = tempfile::TempDir::new().unwrap(); - let mut db = Db::open(&temp.path().join("typed.sqlite"), Limits::default()).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE inventory(value INTEGER NOT NULL)")) - .unwrap(); - - let rejected = db.transaction_with(|tx| { - tx.execute("INSERT INTO inventory VALUES (1)", []) - .map_err(|_| Rejected)?; - Err::<(), _>(Rejected) - }); - assert!(matches!( - rejected, - Err(crate::TransactionError::Operation(Rejected)) - )); - - db.transaction(|tx| { - tx.execute("INSERT INTO inventory VALUES (2)", []) - .map(|_| ()) - }) - .unwrap(); - let count = db - .writer - .query_row("SELECT count(*) FROM inventory", [], |row| { - row.get::<_, u32>(0) - }) - .unwrap(); - assert_eq!(count, 1); -} - -#[test] -fn truncate_checkpoint_and_auto_vacuum_preserve_every_cut() { - let temp = tempfile::TempDir::new().unwrap(); - let path = temp.path().join("source.sqlite"); - let initial = Connection::open(&path).unwrap(); - initial - .execute_batch("PRAGMA auto_vacuum=FULL; VACUUM;") - .unwrap(); - drop(initial); - let mut db = Db::open(&path, Limits::default()).unwrap(); - db.capture.truncate_page_n = 20; - db.capture.min_checkpoint_page_n = 10; - let mut segments = Vec::new(); - let mut last_pages = 0; - let mut shrank = false; - let mut multiple_cuts = false; - for round in 0..6 { - db.transaction(|tx| { - tx.execute("CREATE TABLE IF NOT EXISTS t (data BLOB)", [])?; - if round % 2 == 0 { - for _ in 0..80 { - tx.execute("INSERT INTO t VALUES (randomblob(8000))", [])?; - } - } else { - tx.execute("DELETE FROM t", [])?; - } - Ok(()) - }) - .unwrap(); - let batch = db.capture().unwrap(); - multiple_cuts |= batch.segments.len() > 1; - for segment in &batch.segments { - shrank |= last_pages > segment.info().database_pages; - last_pages = segment.info().database_pages; - } - segments.extend(batch.segments); - let plan = crate::VerifiedPlan::new(&segments, batch.position, Limits::default()).unwrap(); - let restored = temp.path().join(format!("restored-{round}.sqlite")); - crate::restore_exact(&plan, &restored).unwrap(); - let conn = Connection::open(&restored).unwrap(); - let check: String = conn - .query_row("PRAGMA integrity_check", [], |r| r.get(0)) - .unwrap(); - assert_eq!(check, "ok"); - let count: u32 = conn - .query_row("SELECT count(*) FROM t", [], |r| r.get(0)) - .unwrap(); - assert_eq!(count, if round % 2 == 0 { 80 } else { 0 }); - crate::compact_exact(&plan, &temp.path().join(format!("compacted-{round}.ltx"))).unwrap(); - } - assert!(shrank); - assert!(multiple_cuts); -} - -#[cfg(feature = "replica")] -fn committed_database(path: &Path, limits: Limits) -> Db { - let mut db = Db::open(path, limits).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE payload(value INTEGER)")) - .unwrap(); - db.transaction(|tx| tx.execute("INSERT INTO payload VALUES (7)", [])) - .unwrap(); - db.capture().unwrap(); - db -} - -#[cfg(feature = "replica")] -fn resumed_payload(db: &mut Db) -> i64 { - db.query_with(|connection| { - connection.query_row("SELECT count(*) FROM payload", [], |row| row.get(0)) - }) - .unwrap() -} - -#[cfg(feature = "replica")] -#[test] -fn a_recorded_continuation_continues_the_chain_after_a_move() { - let temp = tempfile::TempDir::new().unwrap(); - let limits = Limits::default(); - let source = temp.path().join("session.sqlite"); - let db = committed_database(&source, limits); - assert!(!db.has_pending_capture()); - db.persist_continuation().unwrap(); - let position = db.position(); - db.close().unwrap(); - - // The capture session directory fences the original path, so a resumed - // activation always installs the file somewhere nothing has claimed. - let host = crate::Host::default(); - let destination = temp.path().join("warm.sqlite"); - crate::resume::move_resumed(&source, &destination, &host).unwrap(); - let mut resumed = Db::open_resumed_with_host(&destination, limits, host).unwrap(); - assert_eq!(resumed.position(), position); - assert_eq!(resumed_payload(&mut resumed), 1); - - resumed - .transaction(|tx| tx.execute("INSERT INTO payload VALUES (8)", [])) - .unwrap(); - let next = resumed.capture().unwrap(); - assert_eq!(next.position.txid, position.txid + 1); - assert_eq!(resumed_payload(&mut resumed), 2); - resumed.close().unwrap(); -} - -#[cfg(feature = "replica")] -#[test] -fn a_resume_refuses_a_continuation_that_does_not_match_the_file() { - let temp = tempfile::TempDir::new().unwrap(); - let limits = Limits::default(); - let source = temp.path().join("mismatched.sqlite"); - let db = committed_database(&source, limits); - db.persist_continuation().unwrap(); - db.close().unwrap(); - - let host = crate::Host::default(); - let corrupt = temp.path().join("corrupt.sqlite"); - crate::resume::move_resumed(&source, &corrupt, &host).unwrap(); - let continuation = crate::resume::continuation_path(&corrupt); - let mut bytes = std::fs::read(&continuation).unwrap(); - let recorded: u32 = bytes[32..36].try_into().map(u32::from_be_bytes).unwrap(); - bytes[32..36].copy_from_slice(&(recorded + 1).to_be_bytes()); - std::fs::write(&continuation, bytes).unwrap(); - assert!(Db::open_resumed_with_host(&corrupt, limits, host.clone()).is_err()); - - // A file that no longer holds the recorded image is refused the same way. - let short = temp.path().join("short.sqlite"); - crate::resume::discard_resumed(&corrupt, &host).unwrap(); - let source = temp.path().join("short-source.sqlite"); - let db = committed_database(&source, limits); - db.persist_continuation().unwrap(); - db.close().unwrap(); - crate::resume::move_resumed(&source, &short, &host).unwrap(); - let file = std::fs::OpenOptions::new() - .write(true) - .open(&short) - .unwrap(); - file.set_len(file.metadata().unwrap().len() - 4096).unwrap(); - drop(file); - assert!(Db::open_resumed_with_host(&short, limits, host).is_err()); -} - -#[cfg(feature = "replica")] -#[test] -fn a_resume_refuses_same_length_database_or_sidecar_corruption() { - use std::io::{Read, Seek, SeekFrom, Write}; - - for corrupt_sidecar in [false, true] { - let temp = tempfile::TempDir::new().unwrap(); - let limits = Limits::default(); - let source = temp.path().join("source.sqlite"); - let db = committed_database(&source, limits); - db.persist_continuation().unwrap(); - db.close().unwrap(); - - let host = crate::Host::default(); - let destination = temp.path().join("corrupt.sqlite"); - crate::resume::move_resumed(&source, &destination, &host).unwrap(); - let corrupt = if corrupt_sidecar { - crate::resume::checksum_path(&destination) - } else { - destination.clone() - }; - let mut file = std::fs::OpenOptions::new() - .read(true) - .write(true) - .open(&corrupt) - .unwrap(); - file.seek(SeekFrom::End(-1)).unwrap(); - let mut byte = [0]; - file.read_exact(&mut byte).unwrap(); - file.seek(SeekFrom::End(-1)).unwrap(); - file.write_all(&[byte[0] ^ 1]).unwrap(); - drop(file); - - assert!( - matches!( - Db::open_resumed_with_host(&destination, limits, host), - Err(crate::CrabError::ChecksumMismatch) - ), - "corrupt_sidecar={corrupt_sidecar}" - ); - } -} - -#[cfg(feature = "replica")] -#[test] -fn a_resume_refuses_a_database_that_is_not_checkpointed() { - let temp = tempfile::TempDir::new().unwrap(); - let limits = Limits::default(); - let source = temp.path().join("dirty.sqlite"); - let db = committed_database(&source, limits); - db.persist_continuation().unwrap(); - db.close().unwrap(); - - // A database whose WAL still holds frames may sit behind the continuation, - // so a resumed open must fall back to the authoritative root instead. - std::fs::write(crate::resume::wal_path(&source), [0u8; 32]).unwrap(); - let host = crate::Host::default(); - let destination = temp.path().join("refused.sqlite"); - assert!(crate::resume::move_resumed(&source, &destination, &host).is_err()); - assert!(!destination.exists()); -} - -#[cfg(feature = "replica")] -#[test] -fn a_dense_checksum_copy_refuses_a_base_that_no_longer_folds_to_it() { - let temp = tempfile::TempDir::new().unwrap(); - let limits = Limits::default(); - let source = temp.path().join("folding.sqlite"); - let db = committed_database(&source, limits); - db.persist_continuation().unwrap(); - db.close().unwrap(); - - let host = crate::Host::default(); - let destination = temp.path().join("reopened.sqlite"); - crate::resume::move_resumed(&source, &destination, &host).unwrap(); - let resumed = Db::open_resumed_with_host(&destination, limits, host).unwrap(); - let checksums = crate::resume::checksum_path(&destination); - let mut bytes = std::fs::read(&checksums).unwrap(); - let wrong = u64::from_be_bytes(bytes[0..8].try_into().unwrap()) ^ 1; - bytes[0..8].copy_from_slice(&wrong.to_be_bytes()); - std::fs::write(&checksums, bytes).unwrap(); - assert!(resumed.persist_continuation().is_err()); -} diff --git a/crates/crab-ltx/src/environment.rs b/crates/crab-ltx/src/environment.rs deleted file mode 100644 index 3d7a91228..000000000 --- a/crates/crab-ltx/src/environment.rs +++ /dev/null @@ -1,30 +0,0 @@ -//! Injectable local I/O, clocks and jobs adapted from Celld host.rs. -//! -//! Object-store-only resources, telemetry, directory cache, and executors are -//! gated at the module boundary instead of per item. - -#[cfg(feature = "replica")] -pub mod directory_cache; -#[cfg(feature = "replica")] -pub mod executor; -pub mod host; -#[cfg(feature = "replica")] -pub mod resources; -#[cfg(feature = "replica")] -pub mod telemetry; - -#[cfg(feature = "replica")] -pub use directory_cache::DirectoryCacheStats; -#[cfg(feature = "replica")] -pub use executor::{Executor, Worker}; -pub use host::{ - Clock, DirectFileSystem, DiskBudget, DiskBudgetAdmission, DiskReservation, FileIo, FileSystem, - Host, SystemClock, -}; -#[cfg(feature = "replica")] -pub use resources::{HostResourceAdmission, HostResourceKind, HostResourcePermit}; -#[cfg(feature = "replica")] -pub use telemetry::{LtxPhase, LtxReadOrigin, LtxRequestOutcome, LtxTelemetry, ScratchMonitor}; - -#[cfg(test)] -mod tests; diff --git a/crates/crab-ltx/src/environment/directory_cache.rs b/crates/crab-ltx/src/environment/directory_cache.rs deleted file mode 100644 index 9f09bd742..000000000 --- a/crates/crab-ltx/src/environment/directory_cache.rs +++ /dev/null @@ -1,571 +0,0 @@ -//! Object-store-only verified directory cache. - -use super::*; -use std::{ - io, - path::{Path, PathBuf}, - sync::{ - Arc, Mutex, - atomic::{AtomicU64, Ordering}, - }, -}; - -#[cfg(feature = "replica")] -use std::collections::{BTreeMap, HashMap, HashSet, VecDeque}; - -#[derive(Default)] -pub(super) struct CacheFills { - state: Mutex, - finished: tokio::sync::Notify, -} - -#[derive(Default)] -struct CacheFillState { - paths: HashSet, - opening: HashSet, -} - -impl CacheFills { - pub(super) fn claim(self: &Arc, path: PathBuf) -> Option { - let mut state = self.state.lock().unwrap_or_else(|error| error.into_inner()); - if path - .parent() - .is_some_and(|root| state.opening.contains(root)) - || !state.paths.insert(path.clone()) - { - return None; - } - Some(CacheFill { - fills: Arc::clone(self), - path, - opening: false, - }) - } - - pub(super) async fn open(self: &Arc, root: PathBuf) -> CacheFill { - loop { - let finished = self.finished.notified(); - { - let mut state = self.state.lock().unwrap_or_else(|error| error.into_inner()); - if !state.opening.contains(&root) - && !state.paths.iter().any(|path| path.parent() == Some(&root)) - { - state.opening.insert(root.clone()); - return CacheFill { - fills: Arc::clone(self), - path: root, - opening: true, - }; - } - } - finished.await; - } - } - - pub(super) async fn drain(&self) { - loop { - // Register before checking: completion between the check and await - // must wake shutdown even when no later fill will finish. - let finished = self.finished.notified(); - { - let state = self.state.lock().unwrap_or_else(|error| error.into_inner()); - if state.paths.is_empty() && state.opening.is_empty() { - return; - } - } - finished.await; - } - } -} - -pub(super) struct CacheFill { - fills: Arc, - path: PathBuf, - opening: bool, -} - -impl Drop for CacheFill { - fn drop(&mut self) { - let mut state = self - .fills - .state - .lock() - .unwrap_or_else(|error| error.into_inner()); - if self.opening { - state.opening.remove(&self.path); - } else { - state.paths.remove(&self.path); - } - drop(state); - self.fills.finished.notify_waiters(); - } -} - -#[cfg(feature = "replica")] -#[derive(serde::Serialize, serde::Deserialize)] -pub(crate) struct DirectoryCacheIndex { - version: u8, - entries: BTreeMap, -} - -#[cfg(feature = "replica")] -pub(crate) struct DirectoryCacheState { - entries: BTreeMap, - order: VecDeque, - bytes: u64, - reservations: BTreeMap, -} - -#[cfg(feature = "replica")] -/// Point-in-time usage of the verified immutable directory-node cache. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct DirectoryCacheStats { - entries: usize, - bytes: u64, - capacity_bytes: u64, -} - -#[cfg(feature = "replica")] -impl DirectoryCacheStats { - /// Returns the number of indexed cache entries. - #[must_use] - pub const fn entries(self) -> usize { - self.entries - } - - /// Returns bytes occupied by verified cache entries. - #[must_use] - pub const fn bytes(self) -> u64 { - self.bytes - } - - /// Returns the cache's byte ceiling. - #[must_use] - pub const fn capacity_bytes(self) -> u64 { - self.capacity_bytes - } -} - -#[cfg(feature = "replica")] -pub(crate) struct DirectoryCache { - filesystem: Arc, - pub(super) budget: DiskBudget, - root: PathBuf, - index: PathBuf, - max_bytes: u64, - state: Mutex, - fills: Mutex>>>, -} - -#[cfg(feature = "replica")] -pub(crate) const MAX_DIRECTORY_CACHE_ENTRIES: usize = 16_384; -#[cfg(feature = "replica")] -const MAX_DIRECTORY_CACHE_INDEX_BYTES: u64 = 16 << 20; - -#[cfg(feature = "replica")] -impl DirectoryCache { - #[cfg(test)] - pub(crate) fn new(filesystem: Arc, root: PathBuf, max_bytes: u64) -> Self { - Self::with_budget(filesystem, root, max_bytes, DiskBudget::new(max_bytes)) - } - - pub(crate) fn with_budget( - filesystem: Arc, - root: PathBuf, - max_bytes: u64, - budget: DiskBudget, - ) -> Self { - let _ = filesystem.cleanup_private_temporaries(&root); - let index = root.join("index-v1.json"); - let entries = filesystem - .open(&index) - .and_then(|mut file| { - let length = file.file_len()?; - if length > MAX_DIRECTORY_CACHE_INDEX_BYTES { - return Err(io::Error::new( - io::ErrorKind::InvalidData, - "cache index too large", - )); - } - let length = usize::try_from(length).map_err(io::Error::other)?; - let bytes = file.read_exact_at(0, length)?; - let index: DirectoryCacheIndex = - serde_json::from_slice(&bytes).map_err(io::Error::other)?; - if index.version != 1 { - return Err(io::Error::new( - io::ErrorKind::InvalidData, - "unsupported Cell directory cache index", - )); - } - Ok(index.entries) - }) - .unwrap_or_default(); - let mut retained = BTreeMap::new(); - let mut bytes = 0u64; - let mut reservations = BTreeMap::new(); - for (key, length) in entries { - let path = root.join(hex_digest(blake3::hash(key.as_bytes()).as_bytes())); - let valid = length != 0 - && length <= max_bytes - && retained.len() < MAX_DIRECTORY_CACHE_ENTRIES - && bytes <= max_bytes.saturating_sub(length) - && filesystem.exists(&path).ok() == Some(true) - && filesystem.file_len(&path).ok() == Some(length) - && safe_cache_entry(&filesystem, &root, &path).ok() == Some(true); - let reservation = valid.then(|| budget.try_reserve(length).ok()).flatten(); - if let Some(reservation) = reservation { - // The admission check proves this addition fits the cache cap. - // Re-summing earlier entries makes restart quadratic in count. - bytes += length; - retained.insert(key.clone(), length); - reservations.insert(key, reservation); - } else { - let _ = filesystem.remove_file(&path); - } - } - let order = retained.keys().cloned().collect(); - Self { - filesystem, - budget, - root, - index, - max_bytes, - state: Mutex::new(DirectoryCacheState { - entries: retained, - order, - bytes, - reservations, - }), - fills: Mutex::new(HashMap::new()), - } - } - - pub(crate) fn key_path(&self, key: &str) -> PathBuf { - let digest = blake3::hash(key.as_bytes()); - self.root.join(hex_digest(digest.as_bytes())) - } - - pub(crate) fn stats(&self) -> DirectoryCacheStats { - let state = match self.state.lock() { - Ok(state) => state, - Err(poisoned) => poisoned.into_inner(), - }; - DirectoryCacheStats { - entries: state.entries.len(), - bytes: state.bytes, - capacity_bytes: self.max_bytes, - } - } - - pub(crate) fn fill_lock(&self, key: &str) -> Arc> { - let mut fills = match self.fills.lock() { - Ok(fills) => fills, - Err(poisoned) => poisoned.into_inner(), - }; - if let Some(lock) = fills.get(key) { - return Arc::clone(lock); - } - let lock = Arc::new(Mutex::new(())); - if fills.len() < MAX_DIRECTORY_CACHE_ENTRIES { - fills.insert(key.to_owned(), Arc::clone(&lock)); - } - lock - } - - pub(crate) fn get(&self, key: &str, max_bytes: u64) -> io::Result>> { - let lock = self.fill_lock(key); - let _guard = match lock.try_lock() { - Ok(guard) => guard, - Err(std::sync::TryLockError::WouldBlock) => return Ok(None), - // Interrupted optional fills cannot poison canonical reads. File - // shape and the caller's digest check still authenticate every hit. - Err(std::sync::TryLockError::Poisoned(error)) => error.into_inner(), - }; - let path = self.key_path(key); - if !self.filesystem.exists(&path)? { - if self.remove_entry(key) { - self.persist_index(); - } - return Ok(None); - } - if !safe_cache_entry(&self.filesystem, &self.root, &path)? { - let _ = self.filesystem.remove_file(&path); - self.remove_entry(key); - self.persist_index(); - return Ok(None); - } - let length = self.filesystem.file_len(&path)?; - if length == 0 || length > max_bytes || length > self.max_bytes { - let _ = self.filesystem.remove_file(&path); - self.remove_entry(key); - self.persist_index(); - return Ok(None); - } - let indexed_length = match self.state.lock() { - Ok(state) => state.entries.get(key).copied(), - Err(poisoned) => poisoned.into_inner().entries.get(key).copied(), - }; - if indexed_length != Some(length) { - let _ = self.filesystem.remove_file(&path); - self.remove_entry(key); - self.persist_index(); - return Ok(None); - } - let mut file = match self.filesystem.open(&path) { - Ok(file) => file, - Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(None), - Err(error) => return Err(error), - }; - let bytes = match file.read_exact_at(0, usize::try_from(length).map_err(io::Error::other)?) - { - Ok(bytes) => bytes, - Err(error) - if matches!( - error.kind(), - io::ErrorKind::UnexpectedEof - | io::ErrorKind::InvalidData - | io::ErrorKind::NotFound - ) => - { - let _ = self.filesystem.remove_file(&path); - self.remove_entry(key); - self.persist_index(); - return Ok(None); - } - Err(error) => return Err(error), - }; - if bytes.len() as u64 != length { - return Err(io::Error::new( - io::ErrorKind::UnexpectedEof, - "directory cache entry was truncated", - )); - } - self.touch_entry(key, length); - // Recency lives only in memory. Persisting an unchanged membership map - // turns a verified disk-cache hit into an unnecessary durable write. - Ok(Some(bytes)) - } - - pub(crate) fn put(&self, key: &str, bytes: &[u8], max_entry: u64) -> io::Result<()> { - if bytes.is_empty() || bytes.len() as u64 > max_entry || bytes.len() as u64 > self.max_bytes - { - return Ok(()); - } - let lock = self.fill_lock(key); - let _guard = lock.lock().unwrap_or_else(|error| error.into_inner()); - self.filesystem.create_dir_all(&self.root)?; - let path = self.key_path(key); - if self.filesystem.exists(&path)? { - let length = self.filesystem.file_len(&path)?; - let indexed = self - .state - .lock() - .unwrap_or_else(|error| error.into_inner()) - .entries - .get(key) - .copied() - == Some(length); - // A previous fill may have installed bytes before its index write - // failed. Unindexed files need a fresh disk reservation on reuse. - if length == bytes.len() as u64 && indexed { - self.touch_entry(key, length); - return Ok(()); - } - let _ = self.filesystem.remove_file(&path); - self.remove_entry(key); - } - self.make_room(bytes.len() as u64)?; - let reservation = self - .budget - .try_reserve(bytes.len() as u64) - .map_err(|error| io::Error::other(error.to_string()))?; - let temporary = self.root.join(format!( - ".tmp-{}-{}", - std::process::id(), - NEXT_CACHE_TEMP.fetch_add(1, Ordering::Relaxed) - )); - let mut file = match self.filesystem.create(&temporary) { - Ok(file) => file, - Err(error) => { - drop(reservation); - return Err(error); - } - }; - if let Err(error) = file.write_all(bytes).and_then(|()| file.sync_all()) { - let _ = self.filesystem.remove_file(&temporary); - drop(reservation); - return Err(error); - } - drop(file); - if let Err(error) = self.filesystem.rename(&temporary, &path) { - let _ = self.filesystem.remove_file(&temporary); - drop(reservation); - return Err(error); - } - self.touch_entry_with_reservation(key, bytes.len() as u64, reservation); - self.evict()?; - self.persist_index(); - Ok(()) - } - - pub(crate) fn invalidate(&self, key: &str) -> io::Result<()> { - let lock = self.fill_lock(key); - let _guard = lock.lock().unwrap_or_else(|error| error.into_inner()); - let path = self.key_path(key); - match self.filesystem.remove_file(&path) { - Ok(()) => {} - Err(error) if error.kind() == io::ErrorKind::NotFound => {} - Err(error) => return Err(error), - } - self.remove_entry(key); - self.persist_index(); - Ok(()) - } - - pub(crate) fn touch_entry(&self, key: &str, length: u64) { - let mut state = match self.state.lock() { - Ok(state) => state, - Err(poisoned) => poisoned.into_inner(), - }; - if let Some(previous) = state.entries.insert(key.to_owned(), length) { - state.bytes = state.bytes.saturating_sub(previous); - state.order.retain(|entry| entry != key); - } - state.bytes = state.bytes.saturating_add(length); - state.order.push_back(key.to_owned()); - } - - pub(crate) fn touch_entry_with_reservation( - &self, - key: &str, - length: u64, - reservation: DiskReservation, - ) { - let mut state = match self.state.lock() { - Ok(state) => state, - Err(poisoned) => poisoned.into_inner(), - }; - if let Some(previous) = state.entries.insert(key.to_owned(), length) { - state.bytes = state.bytes.saturating_sub(previous); - state.order.retain(|entry| entry != key); - } - state.bytes = state.bytes.saturating_add(length); - state.order.push_back(key.to_owned()); - state.reservations.insert(key.to_owned(), reservation); - } - - pub(crate) fn remove_entry(&self, key: &str) -> bool { - let mut state = match self.state.lock() { - Ok(state) => state, - Err(poisoned) => poisoned.into_inner(), - }; - let previous = state.entries.remove(key); - if let Some(previous) = previous { - state.bytes = state.bytes.saturating_sub(previous); - } - state.reservations.remove(key); - state.order.retain(|entry| entry != key); - previous.is_some() - } - - pub(crate) fn make_room(&self, required: u64) -> io::Result<()> { - loop { - let victim = { - let state = match self.state.lock() { - Ok(state) => state, - Err(poisoned) => poisoned.into_inner(), - }; - (state.bytes.saturating_add(required) > self.max_bytes - || state.entries.len() >= MAX_DIRECTORY_CACHE_ENTRIES) - .then(|| state.order.front().cloned()) - }; - let Some(Some(key)) = victim else { - return Ok(()); - }; - let path = self.key_path(&key); - match self.filesystem.remove_file(&path) { - Ok(()) => {} - Err(error) if error.kind() == io::ErrorKind::NotFound => {} - Err(error) => return Err(error), - } - self.remove_entry(&key); - } - } - - pub(crate) fn evict(&self) -> io::Result<()> { - loop { - let victim = { - let state = match self.state.lock() { - Ok(state) => state, - Err(poisoned) => poisoned.into_inner(), - }; - (state.bytes > self.max_bytes || state.entries.len() > MAX_DIRECTORY_CACHE_ENTRIES) - .then(|| state.order.front().cloned()) - }; - let Some(Some(key)) = victim else { - return Ok(()); - }; - let path = self.key_path(&key); - match self.filesystem.remove_file(&path) { - Ok(()) => {} - Err(error) if error.kind() == io::ErrorKind::NotFound => {} - Err(error) => return Err(error), - } - self.remove_entry(&key); - } - } - - pub(crate) fn persist_index(&self) { - let entries = match self.state.lock() { - Ok(state) => state.entries.clone(), - Err(poisoned) => poisoned.into_inner().entries.clone(), - }; - if self.filesystem.create_dir_all(&self.root).is_err() { - return; - } - let Ok(bytes) = serde_json::to_vec(&DirectoryCacheIndex { - version: 1, - entries, - }) else { - return; - }; - let temporary = self.root.join(format!( - ".index-tmp-{}-{}", - std::process::id(), - NEXT_CACHE_TEMP.fetch_add(1, Ordering::Relaxed) - )); - let Ok(mut file) = self.filesystem.create(&temporary) else { - return; - }; - if file.write_all(&bytes).is_err() || file.sync_all().is_err() { - let _ = self.filesystem.remove_file(&temporary); - return; - } - drop(file); - let _ = self.filesystem.rename(&temporary, &self.index); - } -} - -#[cfg(feature = "replica")] -fn safe_cache_entry( - filesystem: &Arc, - root: &Path, - path: &Path, -) -> io::Result { - let canonical_root = filesystem.canonicalize(root)?; - let canonical_path = filesystem.canonicalize(path)?; - Ok(canonical_path.parent() == Some(canonical_root.as_path()) - && canonical_path.file_name() == path.file_name()) -} - -#[cfg(feature = "replica")] -pub(crate) static NEXT_CACHE_TEMP: AtomicU64 = AtomicU64::new(1); - -#[cfg(feature = "replica")] -fn hex_digest(bytes: &[u8; 32]) -> String { - let mut output = String::with_capacity(64); - for byte in bytes { - output.push_str(&format!("{byte:02x}")); - } - output -} diff --git a/crates/crab-ltx/src/environment/executor.rs b/crates/crab-ltx/src/environment/executor.rs deleted file mode 100644 index 639d2edff..000000000 --- a/crates/crab-ltx/src/environment/executor.rs +++ /dev/null @@ -1,54 +0,0 @@ -//! Object-store-only executor and worker contracts. - -use std::io; - -/// Blocking dispatch boundary; success means the job was accepted for execution. -/// -/// The dispatcher must eventually run or drop the job. Dropped jobs and panics -/// become errors to the awaiting operation; cancellation does not undo side effects. -#[cfg(feature = "replica")] -pub trait Executor: Send + Sync { - /// Accepts one blocking job; a dropped job becomes an error to its waiter. - fn dispatch(&self, job: Box) -> io::Result<()>; - /// Starts a long-lived worker independently of the caller and dispatch pool. - /// - /// Must not queue behind the blocking caller: SQLite waits synchronously - /// for this worker. The worker drives its own Tokio I/O runtime until closed. - fn start_worker(&self, job: Box) -> io::Result>; -} - -/// An independently progressing worker joined after its input queue closes. -#[cfg(feature = "replica")] -pub trait Worker: Send + Sync { - /// Waits for the worker after its input queue closed; a panic on the - /// worker thread becomes an error here. - fn join(self: Box) -> io::Result<()>; -} - -#[cfg(feature = "replica")] -pub(crate) struct TokioExecutor; -#[cfg(feature = "replica")] -impl Executor for TokioExecutor { - fn dispatch(&self, job: Box) -> io::Result<()> { - tokio::runtime::Handle::try_current() - .map_err(io::Error::other)? - .spawn_blocking(job); - Ok(()) - } - fn start_worker(&self, job: Box) -> io::Result> { - Ok(Box::new( - std::thread::Builder::new() - .name("crab-ltx-paged".into()) - .spawn(job)?, - )) - } -} - -#[cfg(feature = "replica")] -impl Worker for std::thread::JoinHandle<()> { - fn join(self: Box) -> io::Result<()> { - (*self) - .join() - .map_err(|_| io::Error::other("host worker panicked")) - } -} diff --git a/crates/crab-ltx/src/environment/host.rs b/crates/crab-ltx/src/environment/host.rs deleted file mode 100644 index 4c4e6e3ad..000000000 --- a/crates/crab-ltx/src/environment/host.rs +++ /dev/null @@ -1,1071 +0,0 @@ -//! Disk admission, filesystem, clock, and host contracts. - -use std::{ - fmt, io, - path::{Path, PathBuf}, - sync::{ - Arc, Mutex, MutexGuard, - atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}, - }, - time::{Duration, Instant, SystemTime, UNIX_EPOCH}, -}; - -mod budget; - -pub use budget::{DiskBudget, DiskReservation}; - -#[cfg(feature = "replica")] -use crate::environment::directory_cache::{CacheFills, DirectoryCache, DirectoryCacheStats}; -#[cfg(feature = "replica")] -use crate::environment::executor::{Executor, TokioExecutor}; -#[cfg(feature = "replica")] -use crate::environment::resources::{HostResourceAdmission, HostResourceKind, HostResourcePermit}; -#[cfg(feature = "replica")] -use crate::environment::telemetry::{LtxPhase, LtxReadOrigin, LtxRequestOutcome, LtxTelemetry}; -#[cfg(feature = "replica")] -use crate::environment::telemetry::{ScratchMonitor, UnlimitedScratch}; - -/// Admission hook used by an embedding runtime to charge local bytes to its -/// node-wide resource ledger. -/// -/// The hook is called synchronously with the exact aggregate budget usage for -/// every reserve, resize, release, and late installation operation. -pub trait DiskBudgetAdmission: Send + Sync { - /// Reconciles the exact aggregate bytes currently reserved by this budget. - fn reconcile(&self, bytes: u64) -> crate::Result<()>; - - /// Reports whether this admission owner is still alive. - fn is_live(&self) -> bool { - true - } -} - -/// An open local artifact/WAL handle supplied by a host filesystem. -/// -/// Positional reads and writes use their explicit offsets. A handle returned by -/// `FileSystem::open_rw` supports both operations. An open handle remains bound -/// to the selected artifact even if its namespace path is later replaced. -pub trait FileIo: Send { - /// Writes `bytes` at the handle's current offset. - fn write_all(&mut self, bytes: &[u8]) -> io::Result<()>; - /// Writes `bytes` at an explicit offset without moving the handle's offset. - fn write_all_at(&mut self, offset: u64, bytes: &[u8]) -> io::Result<()>; - /// Reads exactly `len` bytes from an explicit offset. - fn read_exact_at(&mut self, offset: u64, len: usize) -> io::Result>; - /// Makes every prior write to this handle durable. - fn sync_all(&mut self) -> io::Result<()>; - /// Returns the current length of the file in bytes. - fn file_len(&self) -> io::Result; - /// Truncates or extends the file to `len` bytes. - fn set_len(&mut self, len: u64) -> io::Result<()>; -} - -/// Local filesystem boundary; SQLite pager I/O remains under its selected VFS. -/// -/// `create` must exclusively create a new file. `open_rw` must not create. -/// `rename` must sync the destination parent before succeeding. The opt-in -/// `rename_uncommitted` variant may install a file before its contents or name -/// are durable; the caller must sync the file and then its parent directory. -/// Implementations must preserve underlying I/O errors. -/// `exists` must detect dangling symlinks. `create_dir` is an exclusive claim. -/// `persist_new` atomically installs fully synced bytes without replacing any -/// destination and syncs its parent; `persist_file_new` does the same for an -/// already synced same-directory scratch file. An error after installation is -/// ambiguous. -pub trait FileSystem: Send + Sync { - /// Opens an existing file for reading. - fn open(&self, path: &Path) -> io::Result>; - /// Opens an existing file for reading and writing; never creates it. - fn open_rw(&self, path: &Path) -> io::Result>; - /// Exclusively creates a new file for writing. - fn create(&self, path: &Path) -> io::Result>; - /// Returns the length of an existing file in bytes. - fn file_len(&self, path: &Path) -> io::Result; - /// Creates the directory and every missing parent. - fn create_dir_all(&self, path: &Path) -> io::Result<()>; - /// Renames within the namespace and syncs the destination parent. - fn rename(&self, from: &Path, to: &Path) -> io::Result<()>; - - /// Atomically renames a file without requiring the destination directory - /// to be durable yet. The default preserves the synchronous `rename` - /// contract for host filesystems that do not support batching. - fn rename_uncommitted(&self, from: &Path, to: &Path) -> io::Result<()> { - self.rename(from, to) - } - - /// Removes one file; a missing file is an error. - fn remove_file(&self, path: &Path) -> io::Result<()>; - /// Resolves a path against the filesystem's canonical namespace root. - fn canonicalize(&self, path: &Path) -> io::Result; - /// Reports whether the path exists, including a dangling symlink. - fn exists(&self, path: &Path) -> io::Result; - /// Exclusively creates one directory. - fn create_dir(&self, path: &Path) -> io::Result<()>; - - /// Syncs every named file before a shared directory barrier. - /// - /// Hosts may coalesce or parallelize these independent flushes. Success - /// must still mean that every file's contents are durable. - fn sync_files(&self, paths: &[PathBuf]) -> io::Result<()> { - for path in paths { - self.open_rw(path)?.sync_all()?; - } - Ok(()) - } - - /// Syncs the parent directory of `path`. - fn sync_parent(&self, path: &Path) -> io::Result<()>; - /// Atomically installs `bytes` as a new file and syncs its parent. - fn persist_new(&self, path: &Path, bytes: &[u8]) -> io::Result<()>; - /// Atomically installs an already synced same-directory scratch file. - fn persist_file_new(&self, source: &Path, destination: &Path) -> io::Result<()>; - - /// Removes abandoned private cache temporaries below `root`. - /// - /// Host filesystems that cannot enumerate a private directory may leave - /// this as a no-op; the cache remains fail-closed because only indexed, - /// canonical entries are ever read. - fn cleanup_private_temporaries(&self, _root: &Path) -> io::Result<()> { - Ok(()) - } -} - -/// Wall-clock observations used in LTX timestamps and checkpoint eligibility. -pub trait Clock: Send + Sync { - /// Returns wall-clock milliseconds since the Unix epoch. - fn unix_millis(&self) -> i64; - /// Returns how long ago the file was last modified. - fn file_age(&self, path: &Path) -> io::Result; - - /// Returns a monotonic instant for observational duration measurements. - /// - /// Implementors that only provide wall-clock behavior can keep this - /// default; tests may override it with a deterministic clock. - fn monotonic(&self) -> Instant { - Instant::now() - } -} - -/// Cloneable host facilities; defaults retain standard filesystem and Tokio behavior. -/// -/// The filesystem and selected SQLite VFS must address the same namespace. -/// Injecting local facilities does not replace object-store transport. -#[derive(Clone)] -pub struct Host { - pub(crate) filesystem: Arc, - pub(crate) clock: Arc, - pub(crate) sqlite_vfs: Option, - pub(crate) local_disk: DiskBudget, - #[cfg(feature = "replica")] - pub(crate) executor: Arc, - #[cfg(feature = "replica")] - pub(crate) paged_driver: crate::paged_io::DriverSlot, - #[cfg(feature = "replica")] - io_slots: Arc, - #[cfg(feature = "replica")] - io_capacity: usize, - #[cfg(feature = "replica")] - job_slots: Arc, - #[cfg(feature = "replica")] - job_capacity: usize, - #[cfg(feature = "replica")] - recovery_slots: Arc, - #[cfg(feature = "replica")] - recovery_capacity: usize, - #[cfg(feature = "replica")] - dirty_slots: Arc, - #[cfg(feature = "replica")] - dirty_capacity: usize, - #[cfg(feature = "replica")] - scratch_slots: Arc, - #[cfg(feature = "replica")] - scratch_capacity: u32, - #[cfg(feature = "replica")] - scratch_monitor: Arc, - #[cfg(feature = "replica")] - directory_cache: Option>, - #[cfg(feature = "replica")] - cache_fills: Arc, - #[cfg(feature = "replica")] - resource_admission: Option>, - #[cfg(feature = "replica")] - telemetry: Option>, - #[cfg(feature = "replica")] - recovery: Option>, - #[cfg(feature = "replica")] - dirty: Option>, - #[cfg(feature = "replica")] - scratch: Option>, - #[cfg(feature = "replica")] - recovery_resource: Option>, - #[cfg(feature = "replica")] - dirty_resource: Option>, - #[cfg(feature = "replica")] - scratch_resource: Option>, -} - -#[cfg(feature = "replica")] -pub(crate) struct HostIoPermit { - _semaphore: tokio::sync::OwnedSemaphorePermit, - _resource: Option>, -} - -impl Host { - /// Charges local-disk reservations to one embedding runtime ledger. - pub fn install_disk_admission( - &self, - admission: Arc, - ) -> crate::Result<()> { - self.local_disk.install_admission(admission) - } - - /// Verifies named local artifacts using this host's bounded filesystem reads. - pub fn verify( - &self, - segments: &[crate::LocalSegment], - target: crate::Position, - limits: crate::Limits, - ) -> crate::Result { - crate::VerifiedPlan::with_host(segments, target, limits, self) - } - - /// Restores a verified cut through this host's atomic new-file installation. - pub fn restore( - &self, - plan: &crate::VerifiedPlan, - destination: &Path, - ) -> crate::Result { - self.restore_materialized(plan.materialize(), destination) - } - - pub(crate) fn restore_materialized( - &self, - materialized: &crate::recovery::MaterializedPlan, - destination: &Path, - ) -> crate::Result { - crate::recovery::reject_sidecars(destination, self)?; - self.filesystem - .persist_new(destination, &materialized.image)?; - Ok(materialized.position) - } - - /// Installs a verified full-chain compaction through this host's filesystem. - pub fn compact( - &self, - plan: &crate::VerifiedPlan, - destination: &Path, - ) -> crate::Result { - crate::recovery::compact_to_file(self, plan, destination) - } - - /// Selects an already registered SQLite VFS for local and sparse databases. - /// - /// The host must keep the registration alive for the process lifetime and - /// make its file namespace agree with `FileSystem`. Unknown names fail closed. - #[must_use] - pub fn with_sqlite_vfs(mut self, name: &str) -> Self { - self.sqlite_vfs = Some(name.to_owned()); - self - } - - /// Shares byte-precise admission across WAL, LTX, sparse pages and staging. - #[must_use] - pub fn with_local_disk_budget(mut self, budget: DiskBudget) -> Self { - self.local_disk = budget; - self - } - - /// Enables the verified immutable directory-node cache below `root`. - /// - /// The cache is an acceleration layer only; directory reachability still - /// reads canonical objects when collecting retention roots. Its byte bound - /// is derived from one eighth of the shared local-disk envelope. Construction - /// uses an admitted blocking job; cancellation retains its disk reservations - /// and job admission until that dispatched work completes. - /// Admission or dispatch failure returns an error. Invalid optional cache - /// membership is ignored and subsequent reads verify origin objects. - #[cfg(feature = "replica")] - pub async fn with_directory_cache(mut self, root: PathBuf) -> crate::Result { - let capacity = (self.local_disk.capacity() / 8).clamp(1, 8 << 30); - let filesystem = Arc::clone(&self.filesystem); - let budget = self.local_disk.clone(); - // Reopening cleans private temporaries. An earlier activation's fill - // may outlive its reader, so exclude fills until construction finishes. - let opening = self.cache_fills.open(root.clone()).await; - let (cache, opening) = self - .run(move || { - let cache = Arc::new(DirectoryCache::with_budget( - filesystem, root, capacity, budget, - )); - (cache, opening) - }) - .await?; - self.directory_cache = Some(cache); - drop(opening); - Ok(self) - } - - /// Waits for accepted cache fills and cache opens on all clones. - /// - /// Call after stopping replica work and before releasing its local directories - /// or executor. Cancellation leaves accepted jobs running; another call can - /// finish the drain. Cache failures never change canonical object durability. - #[cfg(feature = "replica")] - pub async fn drain_cache_fills(&self) { - self.cache_fills.drain().await; - } - - /// Installs one embedding runtime ledger for bounded replica-host work. - #[cfg(feature = "replica")] - pub fn install_resource_admission(&mut self, admission: Arc) { - self.resource_admission = Some(admission); - } - - /// Sends finite replica observations to one embedding runtime. - #[cfg(feature = "replica")] - #[must_use] - pub fn with_ltx_telemetry(mut self, telemetry: Arc) -> Self { - self.telemetry = Some(telemetry); - self - } - - #[cfg(feature = "replica")] - pub(crate) fn observe_ltx_phase(&self, phase: LtxPhase, started: Instant, succeeded: bool) { - if let Some(telemetry) = &self.telemetry { - telemetry.phase( - phase, - self.now_monotonic().saturating_duration_since(started), - succeeded, - ); - } - } - - #[cfg(feature = "replica")] - pub(crate) fn observe_ltx_logical_read(&self, origin: LtxReadOrigin) { - if let Some(telemetry) = &self.telemetry { - telemetry.logical_read(origin); - } - } - - #[cfg(feature = "replica")] - pub(crate) fn observe_ltx_origin_request( - &self, - origin: LtxReadOrigin, - succeeded: bool, - bytes: usize, - ) { - if let Some(telemetry) = &self.telemetry { - telemetry.origin_request( - origin, - if succeeded { - LtxRequestOutcome::Succeeded - } else { - LtxRequestOutcome::Failed - }, - bytes as u64, - ); - } - } - - #[cfg(feature = "replica")] - pub(crate) fn observe_ltx_capture(&self, timing: &crate::CaptureTiming, succeeded: bool) { - if let Some(telemetry) = &self.telemetry { - telemetry.capture(timing, succeeded); - } - } - - /// Returns the currently configured object-store I/O capacity. - #[cfg(feature = "replica")] - #[must_use] - pub fn io_capacity(&self) -> usize { - self.io_capacity - } - - /// Returns the currently configured blocking-job capacity. - #[cfg(feature = "replica")] - #[must_use] - pub fn job_capacity(&self) -> usize { - self.job_capacity - } - - /// Returns the currently configured recovery-job capacity. - #[cfg(feature = "replica")] - #[must_use] - pub fn recovery_capacity(&self) -> usize { - self.recovery_capacity - } - - /// Returns the currently configured dirty-memory capacity. - #[cfg(feature = "replica")] - #[must_use] - pub fn dirty_capacity(&self) -> usize { - self.dirty_capacity - } - - /// Returns the currently configured scratch capacity in MiB units. - #[cfg(feature = "replica")] - #[must_use] - pub const fn scratch_capacity(&self) -> u32 { - self.scratch_capacity - } - - /// Returns the configured byte ceiling shared by local replica artifacts. - #[must_use] - pub fn local_disk_capacity(&self) -> u64 { - self.local_disk.capacity() - } - - /// Returns bytes currently reserved by local replica artifacts. - #[must_use] - pub fn local_disk_used(&self) -> u64 { - self.local_disk.used() - } - - /// Returns the shared local-disk budget used by replica artifacts. - #[must_use] - pub fn local_disk_budget(&self) -> DiskBudget { - self.local_disk.clone() - } - - /// Returns verified directory-cache usage when the cache is enabled. - #[cfg(feature = "replica")] - #[must_use] - pub fn directory_cache_stats(&self) -> Option { - self.directory_cache.as_ref().map(|cache| cache.stats()) - } - - pub(crate) fn reserve_local_disk(&self, bytes: u64) -> crate::Result { - self.local_disk.try_reserve(bytes) - } - - pub(crate) fn now_monotonic(&self) -> Instant { - self.clock.monotonic() - } - - pub(crate) fn read(&self, path: &Path, limit: u64) -> io::Result> { - let host = crate::host::LtxHost { - facilities: self.clone(), - max_database_bytes: limit, - max_file_bytes: limit, - }; - host.read(path) - } - /// Replaces the host filesystem; it must address the same namespace as - /// the selected SQLite VFS. - #[must_use] - pub fn with_filesystem(mut self, filesystem: Arc) -> Self { - self.filesystem = filesystem; - self - } - /// Replaces the host clock used for LTX timestamps. - #[must_use] - pub fn with_clock(mut self, clock: Arc) -> Self { - self.clock = clock; - self - } - /// Replaces the blocking executor that drives capture and compaction. - #[cfg(feature = "replica")] - #[must_use] - pub fn with_executor(mut self, executor: Arc) -> Self { - self.executor = executor; - self.paged_driver = Arc::new(std::sync::Mutex::new(std::sync::Weak::new())); - self - } - - /// Shares an object-store request ceiling across hosts and databases. - /// - /// Defaults share 32 permits process-wide. Closing the semaphore rejects - /// new I/O; permits are held across provider retries and released on drop. - #[cfg(feature = "replica")] - #[must_use] - pub fn with_io_slots(mut self, slots: Arc) -> Self { - self.io_capacity = slots.available_permits(); - self.io_slots = slots; - self - } - - /// Shares a blocking-job ceiling, including jobs whose callers cancel. - /// - /// Defaults share up to 16 jobs process-wide, capped by available CPUs. - /// Independent paged workers do not consume these slots, avoiding deadlock. - #[cfg(feature = "replica")] - #[must_use] - pub fn with_job_slots(mut self, slots: Arc) -> Self { - self.job_capacity = slots.available_permits(); - self.job_slots = slots; - self - } - - /// Bounds simultaneous full restore, resume, bundling and remote compaction. - /// - /// Defaults share two slots process-wide. Admission precedes body downloads - /// and stays with non-cancellable jobs. This bounds cohorts, not process RSS; - /// size per-database `Limits` and these slots to the service memory budget. - #[cfg(feature = "replica")] - #[must_use] - pub fn with_recovery_slots(mut self, slots: Arc) -> Self { - self.recovery_capacity = slots.available_permits(); - self.recovery_slots = slots; - self - } - - /// Shares memory admission for capture, recovery and compaction jobs. - /// - /// One permit represents the embedding service's fixed per-job dirty-memory - /// reservation. The permit follows dispatched work after caller cancellation. - #[cfg(feature = "replica")] - #[must_use] - pub fn with_dirty_slots(mut self, slots: Arc) -> Self { - self.dirty_capacity = slots.available_permits(); - self.dirty_slots = slots; - self - } - - /// Shares temporary local-disk admission in one-MiB permit units. - /// - /// Configure an unused semaphore before cloning the host. Requests larger - /// than its initial capacity fail instead of waiting forever. - #[cfg(feature = "replica")] - #[must_use] - pub fn with_scratch_slots(mut self, slots: Arc) -> Self { - self.scratch_capacity = u32::try_from(slots.available_permits()).unwrap_or(u32::MAX); - self.scratch_slots = slots; - self - } - - /// Rechecks actual host capacity whenever a full scratch job is admitted. - #[cfg(feature = "replica")] - #[must_use] - pub fn with_scratch_monitor(mut self, monitor: Arc) -> Self { - self.scratch_monitor = monitor; - self - } - - #[cfg(feature = "replica")] - fn reserve_resource( - &self, - kind: HostResourceKind, - units: u32, - ) -> crate::Result>> { - self.resource_admission - .as_ref() - .map(|admission| admission.reserve(kind, units).map(Arc::from)) - .transpose() - } - - #[cfg(feature = "replica")] - pub(crate) async fn for_dirty(&self) -> crate::Result { - let mut host = self.clone(); - if host.dirty.is_none() { - let permit = self - .dirty_slots - .clone() - .acquire_owned() - .await - .map_err(|e| crate::CrabError::Other(Box::new(e)))?; - host.dirty_resource = self.reserve_resource(HostResourceKind::Dirty, 1)?; - host.dirty = Some(Arc::new(permit)); - } - Ok(host) - } - - #[cfg(feature = "replica")] - pub(crate) async fn for_recovery(&self) -> crate::Result { - let mut host = self.for_dirty().await?; - if host.recovery.is_none() { - let permit = self - .recovery_slots - .clone() - .acquire_owned() - .await - .map_err(|e| crate::CrabError::Other(Box::new(e)))?; - host.recovery_resource = self.reserve_resource(HostResourceKind::Recovery, 1)?; - host.recovery = Some(Arc::new(permit)); - } - Ok(host) - } - - #[cfg(feature = "replica")] - pub(crate) async fn for_scratch(&self, bytes: u64) -> crate::Result { - const MIB: u64 = 1 << 20; - let units = bytes - .checked_add(MIB - 1) - .ok_or(crate::CrabError::Limit(crate::LimitKind::ScratchDiskBytes))? - / MIB; - let units = u32::try_from(units) - .map_err(|_| crate::CrabError::Limit(crate::LimitKind::ScratchDiskBytes))?; - if units == 0 || units > self.scratch_capacity { - return Err(crate::CrabError::Limit(crate::LimitKind::ScratchDiskBytes)); - } - let mut host = self.clone(); - if let Some(permit) = &host.scratch { - if permit.num_permits() < units as usize { - return Err(crate::CrabError::Limit(crate::LimitKind::ScratchDiskBytes)); - } - return Ok(host); - } - let permit = self - .scratch_slots - .clone() - .acquire_many_owned(units) - .await - .map_err(|e| crate::CrabError::Other(Box::new(e)))?; - let capacity = self.scratch_capacity as usize; - let reserved_units = capacity.saturating_sub(self.scratch_slots.available_permits()); - let reserved_bytes = u64::try_from(reserved_units) - .ok() - .and_then(|units| units.checked_mul(MIB)) - .ok_or(crate::CrabError::Limit(crate::LimitKind::ScratchDiskBytes))?; - self.scratch_monitor - .ensure_available(reserved_bytes) - .map_err(crate::CrabError::Io)?; - host.scratch_resource = self.reserve_resource(HostResourceKind::Scratch, units)?; - host.scratch = Some(Arc::new(permit)); - Ok(host) - } - - #[cfg(feature = "replica")] - pub(crate) fn without_recovery(mut self) -> Self { - self.recovery = None; - self.recovery_resource = None; - self - } - - #[cfg(feature = "replica")] - pub(crate) fn without_dirty(mut self) -> Self { - self.dirty = None; - self.dirty_resource = None; - self - } - - #[cfg(feature = "replica")] - pub(crate) fn without_scratch(mut self) -> Self { - self.scratch = None; - self.scratch_resource = None; - self - } - - #[cfg(feature = "replica")] - pub(crate) async fn io_permit(&self) -> crate::Result { - let permit = self - .io_slots - .clone() - .acquire_owned() - .await - .map_err(|e| crate::CrabError::Other(Box::new(e)))?; - let resource = self.reserve_resource(HostResourceKind::Io, 1)?; - Ok(HostIoPermit { - _semaphore: permit, - _resource: resource, - }) - } - - #[cfg(feature = "replica")] - pub(crate) async fn directory_cache_get( - &self, - key: String, - max_bytes: u64, - ) -> crate::Result>> { - let Some(cache) = &self.directory_cache else { - return Ok(None); - }; - // Cache access is optional. A busy fill must not queue a verified read - // behind local syncs; origin remains the authenticated source on a miss. - let permit = match self.job_slots.clone().try_acquire_owned() { - Ok(permit) => permit, - Err(tokio::sync::TryAcquireError::NoPermits) => return Ok(None), - Err(error) => return Err(crate::CrabError::Other(Box::new(error))), - }; - let cache = Arc::clone(cache); - self.run_admitted(permit, move || cache.get(&key, max_bytes)) - .await? - .map_err(crate::CrabError::Io) - } - - #[cfg(feature = "replica")] - pub(crate) fn directory_cache_put( - &self, - key: String, - bytes: Vec, - max_bytes: u64, - ) -> crate::Result<()> { - let Some(cache) = &self.directory_cache else { - return Ok(()); - }; - if bytes.is_empty() || bytes.len() as u64 > max_bytes { - return Ok(()); - } - // A cache fill is optional. Waiting here after releasing origin - // admission would retain one node buffer per waiter without a bound. - let permit = match self.job_slots.clone().try_acquire_owned() { - Ok(permit) => permit, - Err(tokio::sync::TryAcquireError::NoPermits) => return Ok(()), - Err(error) => return Err(crate::CrabError::Other(Box::new(error))), - }; - let Some(fill) = self.cache_fills.claim(cache.key_path(&key)) else { - return Ok(()); - }; - let resource = self.reserve_resource(HostResourceKind::BlockingJob, 1)?; - let cache = Arc::clone(cache); - // Verified buffers are bounded by admitted jobs and the node-size cap. - // Derived fills own no capture/recovery cohort; only their job and disk - // accounting survive the read, including executor rejection or panic. - self.executor.dispatch(Box::new(move || { - let _ = std::panic::catch_unwind(std::panic::AssertUnwindSafe(move || { - cache.put(&key, &bytes, max_bytes) - })); - drop(resource); - drop(permit); - // Shutdown observes completion only after the job's buffers and - // admission are released, so returning cannot hide running fills. - drop(fill); - }))?; - Ok(()) - } - - #[cfg(feature = "replica")] - pub(crate) async fn directory_cache_invalidate(&self, key: String) -> crate::Result<()> { - let Some(cache) = &self.directory_cache else { - return Ok(()); - }; - let cache = Arc::clone(cache); - self.run(move || cache.invalidate(&key)) - .await? - .map_err(crate::CrabError::Io) - } - - #[cfg(feature = "replica")] - pub(crate) async fn run( - &self, - operation: impl FnOnce() -> T + Send + 'static, - ) -> crate::Result { - let permit = self - .job_slots - .clone() - .acquire_owned() - .await - .map_err(|e| crate::CrabError::Other(Box::new(e)))?; - self.run_admitted(permit, operation).await - } - - #[cfg(feature = "replica")] - async fn run_admitted( - &self, - permit: tokio::sync::OwnedSemaphorePermit, - operation: impl FnOnce() -> T + Send + 'static, - ) -> crate::Result { - let resource = self.reserve_resource(HostResourceKind::BlockingJob, 1)?; - let (send, receive) = tokio::sync::oneshot::channel(); - let recovery = self.recovery.clone(); - let dirty = self.dirty.clone(); - let scratch = self.scratch.clone(); - self.executor.dispatch(Box::new(move || { - // Dispatched work can outlive its future. Keep admission with the - // job, not the waiter, so cancellation cannot oversubscribe the pool. - let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(operation)) - .map_err(|_| crate::CrabError::InvalidState("host job panicked")); - // Result delivery is the operation's completion boundary. Release - // admission first so returned long-lived handles cannot appear to - // retain capacity while this closure is still being torn down. - drop(recovery); - drop(dirty); - drop(scratch); - drop(resource); - drop(permit); - let _ = send.send(result); - }))?; - receive - .await - .map_err(|error| crate::CrabError::Other(Box::new(error)))? - } -} - -impl Default for Host { - fn default() -> Self { - #[cfg(feature = "replica")] - static IO: std::sync::OnceLock> = std::sync::OnceLock::new(); - #[cfg(feature = "replica")] - static JOBS: std::sync::OnceLock> = std::sync::OnceLock::new(); - #[cfg(feature = "replica")] - static RECOVERY: std::sync::OnceLock> = - std::sync::OnceLock::new(); - #[cfg(feature = "replica")] - static DIRTY: std::sync::OnceLock> = std::sync::OnceLock::new(); - #[cfg(feature = "replica")] - static SCRATCH: std::sync::OnceLock> = - std::sync::OnceLock::new(); - static LOCAL_DISK: std::sync::OnceLock = std::sync::OnceLock::new(); - Self { - filesystem: Arc::new(DirectFileSystem), - clock: Arc::new(SystemClock), - sqlite_vfs: None, - local_disk: LOCAL_DISK - .get_or_init(|| DiskBudget::new(64 * 1024 * 1024 * 1024)) - .clone(), - #[cfg(feature = "replica")] - executor: Arc::new(TokioExecutor), - #[cfg(feature = "replica")] - paged_driver: crate::paged_io::default_slot(), - #[cfg(feature = "replica")] - io_slots: IO - .get_or_init(|| Arc::new(tokio::sync::Semaphore::new(32))) - .clone(), - #[cfg(feature = "replica")] - io_capacity: 32, - #[cfg(feature = "replica")] - job_slots: JOBS - .get_or_init(|| { - Arc::new(tokio::sync::Semaphore::new( - std::thread::available_parallelism().map_or(1, |n| n.get().min(16)), - )) - }) - .clone(), - #[cfg(feature = "replica")] - job_capacity: std::thread::available_parallelism().map_or(1, |n| n.get().min(16)), - #[cfg(feature = "replica")] - recovery_slots: RECOVERY - .get_or_init(|| Arc::new(tokio::sync::Semaphore::new(2))) - .clone(), - #[cfg(feature = "replica")] - recovery_capacity: 2, - #[cfg(feature = "replica")] - dirty_slots: DIRTY - .get_or_init(|| { - Arc::new(tokio::sync::Semaphore::new( - std::thread::available_parallelism().map_or(1, |n| n.get().min(16)), - )) - }) - .clone(), - #[cfg(feature = "replica")] - dirty_capacity: std::thread::available_parallelism().map_or(1, |n| n.get().min(16)), - #[cfg(feature = "replica")] - scratch_slots: SCRATCH - .get_or_init(|| Arc::new(tokio::sync::Semaphore::new(64 * 1024))) - .clone(), - #[cfg(feature = "replica")] - scratch_capacity: 64 * 1024, - #[cfg(feature = "replica")] - scratch_monitor: Arc::new(UnlimitedScratch), - #[cfg(feature = "replica")] - directory_cache: None, - #[cfg(feature = "replica")] - cache_fills: Arc::new(CacheFills::default()), - #[cfg(feature = "replica")] - resource_admission: None, - #[cfg(feature = "replica")] - telemetry: None, - #[cfg(feature = "replica")] - recovery: None, - #[cfg(feature = "replica")] - dirty: None, - #[cfg(feature = "replica")] - scratch: None, - #[cfg(feature = "replica")] - recovery_resource: None, - #[cfg(feature = "replica")] - dirty_resource: None, - #[cfg(feature = "replica")] - scratch_resource: None, - } - } -} - -/// Standard local filesystem with exclusive private artifact creation. -#[derive(Default)] -pub struct DirectFileSystem; - -impl FileIo for std::fs::File { - fn write_all(&mut self, bytes: &[u8]) -> io::Result<()> { - io::Write::write_all(self, bytes) - } - fn write_all_at(&mut self, offset: u64, bytes: &[u8]) -> io::Result<()> { - io::Seek::seek(self, io::SeekFrom::Start(offset))?; - io::Write::write_all(self, bytes) - } - fn read_exact_at(&mut self, offset: u64, len: usize) -> io::Result> { - io::Seek::seek(self, io::SeekFrom::Start(offset))?; - let mut bytes = vec![0; len]; - io::Read::read_exact(self, &mut bytes)?; - Ok(bytes) - } - fn sync_all(&mut self) -> io::Result<()> { - std::fs::File::sync_all(self) - } - fn file_len(&self) -> io::Result { - Ok(self.metadata()?.len()) - } - fn set_len(&mut self, len: u64) -> io::Result<()> { - std::fs::File::set_len(self, len) - } -} - -impl FileSystem for DirectFileSystem { - fn open(&self, path: &Path) -> io::Result> { - Ok(Box::new(std::fs::File::open(path)?)) - } - fn open_rw(&self, path: &Path) -> io::Result> { - Ok(Box::new( - std::fs::OpenOptions::new() - .read(true) - .write(true) - .open(path)?, - )) - } - fn create(&self, path: &Path) -> io::Result> { - let mut options = std::fs::OpenOptions::new(); - options.write(true).create_new(true); - #[cfg(unix)] - { - use std::os::unix::fs::OpenOptionsExt; - options.mode(0o600); - } - Ok(Box::new(options.open(path)?)) - } - fn file_len(&self, path: &Path) -> io::Result { - Ok(std::fs::metadata(path)?.len()) - } - fn create_dir_all(&self, path: &Path) -> io::Result<()> { - std::fs::create_dir_all(path) - } - fn rename(&self, from: &Path, to: &Path) -> io::Result<()> { - std::fs::rename(from, to)?; - crate::host::sync_parent(to) - } - fn rename_uncommitted(&self, from: &Path, to: &Path) -> io::Result<()> { - std::fs::rename(from, to) - } - fn remove_file(&self, path: &Path) -> io::Result<()> { - std::fs::remove_file(path) - } - fn canonicalize(&self, path: &Path) -> io::Result { - path.canonicalize() - } - fn exists(&self, path: &Path) -> io::Result { - match std::fs::symlink_metadata(path) { - Ok(_) => Ok(true), - Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(false), - Err(error) => Err(error), - } - } - fn create_dir(&self, path: &Path) -> io::Result<()> { - std::fs::create_dir(path) - } - fn sync_files(&self, paths: &[PathBuf]) -> io::Result<()> { - let workers = std::thread::available_parallelism() - .map_or(1, std::num::NonZeroUsize::get) - .min(paths.len()); - if workers <= 1 { - for path in paths { - self.open_rw(path)?.sync_all()?; - } - return Ok(()); - } - let next = AtomicUsize::new(0); - std::thread::scope(|scope| { - let handles = (0..workers) - .map(|_| { - let next = &next; - scope.spawn(move || { - loop { - let index = next.fetch_add(1, Ordering::Relaxed); - let Some(path) = paths.get(index) else { - break; - }; - std::fs::OpenOptions::new() - .read(true) - .write(true) - .open(path)? - .sync_all()?; - } - Ok::<(), io::Error>(()) - }) - }) - .collect::>(); - let mut first_error = None; - for handle in handles { - let result = handle - .join() - .map_err(|_| io::Error::other("file sync worker panicked")) - .and_then(|result| result); - if let Err(error) = result - && first_error.is_none() - { - first_error = Some(error); - } - } - if let Some(error) = first_error { - return Err(error); - } - Ok(()) - }) - } - fn sync_parent(&self, path: &Path) -> io::Result<()> { - crate::host::sync_parent(path) - } - fn persist_new(&self, path: &Path, bytes: &[u8]) -> io::Result<()> { - let parent = path - .parent() - .filter(|p| !p.as_os_str().is_empty()) - .unwrap_or(Path::new(".")); - let mut file = tempfile::NamedTempFile::new_in(parent)?; - io::Write::write_all(&mut file, bytes)?; - file.as_file().sync_all()?; - file.persist_noclobber(path).map_err(|error| error.error)?; - self.sync_parent(path) - } - fn persist_file_new(&self, source: &Path, destination: &Path) -> io::Result<()> { - if source.parent() != destination.parent() { - return Err(io::Error::new( - io::ErrorKind::InvalidInput, - "scratch and destination must share a directory", - )); - } - std::fs::hard_link(source, destination)?; - // Both names share one directory, so one barrier after link + unlink - // makes the no-clobber install and scratch cleanup durable together. - // A barrier error is ambiguous because the destination may now exist. - std::fs::remove_file(source)?; - self.sync_parent(destination) - } - - fn cleanup_private_temporaries(&self, root: &Path) -> io::Result<()> { - let entries = match std::fs::read_dir(root) { - Ok(entries) => entries, - Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(()), - Err(error) => return Err(error), - }; - for entry in entries { - let entry = entry?; - let file_type = entry.file_type()?; - let name = entry.file_name(); - let name = name.to_string_lossy(); - if file_type.is_file() && (name.starts_with(".tmp-") || name.starts_with(".index-tmp-")) - { - match std::fs::remove_file(entry.path()) { - Ok(()) => {} - Err(error) if error.kind() == io::ErrorKind::NotFound => {} - Err(error) => return Err(error), - } - } - } - Ok(()) - } -} - -/// Operating-system wall clock; future mtimes have age zero. -#[derive(Default)] -pub struct SystemClock; - -impl Clock for SystemClock { - fn unix_millis(&self) -> i64 { - SystemTime::now() - .duration_since(UNIX_EPOCH) - .map(|d| i64::try_from(d.as_millis()).unwrap_or(i64::MAX)) - .unwrap_or(0) - } - fn file_age(&self, path: &Path) -> io::Result { - Ok(SystemTime::now() - .duration_since(std::fs::metadata(path)?.modified()?) - .unwrap_or_default()) - } -} diff --git a/crates/crab-ltx/src/environment/host/budget.rs b/crates/crab-ltx/src/environment/host/budget.rs deleted file mode 100644 index acd1c21ef..000000000 --- a/crates/crab-ltx/src/environment/host/budget.rs +++ /dev/null @@ -1,272 +0,0 @@ -//! Byte-precise local disk admission for active database work. -//! -//! A budget owns the capacity an embedding runtime granted it, charges the -//! exact bytes each reservation holds, and reconciles every change with the -//! registered admissions so a node ledger never drifts from the files. - -use super::*; - -type DiskAdmissions = Vec>; - -/// Shared byte-precise admission for local files owned by active database work. -#[derive(Clone)] -pub struct DiskBudget { - inner: Arc, -} - -pub(crate) struct DiskBudgetInner { - capacity: u64, - used: AtomicU64, - has_admissions: AtomicBool, - admissions: Mutex, -} - -impl fmt::Debug for DiskBudget { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter - .debug_struct("DiskBudget") - .field("capacity", &self.capacity()) - .field("used", &self.used()) - .finish_non_exhaustive() - } -} - -impl DiskBudget { - /// Creates a budget. A zero capacity rejects every non-empty reservation. - #[must_use] - pub fn new(capacity: u64) -> Self { - Self { - inner: Arc::new(DiskBudgetInner { - capacity, - used: AtomicU64::new(0), - has_admissions: AtomicBool::new(false), - admissions: Mutex::new(Vec::new()), - }), - } - } - - /// Installs one embedding ledger and immediately reconciles existing bytes. - /// - /// All clones of this budget observe the same hook. Multiple live runtimes - /// may observe the same process-wide budget; dead hooks are removed before - /// the new hook is registered. - pub fn install_admission(&self, admission: Arc) -> crate::Result<()> { - let mut current = self - .inner - .admissions - .lock() - .map_err(|_| crate::CrabError::InvalidState("disk admission lock poisoned"))?; - current.retain(|admission| admission.is_live()); - let had_admissions = !current.is_empty(); - self.inner.has_admissions.store(true, Ordering::Release); - current.push(admission); - let result = current.last().map_or_else( - || { - Err(crate::CrabError::InvalidState( - "disk admission was not installed", - )) - }, - |admission| admission.reconcile(self.used()), - ); - if let Err(error) = result { - current.pop(); - self.inner - .has_admissions - .store(had_admissions, Ordering::Release); - return Err(error); - } - Ok(()) - } - - /// Reserves bytes without waiting or overcommitting the configured capacity. - pub fn try_reserve(&self, bytes: u64) -> crate::Result { - self.add(bytes)?; - if let Err(error) = self.reconcile_admissions(self.used()) { - let _ = self.remove(bytes); - let _ = self.reconcile_admissions(self.used()); - return Err(error); - } - Ok(DiskReservation { - budget: self.clone(), - bytes: Mutex::new(bytes), - }) - } - - /// Returns the bytes this budget may reserve in total. - #[must_use] - pub fn capacity(&self) -> u64 { - self.inner.capacity - } - - /// Returns the bytes held by live reservations. - #[must_use] - pub fn used(&self) -> u64 { - self.inner.used.load(Ordering::Acquire) - } - - /// Returns the bytes no reservation holds. - #[must_use] - pub fn available(&self) -> u64 { - self.capacity().saturating_sub(self.used()) - } - - fn add(&self, bytes: u64) -> crate::Result<()> { - self.inner - .used - .fetch_update(Ordering::AcqRel, Ordering::Acquire, |used| { - used.checked_add(bytes) - .filter(|next| *next <= self.inner.capacity) - }) - .map(|_| ()) - .map_err(|_| crate::CrabError::Limit(crate::LimitKind::LocalDiskBytes)) - } - - fn reconcile_admissions(&self, bytes: u64) -> crate::Result<()> { - let Some(mut admissions) = self.live_admissions()? else { - return Ok(()); - }; - Self::reconcile_admissions_locked(&mut admissions, bytes) - } - - fn live_admissions(&self) -> crate::Result>> { - if !self.inner.has_admissions.load(Ordering::Acquire) { - return Ok(None); - } - let mut admissions = self - .inner - .admissions - .lock() - .map_err(|_| crate::CrabError::InvalidState("disk admission lock poisoned"))?; - admissions.retain(|admission| admission.is_live()); - if admissions.is_empty() { - self.inner.has_admissions.store(false, Ordering::Release); - return Ok(None); - } - Ok(Some(admissions)) - } - - fn reconcile_admissions_locked( - admissions: &mut DiskAdmissions, - bytes: u64, - ) -> crate::Result<()> { - let mut index = 0; - while index < admissions.len() { - match admissions[index].reconcile(bytes) { - Ok(()) => index += 1, - // The owner can drop between the liveness prune and this call. - // Its reservations died with it, and a closed hook must never - // fail a reservation that the live runtimes still own. - Err(_) if !admissions[index].is_live() => { - admissions.remove(index); - } - Err(error) => return Err(error), - } - } - Ok(()) - } - - fn remove(&self, bytes: u64) -> crate::Result<()> { - self.inner - .used - .fetch_update(Ordering::AcqRel, Ordering::Acquire, |used| { - used.checked_sub(bytes) - }) - .map(|_| ()) - .map_err(|_| crate::CrabError::InvalidState("local disk reservation underflow")) - } -} - -/// Owned local-disk admission released when its owner drops it. -pub struct DiskReservation { - budget: DiskBudget, - bytes: Mutex, -} - -impl fmt::Debug for DiskReservation { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter - .debug_struct("DiskReservation") - .field("bytes", &self.bytes()) - .finish_non_exhaustive() - } -} - -impl DiskReservation { - /// Adds bytes to this reservation without exceeding the shared budget. - pub fn try_grow(&self, bytes: u64) -> crate::Result<()> { - let mut held = match self.bytes.lock() { - Ok(held) => held, - Err(poisoned) => poisoned.into_inner(), - }; - let next = held - .checked_add(bytes) - .ok_or(crate::CrabError::Limit(crate::LimitKind::LocalDiskBytes))?; - self.budget.add(bytes)?; - if let Err(error) = self.budget.reconcile_admissions(self.budget.used()) { - let _ = self.budget.remove(bytes); - let _ = self.budget.reconcile_admissions(self.budget.used()); - return Err(error); - } - *held = next; - Ok(()) - } - - /// Changes the exact held byte count, releasing capacity when it shrinks. - pub fn resize(&self, bytes: u64) -> crate::Result<()> { - let mut held = match self.bytes.lock() { - Ok(held) => held, - Err(poisoned) => poisoned.into_inner(), - }; - let current = *held; - if bytes > current { - let added = bytes - current; - self.budget.add(added)?; - if let Err(error) = self.budget.reconcile_admissions(self.budget.used()) { - let _ = self.budget.remove(added); - let _ = self.budget.reconcile_admissions(self.budget.used()); - return Err(error); - } - *held = bytes; - return Ok(()); - } - let released = current - bytes; - *held = bytes; - if let Err(error) = self.budget.remove(released) { - *held = current; - return Err(error); - } - if let Err(error) = self.budget.reconcile_admissions(self.budget.used()) { - self.budget.add(released)?; - *held = current; - let _ = self.budget.reconcile_admissions(self.budget.used()); - return Err(error); - } - Ok(()) - } - - fn release(&self) { - let mut held = match self.bytes.lock() { - Ok(held) => held, - Err(poisoned) => poisoned.into_inner(), - }; - let released = *held; - *held = 0; - let _ = self.budget.remove(released); - let _ = self.budget.reconcile_admissions(self.budget.used()); - } - - /// Returns the bytes this reservation still holds. - #[must_use] - pub fn bytes(&self) -> u64 { - match self.bytes.lock() { - Ok(held) => *held, - Err(poisoned) => *poisoned.into_inner(), - } - } -} - -impl Drop for DiskReservation { - fn drop(&mut self) { - self.release(); - } -} diff --git a/crates/crab-ltx/src/environment/resources.rs b/crates/crab-ltx/src/environment/resources.rs deleted file mode 100644 index 407b34b7b..000000000 --- a/crates/crab-ltx/src/environment/resources.rs +++ /dev/null @@ -1,33 +0,0 @@ -//! Object-store-only host resource admission contracts. - -/// Resource class charged by an embedding runtime for replica-host work. -#[cfg(feature = "replica")] -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum HostResourceKind { - /// One bounded object-store or immutable-file I/O operation. - Io, - /// One blocking host job dispatched to the replica executor. - BlockingJob, - /// One full recovery or restore cohort. - Recovery, - /// One capture/compaction dirty-memory cohort. - Dirty, - /// One MiB of temporary scratch admission. - Scratch, -} - -/// Admission hook used by an embedding runtime to charge host work to its -/// node-wide resource ledger. The returned permit owns the charge until drop. -#[cfg(feature = "replica")] -pub trait HostResourceAdmission: Send + Sync { - /// Reserves `units` of one host resource without waiting. - fn reserve( - &self, - kind: HostResourceKind, - units: u32, - ) -> crate::Result>; -} - -/// Opaque lifetime token returned by [`HostResourceAdmission::reserve`]. -#[cfg(feature = "replica")] -pub trait HostResourcePermit: Send + Sync {} diff --git a/crates/crab-ltx/src/environment/telemetry.rs b/crates/crab-ltx/src/environment/telemetry.rs deleted file mode 100644 index 2c0bbb5dd..000000000 --- a/crates/crab-ltx/src/environment/telemetry.rs +++ /dev/null @@ -1,109 +0,0 @@ -//! Object-store-only telemetry and scratch-monitor contracts. - -use std::{io, time::Duration}; - -/// Finite replica phases exposed to an embedding runtime. -#[cfg(feature = "replica")] -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum LtxPhase { - /// Local WAL capture. - Capture, - /// Managed-database preparation before capture. - Preparation, - /// Managed control-table schema validation. - SchemaCheck, - /// Confirming the WAL holds a readable frame. - WalExistence, - /// Resolving the checksum-linked WAL position. - PositionResolution, - /// Reading and parsing the WAL image. - WalRead, - /// Collecting the committed page map. - PageCollection, - /// Validating WAL or produced LTX data. - Verification, - /// Encoding LTX bytes and page records. - Encode, - /// Writing LTX and index bytes locally. - LocalWrite, - /// Syncing completed LTX contents. - Fsync, - /// Syncing the published name's parent directory. - ParentSync, - /// Checkpoint maintenance for the capture. - Checkpoint, - /// Preparing one immutable successor root from captured cuts. - RootPreparation, - /// Opening an exact root. - RootOpen, - /// Reading root directory pages. - Directory, - /// Fetching segment frames from the provider. - FrameFetch, - /// Writing restored pages locally. - RestoreWrite, - /// Compacting segments. - Compaction, -} - -/// Finite origin-read classes exposed to an embedding runtime. -#[cfg(feature = "replica")] -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum LtxReadOrigin { - /// A read that had to reach the provider. - Cold, - /// A read served by the sparse local activation. - Sparse, - /// A read that triggered hydration. - Hydrating, - /// A read served by a fully resident local database. - Resident, -} - -/// Finite provider-attempt outcomes exposed to an embedding runtime. -#[cfg(feature = "replica")] -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum LtxRequestOutcome { - /// The provider attempt returned bytes. - Succeeded, - /// The provider attempt failed. - Failed, -} - -/// Non-blocking, bounded-cardinality observations emitted by replica work. -#[cfg(feature = "replica")] -pub trait LtxTelemetry: Send + Sync { - /// Records one completed phase and whether it succeeded. - fn phase(&self, _phase: LtxPhase, _elapsed: Duration, _succeeded: bool) {} - - /// Records one logical read issued by the runtime or a paged database. - fn logical_read(&self, _origin: LtxReadOrigin) {} - - /// Records one provider attempt and bytes returned before its outcome. - fn origin_request(&self, _origin: LtxReadOrigin, _outcome: LtxRequestOutcome, _bytes: u64) {} - - /// Records one complete capture attempt, including failed attempts. - fn capture(&self, _timing: &crate::CaptureTiming, _succeeded: bool) {} -} - -/// Rechecks host disk pressure after full-job scratch admission. -/// -/// `reserved_bytes` is the process-wide scratch reservation, including the -/// current job. An embedding service can combine it with other local-disk -/// reservations and an operator reserve before allowing remote downloads. -#[cfg(feature = "replica")] -pub trait ScratchMonitor: Send + Sync { - /// Rechecks host disk pressure, failing when the reservation cannot be - /// admitted alongside the process-wide scratch a job needs. - fn ensure_available(&self, reserved_bytes: u64) -> io::Result<()>; -} - -#[cfg(feature = "replica")] -pub(crate) struct UnlimitedScratch; - -#[cfg(feature = "replica")] -impl ScratchMonitor for UnlimitedScratch { - fn ensure_available(&self, _: u64) -> io::Result<()> { - Ok(()) - } -} diff --git a/crates/crab-ltx/src/environment/tests.rs b/crates/crab-ltx/src/environment/tests.rs deleted file mode 100644 index 0a695b997..000000000 --- a/crates/crab-ltx/src/environment/tests.rs +++ /dev/null @@ -1,542 +0,0 @@ -#[cfg(feature = "replica")] -use std::sync::atomic::AtomicU64; - -use std::{ - io, - path::{Path, PathBuf}, - sync::{ - Arc, - atomic::{AtomicBool, AtomicUsize, Ordering}, - }, - time::Duration, -}; - -use super::*; -#[cfg(feature = "replica")] -use crate::environment::directory_cache::DirectoryCache; -#[cfg(feature = "replica")] -use crate::environment::executor::TokioExecutor; - -#[cfg(feature = "replica")] -#[tokio::test] -async fn directory_cache_fill_does_not_queue_bytes_behind_busy_jobs() { - let directory = tempfile::TempDir::new().unwrap(); - let jobs = Arc::new(tokio::sync::Semaphore::new(1)); - let host = Host::default() - .with_job_slots(jobs.clone()) - .with_directory_cache(directory.path().join("cache")) - .await - .unwrap(); - let occupied = jobs.acquire().await.unwrap(); - host.directory_cache_put("node".into(), b"verified".to_vec(), 64) - .unwrap(); - assert_eq!(host.directory_cache_stats().unwrap().entries(), 0); - drop(occupied); - host.directory_cache_put("node".into(), b"verified".to_vec(), 64) - .unwrap(); - host.drain_cache_fills().await; - assert_eq!(host.directory_cache_stats().unwrap().entries(), 1); -} - -#[cfg(feature = "replica")] -#[tokio::test] -async fn cache_fills_deduplicate_bound_dispatch_and_release_dropped_jobs() { - #[derive(Default)] - struct HeldExecutor(std::sync::Mutex>>); - impl Executor for HeldExecutor { - fn dispatch(&self, job: Box) -> io::Result<()> { - self.0.lock().unwrap().push(job); - Ok(()) - } - fn start_worker(&self, job: Box) -> io::Result> { - TokioExecutor.start_worker(job) - } - } - let directory = tempfile::TempDir::new().unwrap(); - let jobs = Arc::new(tokio::sync::Semaphore::new(2)); - let filesystem = Arc::new(FaultFs::default()); - let executor = Arc::new(HeldExecutor::default()); - let host = Host::default() - .with_filesystem(filesystem.clone()) - .with_job_slots(jobs.clone()) - .with_directory_cache(directory.path().join("cache")) - .await - .unwrap() - .with_executor(executor.clone()); - for _ in 0..100 { - host.directory_cache_put("same".into(), b"verified".to_vec(), 64) - .unwrap(); - } - assert_eq!(executor.0.lock().unwrap().len(), 1); - host.directory_cache_put("second".into(), b"verified".to_vec(), 64) - .unwrap(); - host.directory_cache_put("overflow".into(), b"verified".to_vec(), 64) - .unwrap(); - assert_eq!(executor.0.lock().unwrap().len(), 2); - assert_eq!(jobs.available_permits(), 0); - assert!( - host.directory_cache_get("same".into(), 64) - .await - .unwrap() - .is_none() - ); - executor.0.lock().unwrap().clear(); - tokio::time::timeout(Duration::from_secs(1), host.drain_cache_fills()) - .await - .unwrap(); - assert_eq!(jobs.available_permits(), 2); - assert_eq!(host.directory_cache_stats().unwrap().entries(), 0); - - for (reject, panic) in [(true, false), (false, true), (false, false)] { - filesystem.reject_create.store(reject, Ordering::SeqCst); - filesystem.panic_create.store(panic, Ordering::SeqCst); - host.directory_cache_put("same".into(), b"verified".to_vec(), 64) - .unwrap(); - let job = executor.0.lock().unwrap().pop().unwrap(); - job(); - tokio::time::timeout(Duration::from_secs(1), host.drain_cache_fills()) - .await - .unwrap(); - assert_eq!(jobs.available_permits(), 2); - } - assert_eq!(host.directory_cache_stats().unwrap().entries(), 1); -} - -#[test] -fn disk_budget_reservations_resize_and_release_exact_bytes() { - let budget = DiskBudget::new(10); - let first = budget.try_reserve(4).unwrap(); - let second = budget.try_reserve(6).unwrap(); - assert_eq!(budget.available(), 0); - assert!(matches!( - budget.try_reserve(1), - Err(crate::CrabError::Limit(crate::LimitKind::LocalDiskBytes)) - )); - - first.resize(2).unwrap(); - assert_eq!(budget.available(), 2); - drop(second); - assert_eq!(budget.available(), 8); - drop(first); - assert_eq!(budget.available(), 10); -} - -#[cfg(feature = "replica")] -#[test] -fn directory_cache_survives_restart_and_evicts_by_bytes() { - let directory = tempfile::TempDir::new().unwrap(); - let root = directory.path().join("directory-cache"); - let filesystem: Arc = Arc::new(DirectFileSystem); - let cache = DirectoryCache::new(Arc::clone(&filesystem), root.clone(), 5); - cache.put("first", b"1234", 64).unwrap(); - assert_eq!(cache.budget.used(), 4); - drop(cache); - - let cache = DirectoryCache::new(Arc::clone(&filesystem), root, 5); - assert_eq!(cache.get("first", 64).unwrap(), Some(b"1234".to_vec())); - assert_eq!(cache.stats().entries(), 1); - assert_eq!(cache.stats().bytes(), 4); - cache.put("second", b"abcde", 64).unwrap(); - assert!(cache.get("first", 64).unwrap().is_none()); - assert_eq!(cache.get("second", 64).unwrap(), Some(b"abcde".to_vec())); - assert_eq!(cache.stats().entries(), 1); - assert_eq!(cache.budget.used(), 5); -} - -#[cfg(feature = "replica")] -#[test] -fn directory_cache_reaccounts_a_file_without_membership() { - let directory = tempfile::TempDir::new().unwrap(); - let cache = DirectoryCache::new(Arc::new(DirectFileSystem), directory.path().to_owned(), 64); - // A fill can install its bytes before the membership index is persisted. - // Reusing that file after restart must still acquire its disk reservation. - std::fs::write(cache.key_path("orphan"), b"verified").unwrap(); - cache.put("orphan", b"verified", 64).unwrap(); - assert_eq!(cache.budget.used(), 8); - assert_eq!(cache.get("orphan", 64).unwrap(), Some(b"verified".to_vec())); -} - -#[cfg(feature = "replica")] -#[test] -fn directory_cache_reads_do_not_rewrite_unchanged_membership() { - let directory = tempfile::TempDir::new().unwrap(); - let filesystem = Arc::new(FaultFs::default()); - let cache = DirectoryCache::new(filesystem.clone(), directory.path().to_owned(), 64); - cache.put("present", b"verified", 64).unwrap(); - filesystem.creates.store(0, Ordering::SeqCst); - - for (key, expected) in [("present", Some(b"verified".to_vec())), ("absent", None)] { - assert_eq!(cache.get(key, 64).unwrap(), expected); - assert_eq!(filesystem.creates.load(Ordering::SeqCst), 0, "{key}"); - } -} - -#[cfg(feature = "replica")] -#[test] -fn directory_cache_discards_truncated_and_symlink_entries() { - let directory = tempfile::TempDir::new().unwrap(); - let root = directory.path().join("directory-cache"); - let filesystem: Arc = Arc::new(DirectFileSystem); - let cache = DirectoryCache::new(Arc::clone(&filesystem), root.clone(), 64); - cache.put("entry", b"verified", 64).unwrap(); - let path = cache.key_path("entry"); - std::fs::write(&path, b"short").unwrap(); - assert!(cache.get("entry", 64).unwrap().is_none()); - - let target = directory.path().join("outside"); - std::fs::write(&target, b"outside").unwrap(); - let symlink = cache.key_path("symlink"); - #[cfg(unix)] - std::os::unix::fs::symlink(&target, &symlink).unwrap(); - #[cfg(unix)] - assert!(cache.get("symlink", 64).unwrap().is_none()); -} - -#[cfg(feature = "replica")] -#[test] -fn directory_cache_cleans_abandoned_private_temporaries_on_restart() { - let directory = tempfile::TempDir::new().unwrap(); - let root = directory.path().join("directory-cache"); - std::fs::create_dir_all(&root).unwrap(); - std::fs::write(root.join(".tmp-old"), b"partial").unwrap(); - std::fs::write(root.join(".index-tmp-old"), b"partial").unwrap(); - std::fs::write(root.join("unrelated"), b"keep").unwrap(); - let filesystem: Arc = Arc::new(DirectFileSystem); - let _cache = DirectoryCache::new(filesystem, root.clone(), 64); - assert!(!root.join(".tmp-old").exists()); - assert!(!root.join(".index-tmp-old").exists()); - assert!(root.join("unrelated").exists()); -} - -#[cfg(feature = "replica")] -#[test] -fn directory_cache_serializes_concurrent_fills_for_one_key() { - let directory = tempfile::TempDir::new().unwrap(); - let root = directory.path().join("directory-cache"); - let filesystem: Arc = Arc::new(DirectFileSystem); - let cache = Arc::new(DirectoryCache::new(Arc::clone(&filesystem), root, 64)); - std::thread::scope(|scope| { - for _ in 0..8 { - let cache = Arc::clone(&cache); - scope.spawn(move || { - cache.put("same-key", b"verified", 64).unwrap(); - }); - } - }); - assert_eq!( - cache.get("same-key", 64).unwrap(), - Some(b"verified".to_vec()) - ); - assert_eq!(cache.stats().entries(), 1); - assert_eq!(cache.stats().bytes(), 8); - assert_eq!(cache.budget.used(), 8); -} - -struct TestClock; -impl Clock for TestClock { - fn unix_millis(&self) -> i64 { - 123456789 - } - fn file_age(&self, _: &Path) -> io::Result { - Ok(Duration::ZERO) - } -} - -#[derive(Default)] -struct FaultFs { - reject_create: AtomicBool, - panic_create: AtomicBool, - creates: AtomicUsize, -} -impl FileSystem for FaultFs { - fn open(&self, path: &Path) -> io::Result> { - DirectFileSystem.open(path) - } - fn open_rw(&self, path: &Path) -> io::Result> { - DirectFileSystem.open_rw(path) - } - fn create(&self, path: &Path) -> io::Result> { - self.creates.fetch_add(1, Ordering::SeqCst); - assert!( - !self.panic_create.load(Ordering::SeqCst), - "injected cache panic" - ); - if self.reject_create.load(Ordering::SeqCst) { - return Err(io::Error::new( - io::ErrorKind::StorageFull, - "injected artifact failure", - )); - } - DirectFileSystem.create(path) - } - fn file_len(&self, path: &Path) -> io::Result { - DirectFileSystem.file_len(path) - } - fn create_dir_all(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.create_dir_all(path) - } - fn rename(&self, from: &Path, to: &Path) -> io::Result<()> { - DirectFileSystem.rename(from, to) - } - fn remove_file(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.remove_file(path) - } - fn canonicalize(&self, path: &Path) -> io::Result { - DirectFileSystem.canonicalize(path) - } - fn exists(&self, path: &Path) -> io::Result { - DirectFileSystem.exists(path) - } - fn create_dir(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.create_dir(path) - } - fn sync_parent(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.sync_parent(path) - } - fn persist_new(&self, path: &Path, bytes: &[u8]) -> io::Result<()> { - DirectFileSystem.persist_new(path, bytes) - } - fn persist_file_new(&self, source: &Path, destination: &Path) -> io::Result<()> { - DirectFileSystem.persist_file_new(source, destination) - } -} - -#[test] -fn injected_clock_and_capture_filesystem_reach_real_sqlite_transactions() { - let directory = tempfile::TempDir::new().unwrap(); - let filesystem = Arc::new(FaultFs::default()); - let host = Host::default() - .with_clock(Arc::new(TestClock)) - .with_filesystem(filesystem.clone()); - let mut db = crate::Db::open_with_host( - &directory.path().join("db.sqlite"), - crate::Limits::default(), - host, - ) - .unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - let batch = db.capture().unwrap(); - let bytes = std::fs::read(batch.segments[0].path()).unwrap(); - assert_eq!( - crate::ltx::Header::parse(&bytes).unwrap().timestamp, - 123456789 - ); - db.transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(2)")) - .unwrap(); - filesystem.reject_create.store(true, Ordering::SeqCst); - assert!( - matches!(db.capture(), Err(crate::CrabError::Io(error)) if error.kind() == io::ErrorKind::StorageFull) - ); - assert!(matches!( - db.transaction(|_| Ok(())), - Err(crate::CrabError::Fenced) - )); -} - -#[test] -fn synced_scratch_install_never_replaces_a_destination() { - let directory = tempfile::TempDir::new().unwrap(); - let destination = directory.path().join("database.sqlite"); - let first = directory.path().join("first.scratch"); - let mut file = DirectFileSystem.create(&first).unwrap(); - file.write_all(b"first").unwrap(); - file.sync_all().unwrap(); - drop(file); - DirectFileSystem - .persist_file_new(&first, &destination) - .unwrap(); - assert_eq!(std::fs::read(&destination).unwrap(), b"first"); - assert!(!first.exists()); - - let second = directory.path().join("second.scratch"); - let mut file = DirectFileSystem.create(&second).unwrap(); - file.write_all(b"second").unwrap(); - file.sync_all().unwrap(); - drop(file); - assert_eq!( - DirectFileSystem - .persist_file_new(&second, &destination) - .unwrap_err() - .kind(), - io::ErrorKind::AlreadyExists - ); - assert_eq!(std::fs::read(destination).unwrap(), b"first"); - assert_eq!(std::fs::read(second).unwrap(), b"second"); -} - -#[cfg(feature = "replica")] -#[tokio::test] -async fn dropped_executor_jobs_return_errors_without_hanging() { - struct DroppingExecutor; - impl Executor for DroppingExecutor { - fn dispatch(&self, _: Box) -> io::Result<()> { - Ok(()) - } - fn start_worker(&self, job: Box) -> io::Result> { - TokioExecutor.start_worker(job) - } - } - let host = Host::default().with_executor(Arc::new(DroppingExecutor)); - assert!(host.run(|| 42).await.is_err()); - assert_eq!(Host::default().run(|| 42).await.unwrap(), 42); - let directory = tempfile::TempDir::new().unwrap(); - let cache = Host::default() - .with_directory_cache(directory.path().join("cache")) - .await - .unwrap() - .with_executor(Arc::new(DroppingExecutor)); - cache - .directory_cache_put("node".into(), b"verified".to_vec(), 64) - .unwrap(); - tokio::time::timeout(Duration::from_secs(1), cache.drain_cache_fills()) - .await - .unwrap(); - assert_eq!(cache.directory_cache_stats().unwrap().entries(), 0); -} - -#[cfg(feature = "replica")] -#[tokio::test] -async fn rejected_cache_dispatch_releases_its_key_and_admission() { - struct RejectExecutor; - impl Executor for RejectExecutor { - fn dispatch(&self, _: Box) -> io::Result<()> { - Err(io::Error::other("injected executor rejection")) - } - fn start_worker(&self, job: Box) -> io::Result> { - TokioExecutor.start_worker(job) - } - } - let directory = tempfile::TempDir::new().unwrap(); - let jobs = Arc::new(tokio::sync::Semaphore::new(1)); - let host = Host::default() - .with_job_slots(jobs.clone()) - .with_directory_cache(directory.path().join("cache")) - .await - .unwrap(); - assert!( - host.clone() - .with_executor(Arc::new(RejectExecutor)) - .directory_cache_put("node".into(), b"verified".to_vec(), 64) - .is_err() - ); - assert_eq!(jobs.available_permits(), 1); - host.directory_cache_put("node".into(), b"verified".to_vec(), 64) - .unwrap(); - tokio::time::timeout(Duration::from_secs(1), host.drain_cache_fills()) - .await - .unwrap(); - assert_eq!(host.directory_cache_stats().unwrap().entries(), 1); -} - -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cancelled_waiters_do_not_release_running_job_or_recovery_admission() { - let jobs = Arc::new(tokio::sync::Semaphore::new(1)); - let recovery = Arc::new(tokio::sync::Semaphore::new(1)); - let dirty = Arc::new(tokio::sync::Semaphore::new(1)); - let scratch = Arc::new(tokio::sync::Semaphore::new(1)); - let host = Host::default() - .with_job_slots(jobs.clone()) - .with_recovery_slots(recovery.clone()) - .with_dirty_slots(dirty.clone()) - .with_scratch_slots(scratch.clone()); - let scope = host - .for_recovery() - .await - .unwrap() - .for_scratch(1 << 20) - .await - .unwrap(); - let (started, entered) = tokio::sync::oneshot::channel(); - let (release, blocked) = std::sync::mpsc::channel(); - let task = tokio::spawn(async move { - scope - .run(move || { - let _ = started.send(()); - let _ = blocked.recv(); - }) - .await - }); - entered.await.unwrap(); - task.abort(); - assert!(task.await.unwrap_err().is_cancelled()); - assert_eq!(jobs.available_permits(), 0); - assert_eq!(recovery.available_permits(), 0); - assert_eq!(dirty.available_permits(), 0); - assert_eq!(scratch.available_permits(), 0); - release.send(()).unwrap(); - let _job = tokio::time::timeout(Duration::from_secs(2), jobs.acquire()) - .await - .unwrap() - .unwrap(); - let _recovery = tokio::time::timeout(Duration::from_secs(2), recovery.acquire()) - .await - .unwrap() - .unwrap(); - let _dirty = tokio::time::timeout(Duration::from_secs(2), dirty.acquire()) - .await - .unwrap() - .unwrap(); - let _scratch = tokio::time::timeout(Duration::from_secs(2), scratch.acquire()) - .await - .unwrap() - .unwrap(); -} - -#[cfg(feature = "replica")] -#[tokio::test] -async fn closed_admission_returns_errors_instead_of_panicking() { - let slots = Arc::new(tokio::sync::Semaphore::new(1)); - slots.close(); - let host = Host::default() - .with_io_slots(slots.clone()) - .with_job_slots(slots.clone()) - .with_recovery_slots(slots.clone()) - .with_scratch_slots(slots); - assert!(host.run(|| 1).await.is_err()); - assert!(host.io_permit().await.is_err()); - assert!(host.for_recovery().await.is_err()); - assert!(host.for_scratch(1 << 20).await.is_err()); -} - -#[cfg(feature = "replica")] -#[tokio::test] -async fn scratch_monitor_rechecks_total_reservation_and_releases_rejection() { - struct RecordingScratch { - bytes: AtomicU64, - reject: AtomicBool, - } - impl ScratchMonitor for RecordingScratch { - fn ensure_available(&self, reserved_bytes: u64) -> io::Result<()> { - self.bytes.store(reserved_bytes, Ordering::Release); - if self.reject.load(Ordering::Acquire) { - return Err(io::Error::from(io::ErrorKind::StorageFull)); - } - Ok(()) - } - } - - let slots = Arc::new(tokio::sync::Semaphore::new(3)); - let monitor = Arc::new(RecordingScratch { - bytes: AtomicU64::new(0), - reject: AtomicBool::new(false), - }); - let host = Host::default() - .with_scratch_slots(slots.clone()) - .with_scratch_monitor(monitor.clone()); - let first = host.for_scratch(1 << 20).await.unwrap(); - assert_eq!(monitor.bytes.load(Ordering::Acquire), 1 << 20); - let second = host.for_scratch(2 << 20).await.unwrap(); - assert_eq!(monitor.bytes.load(Ordering::Acquire), 3 << 20); - drop((first, second)); - - monitor.reject.store(true, Ordering::Release); - - assert!(matches!( - host.for_scratch(1 << 20).await, - Err(crate::CrabError::Io(error)) if error.kind() == io::ErrorKind::StorageFull - )); - assert_eq!(monitor.bytes.load(Ordering::Acquire), 1 << 20); - assert_eq!(slots.available_permits(), 3); -} diff --git a/crates/crab-ltx/src/error.rs b/crates/crab-ltx/src/error.rs deleted file mode 100644 index 8af529a9a..000000000 --- a/crates/crab-ltx/src/error.rs +++ /dev/null @@ -1,506 +0,0 @@ -//! Errors retain their local I/O, SQLite, or codec cause. - -use std::fmt; -use std::time::Duration; - -/// Result of a local replication operation. -pub type Result = std::result::Result; - -/// Bounded resource that a capture, recovery, or replica step refused to exceed. -/// -/// Callers map these classes onto their own capacity errors, so the variant — -/// not a message string — is the contract. `InvalidState` messages stay free -/// form because no caller dispatches on them. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum LimitKind { - /// One captured Cell bundle exceeded the bundle byte budget. - CapturedCellBundleBytes, - /// One captured bundle exceeded the bundle entry budget. - BundleEntries, - /// One bundle exceeded the bundle byte budget. - BundleBytes, - /// One bundle footer exceeded its byte budget. - BundleFooter, - /// One captured Cell LTX object exceeded the object byte budget. - CapturedCellLtxBytes, - /// One captured LTX object exceeded the object byte budget. - CapturedLtxBytes, - /// One captured Cell bundle exceeded the bundle byte budget. - CellBundleBytes, - /// The Cell root exceeded its byte budget. - CellRootBytes, - /// The Cell root exceeded its segment-count budget. - CellRootSegments, - /// One Cell scale-load batch exceeded its byte budget. - CellScaleBytes, - /// The Cell scale-load checksum exceeded its byte budget. - CellScaleChecksumLength, - /// The Cell scale-load run exceeded its sequence budget. - CellScaleSequence, - /// The Cell root descriptor pages exceeded their page budget. - CellRootSegmentPages, - /// One Cell segment page exceeded its byte budget. - CellSegmentPageBytes, - /// The checksum sidecar exceeded its byte budget. - ChecksumFileBytes, - /// A compaction body spool exceeded its byte budget. - CompactionBodySpool, - /// A compaction index spool exceeded its byte budget. - CompactionIndexSpool, - /// A compaction read set exceeded its input budget. - CompactionInputs, - /// The database exceeded its byte budget. - DatabaseBytes, - /// The database page size fell outside the supported range. - DatabasePageSize, - /// The database exceeded its page-count budget. - DatabasePages, - /// The host refused to charge the reserved host resource units. - HostResourceUnits, - /// The host refused to admit the requested local disk bytes. - LocalDiskBytes, - /// One LTX object exceeded the object byte budget. - LtxBytes, - /// One LTX file exceeded its byte budget. - LtxFileBytes, - /// The LTX page index exceeded its byte budget. - LtxPageIndexBytes, - /// One node frame body exceeded its byte budget. - NodeFrameBody, - /// One node frame exceeded its byte budget. - NodeFrameBytes, - /// The paged request queue exceeded its entry budget. - PagedRequestQueue, - /// A recovery plan exceeded its byte budget. - PlanBytes, - /// A recovery plan exceeded its segment budget. - PlanSegments, - /// Retained replica bytes exceeded the retained-bytes budget. - RetainedBytes, - /// Retained capture artifacts exceeded the session plan budget. - RetainedCaptureArtifacts, - /// Scratch files exceeded the scratch disk byte budget. - ScratchDiskBytes, -} - -impl LimitKind { - /// Stable diagnostic text for this limit class. - #[must_use] - pub const fn as_str(self) -> &'static str { - match self { - Self::CapturedCellBundleBytes => "captured Cell bundle bytes", - Self::BundleEntries => "bundle entries", - Self::BundleBytes => "bundle bytes", - Self::BundleFooter => "bundle footer", - Self::CapturedCellLtxBytes => "captured Cell LTX bytes", - Self::CapturedLtxBytes => "captured LTX bytes", - Self::CellBundleBytes => "Cell bundle bytes", - Self::CellRootBytes => "Cell root bytes", - Self::CellRootSegments => "Cell root segments", - Self::CellScaleBytes => "Cell scale bytes", - Self::CellScaleChecksumLength => "Cell scale checksum length", - Self::CellScaleSequence => "Cell scale sequence", - Self::CellRootSegmentPages => "Cell root segment pages", - Self::CellSegmentPageBytes => "Cell segment page bytes", - Self::ChecksumFileBytes => "checksum file bytes", - Self::CompactionBodySpool => "compaction body spool", - Self::CompactionIndexSpool => "compaction index spool", - Self::CompactionInputs => "compaction inputs", - Self::DatabaseBytes => "database bytes", - Self::DatabasePageSize => "database page size", - Self::DatabasePages => "database pages", - Self::HostResourceUnits => "host resource units", - Self::LocalDiskBytes => "local disk bytes", - Self::LtxBytes => "LTX bytes", - Self::LtxFileBytes => "LTX file bytes", - Self::LtxPageIndexBytes => "LTX page index bytes", - Self::NodeFrameBody => "node frame body", - Self::NodeFrameBytes => "node frame bytes", - Self::PagedRequestQueue => "paged request queue", - Self::PlanBytes => "plan bytes", - Self::PlanSegments => "plan segments", - Self::RetainedBytes => "retained bytes", - Self::RetainedCaptureArtifacts => "retained capture artifacts; rotate session", - Self::ScratchDiskBytes => "scratch disk bytes", - } - } -} - -impl fmt::Display for LimitKind { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter.write_str(self.as_str()) - } -} - -/// Failure from a managed transaction without erasing a handler's domain error. -/// -/// `Operation` proves SQLite rolled the transaction back. `Sqlite` includes -/// begin, rollback, or commit failures; commit failures fence the writer because -/// their outcome can be ambiguous. `Admission` occurs before SQLite starts and -/// leaves the writer reusable. `Capture` occurs after a successful commit while -/// establishing the WAL cut required for later LTX capture. -#[derive(Debug, thiserror::Error)] -pub enum TransactionError { - /// Resource admission failed before SQLite started; the writer stays usable. - #[error("transaction resource admission failed")] - Admission(#[source] CrabError), - /// The caller's operation failed after SQLite rolled the transaction back. - #[error("transaction operation failed")] - Operation(#[source] E), - /// SQLite rejected the transaction; an ambiguous commit fences the writer. - #[error("SQLite transaction failed")] - Sqlite(#[source] rusqlite::Error), - /// The commit succeeded but its WAL cut could not be established. - #[error("WAL capture boundary failed")] - Capture(#[source] CrabError), -} - -/// Failure from a read-only managed-database callback. -#[derive(Debug, thiserror::Error)] -pub enum QueryError { - /// The caller's read-only operation failed. - #[error("query operation failed")] - Operation(#[source] E), - /// SQLite rejected the read-only boundary. - #[error("SQLite read-only boundary failed")] - Sqlite(#[source] rusqlite::Error), - /// The database cannot serve the query in its current state. - #[error("managed database cannot serve the query")] - State(#[source] CrabError), -} - -/// Capture and recovery failures; none imply remote publication succeeded. -#[derive(Debug, thiserror::Error)] -pub enum CrabError { - /// Object-store transport failed. - #[cfg(feature = "replica")] - #[error("object-store replication failure: {0}")] - Storage(#[from] crab_storage::StorageError), - /// Replica metadata was not valid JSON. - #[cfg(feature = "replica")] - #[error("invalid replica metadata: {0}")] - Json(#[from] serde_json::Error), - /// A replication task failed to join. - #[cfg(feature = "replica")] - #[error("replication task failed: {0}")] - Task(#[from] tokio::task::JoinError), - /// LTX data did not carry the checksum its lineage expects. - #[error("LTX checksum mismatch")] - ChecksumMismatch, - /// LTX data failed structural validation. - #[error("LTX file corrupted")] - LTXCorrupted, - /// A referenced LTX file does not exist. - #[error("LTX file missing")] - LTXMissing, - /// A transaction was requested while none is open. - #[error("transaction not available")] - TxNotAvailable, - /// Local I/O failed. - #[error("I/O failure: {0}")] - Io(#[from] std::io::Error), - /// SQLite failed outside a caller-managed transaction. - #[error("SQLite failure: {0}")] - Sqlite(#[from] rusqlite::Error), - /// A configured admission bound was exceeded. - #[error("resource limit exceeded: {0}")] - Limit(LimitKind), - /// Sparse page I/O exceeded its deadline. - #[cfg(feature = "replica")] - #[error("sparse page I/O exceeded its deadline")] - Deadline, - /// Local replication state is invalid; the message names the invariant. - #[error("invalid local replication state: {0}")] - InvalidState(&'static str), - /// Capture was fenced; close the handle and restore an authoritative plan. - #[error("capture failed; close this handle and restore an authoritative plan")] - Fenced, - /// Any other failure, kept boxed so the source survives. - #[error("{0}")] - Other(#[from] Box), -} - -impl CrabError { - /// Reports whether compacting an existing Cell graph can admit an append. - #[must_use] - pub fn is_cell_graph_limit(&self) -> bool { - matches!( - self, - Self::Limit(LimitKind::CellRootSegments | LimitKind::CellRootBytes) - ) - } - - /// Classifies what a caller may do after this failure. - /// - /// The class is the contract: a caller decides between retrying, refusing - /// the request, reconciling an ambiguous outcome, and fencing its handle - /// from this value alone. No caller may dispatch on error messages. - #[must_use] - pub fn classify(&self) -> FailureClass { - match self { - #[cfg(feature = "replica")] - Self::Storage(error) => storage_failure_class(crab_storage::retry_class(error)), - #[cfg(feature = "replica")] - Self::Json(_) => FailureClass::Permanent, - #[cfg(feature = "replica")] - Self::Task(_) => FailureClass::Ambiguous, - Self::ChecksumMismatch - | Self::LTXCorrupted - | Self::LTXMissing - | Self::TxNotAvailable => FailureClass::Permanent, - Self::Io(error) => io_failure_class(error), - Self::Sqlite(error) => sqlite_failure_class(error), - Self::Limit(_) => FailureClass::Capacity, - #[cfg(feature = "replica")] - Self::Deadline => FailureClass::Retryable { after: None }, - Self::InvalidState(_) | Self::Fenced => FailureClass::Fenced, - Self::Other(_) => FailureClass::Ambiguous, - } - } -} - -/// What a caller may do after a [`CrabError`]. -/// -/// The class never depends on the failure text, so a caller can branch on it -/// across versions without matching strings. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum FailureClass { - /// No acknowledged side effect exists, so the same request may be attempted - /// again, no earlier than `after` when a provider named a delay. - Retryable { - /// Provider-requested minimum delay before the next attempt. - after: Option, - }, - /// The request cannot succeed while its inputs or selected state stay the - /// same; the caller must choose different inputs, artifacts, or state. - Permanent, - /// A declared bound or the environment refused the work before it could - /// produce an acknowledged side effect. Freeing the resource or raising the - /// bound is required: an identical retry fails the same way. - Capacity, - /// Work that may already have taken effect has an unknown outcome; the - /// caller must reconcile before retrying. - Ambiguous, - /// The managed handle or session is fenced; close it and restore - /// authoritative state before serving the Cell again. - Fenced, -} - -impl FailureClass { - /// Reports whether a caller holding its own attempt budget may retry. - #[must_use] - pub fn is_retryable(self) -> bool { - matches!(self, Self::Retryable { .. }) - } - - /// Returns the provider's retry delay, when it named one. - #[must_use] - pub fn retry_after(self) -> Option { - match self { - Self::Retryable { after } => after, - _ => None, - } - } -} - -#[cfg(feature = "replica")] -fn storage_failure_class(class: crab_storage::RetryClass) -> FailureClass { - use crab_storage::RetryClass; - match class { - RetryClass::Transient | RetryClass::StateDependent | RetryClass::InspectErrno => { - FailureClass::Retryable { after: None } - } - RetryClass::Throttled { retry_after } => FailureClass::Retryable { after: retry_after }, - // The storage layer already spent its one bounded retry for these - // classes, so a caller that retried again could loop on a corrupt read. - RetryClass::FatalAfterOneRetry | RetryClass::Fatal => FailureClass::Permanent, - } -} - -fn io_failure_class(error: &std::io::Error) -> FailureClass { - use std::io::ErrorKind; - match error.kind() { - ErrorKind::Interrupted | ErrorKind::WouldBlock | ErrorKind::TimedOut => { - FailureClass::Retryable { after: None } - } - ErrorKind::StorageFull | ErrorKind::QuotaExceeded | ErrorKind::OutOfMemory => { - FailureClass::Capacity - } - ErrorKind::NotFound - | ErrorKind::PermissionDenied - | ErrorKind::AlreadyExists - | ErrorKind::InvalidInput - | ErrorKind::InvalidData - | ErrorKind::Unsupported => FailureClass::Permanent, - // Any other I/O failure can leave a partially written artifact, so the - // caller reconciles instead of assuming the operation did nothing. - _ => FailureClass::Ambiguous, - } -} - -fn sqlite_failure_class(error: &rusqlite::Error) -> FailureClass { - use rusqlite::ErrorCode; - match error.sqlite_error_code() { - Some( - ErrorCode::DatabaseBusy | ErrorCode::DatabaseLocked | ErrorCode::OperationInterrupted, - ) => FailureClass::Retryable { after: None }, - Some(ErrorCode::DiskFull | ErrorCode::OutOfMemory) => FailureClass::Capacity, - // SQLite rolls a failed statement back, but the underlying file I/O - // failure may have written a partial page. - Some(ErrorCode::SystemIoFailure) => FailureClass::Ambiguous, - // Every remaining SQLite failure is statement-atomic and repeatable with - // the same inputs, so a caller must change the request or fence. - Some(_) | None => FailureClass::Permanent, - } -} - -#[cfg(test)] -mod tests { - use super::{CrabError, FailureClass, LimitKind}; - use std::io::ErrorKind; - use std::time::Duration; - - #[test] - fn only_cell_graph_admission_limits_are_compaction_retryable() { - assert!(CrabError::Limit(LimitKind::CellRootSegments).is_cell_graph_limit()); - assert!(CrabError::Limit(LimitKind::CellRootBytes).is_cell_graph_limit()); - assert!(!CrabError::Limit(LimitKind::LtxFileBytes).is_cell_graph_limit()); - assert!(!CrabError::ChecksumMismatch.is_cell_graph_limit()); - } - - #[test] - fn limit_kinds_keep_their_diagnostic_text() { - assert_eq!( - CrabError::Limit(LimitKind::LocalDiskBytes).to_string(), - "resource limit exceeded: local disk bytes" - ); - assert_eq!(LimitKind::CellRootBytes.as_str(), "Cell root bytes"); - } - - #[test] - fn state_selection_failures_are_permanent() { - for error in [ - CrabError::ChecksumMismatch, - CrabError::LTXCorrupted, - CrabError::LTXMissing, - CrabError::TxNotAvailable, - ] { - assert_eq!(error.classify(), FailureClass::Permanent, "{error}"); - } - } - - #[test] - fn declared_limits_are_capacity_refusals() { - for kind in [ - LimitKind::LtxFileBytes, - LimitKind::LocalDiskBytes, - LimitKind::CapturedLtxBytes, - LimitKind::HostResourceUnits, - ] { - assert_eq!( - CrabError::Limit(kind).classify(), - FailureClass::Capacity, - "{kind}" - ); - } - } - - #[test] - fn fenced_states_require_restoring_authoritative_state() { - assert_eq!(CrabError::Fenced.classify(), FailureClass::Fenced); - assert_eq!( - CrabError::InvalidState("sessions cannot be reused").classify(), - FailureClass::Fenced - ); - } - - #[test] - fn io_failures_separate_retryable_capacity_and_ambiguous_outcomes() { - let cases = [ - ( - ErrorKind::Interrupted, - FailureClass::Retryable { after: None }, - ), - ( - ErrorKind::WouldBlock, - FailureClass::Retryable { after: None }, - ), - (ErrorKind::TimedOut, FailureClass::Retryable { after: None }), - (ErrorKind::StorageFull, FailureClass::Capacity), - (ErrorKind::QuotaExceeded, FailureClass::Capacity), - (ErrorKind::OutOfMemory, FailureClass::Capacity), - (ErrorKind::NotFound, FailureClass::Permanent), - (ErrorKind::PermissionDenied, FailureClass::Permanent), - (ErrorKind::AlreadyExists, FailureClass::Permanent), - (ErrorKind::InvalidInput, FailureClass::Permanent), - (ErrorKind::InvalidData, FailureClass::Permanent), - (ErrorKind::Unsupported, FailureClass::Permanent), - // A failed page write can leave a partial artifact behind. - (ErrorKind::Other, FailureClass::Ambiguous), - ]; - for (kind, expected) in cases { - let error = CrabError::Io(std::io::Error::from(kind)); - assert_eq!(error.classify(), expected, "{kind:?}"); - } - } - - #[test] - fn sqlite_failures_classify_by_code_not_message() { - use rusqlite::ffi; - // Raw SQLite result codes, one of them extended, so the classification - // is pinned to the wire code. `rusqlite::ErrorCode::DatabaseBusy as i32` - // is the Rust enum discriminant (3 = SQLITE_PERM), never the result - // code, so a caller must not use it to build a failure. - let busy_snapshot = ffi::SQLITE_BUSY | (4 << 8); - let cases = [ - (ffi::SQLITE_BUSY, FailureClass::Retryable { after: None }), - (ffi::SQLITE_LOCKED, FailureClass::Retryable { after: None }), - ( - ffi::SQLITE_INTERRUPT, - FailureClass::Retryable { after: None }, - ), - (ffi::SQLITE_FULL, FailureClass::Capacity), - (ffi::SQLITE_NOMEM, FailureClass::Capacity), - (ffi::SQLITE_IOERR, FailureClass::Ambiguous), - (ffi::SQLITE_CORRUPT, FailureClass::Permanent), - (ffi::SQLITE_READONLY, FailureClass::Permanent), - (ffi::SQLITE_NOTADB, FailureClass::Permanent), - ( - rusqlite::ErrorCode::DatabaseBusy as i32, - FailureClass::Permanent, - ), - ]; - for (code, expected) in cases { - let error = - CrabError::Sqlite(rusqlite::Error::SqliteFailure(ffi::Error::new(code), None)); - assert_eq!(error.classify(), expected, "{code}"); - } - - let extended = CrabError::Sqlite(rusqlite::Error::SqliteFailure( - ffi::Error::new(busy_snapshot), - None, - )); - assert_eq!( - extended.classify(), - FailureClass::Retryable { after: None }, - "extended codes keep their primary code" - ); - } - - #[test] - fn unknown_failures_stay_ambiguous() { - let error = CrabError::Other(Box::new(std::io::Error::other("unclassified"))); - assert_eq!(error.classify(), FailureClass::Ambiguous); - } - - #[test] - fn retry_classes_carry_their_provider_hint() { - let hinted = FailureClass::Retryable { - after: Some(Duration::from_millis(250)), - }; - assert!(hinted.is_retryable()); - assert_eq!(hinted.retry_after(), Some(Duration::from_millis(250))); - assert!(!FailureClass::Capacity.is_retryable()); - assert_eq!(FailureClass::Capacity.retry_after(), None); - } -} diff --git a/crates/crab-ltx/src/format_tests.rs b/crates/crab-ltx/src/format_tests.rs deleted file mode 100644 index 0c0d6861f..000000000 --- a/crates/crab-ltx/src/format_tests.rs +++ /dev/null @@ -1,500 +0,0 @@ -//! Independent literal-block vectors and a bitwise CRC oracle. These do not -//! call the imported compressor, encoder, or crc-fast implementation. - -use crate::{ - CHECKSUM_FLAG, CrabError, Limits, LocalSegment, Position, SegmentInfo, VerifiedPlan, ltx, - restore_exact, -}; -use std::path::PathBuf; - -pub(crate) fn decode_file(bytes: &[u8]) -> crate::Result { - ltx::inspect_reader(std::io::Cursor::new(bytes)).map(|(file, _, _)| file) -} - -/// Page body the external vectors were generated with. -/// -/// `tests/vectors/generate` writes the same pattern, so a change to one side -/// without the other fails the round-trip comparison instead of passing on -/// regenerated bytes. -fn vector_page(page_size: u32, pgno: u32) -> Vec { - (0..page_size) - .map(|index| ((pgno * 37 + index) % 251) as u8) - .collect() -} - -fn vector_path(name: &str) -> PathBuf { - PathBuf::from(env!("CARGO_MANIFEST_DIR")) - .join("tests/vectors") - .join(name) -} - -/// Loads one external vector as bytes plus its exact manifest expectations. -fn external_segment(name: &str) -> (Vec, SegmentInfo) { - let bytes = std::fs::read(vector_path(name)).unwrap(); - let summary = crate::internal::inspect_ltx(&bytes).unwrap(); - let info = SegmentInfo { - min_txid: summary.min_txid, - max_txid: summary.max_txid, - page_size: summary.page_size, - database_pages: summary.commit, - pre_checksum: summary.pre_apply_checksum, - post_checksum: summary.post_apply_checksum, - size_bytes: summary.size_bytes, - blake3: summary.blake3, - }; - (bytes, info) -} - -/// External vectors written by the upstream celld encoder (see UPSTREAM.md). -const EXTERNAL_VECTORS: &[(&str, u32, u32, bool)] = &[ - ("celld-10cb130-snapshot-block-512.ltx", 512, 3, true), - ("celld-10cb130-snapshot-frame-512.ltx", 512, 3, false), - ("celld-10cb130-snapshot-block-4096.ltx", 4096, 4, true), - ("superfly-ltx-v0.5.2-snapshot-block-512.ltx", 512, 3, true), -]; - -#[test] -fn independent_producers_agree_byte_for_byte() { - // The reference Go writer (superfly/ltx v0.5.2, the library Litestream - // uses) and the pinned Celld port must produce identical bytes for the same - // input; `external_snapshots_decode_restore_and_match_our_writer` then - // proves this crate writes those same bytes. - assert_eq!( - std::fs::read(vector_path("superfly-ltx-v0.5.2-snapshot-block-512.ltx")).unwrap(), - std::fs::read(vector_path("celld-10cb130-snapshot-block-512.ltx")).unwrap() - ); -} - -#[test] -fn external_delta_chain_restores_and_matches_our_writer() { - let celld_delta = "celld-10cb130-delta-2-2-512.ltx"; - let reference_delta = "superfly-ltx-v0.5.2-delta-2-2-512.ltx"; - assert_eq!( - std::fs::read(vector_path(celld_delta)).unwrap(), - std::fs::read(vector_path(reference_delta)).unwrap(), - "both reference producers must agree on the successor file" - ); - - let (snapshot_bytes, snapshot_info) = external_segment("celld-10cb130-snapshot-block-512.ltx"); - let (delta_bytes, delta_info) = external_segment(celld_delta); - let delta_pre = delta_info.pre_checksum; - let delta_post = delta_info.post_checksum; - assert_eq!( - delta_pre, snapshot_info.post_checksum, - "the successor continues the snapshot's checksum" - ); - - // A chain of independently produced files restores the exact image: page 1 - // and page 3 from the snapshot, page 2 from the successor. - let temp = tempfile::TempDir::new().unwrap(); - let snapshot_path = temp.path().join("snapshot.ltx"); - let delta_path = temp.path().join("delta.ltx"); - std::fs::write(&snapshot_path, &snapshot_bytes).unwrap(); - std::fs::write(&delta_path, &delta_bytes).unwrap(); - let position = Position { - txid: delta_info.max_txid, - checksum: delta_post, - }; - let plan = VerifiedPlan::new( - &[ - LocalSegment::new(snapshot_path, snapshot_info), - LocalSegment::new(delta_path, delta_info), - ], - position, - Limits::default(), - ) - .unwrap(); - let restored = temp.path().join("restored.sqlite"); - assert_eq!(restore_exact(&plan, &restored).unwrap(), position); - let mut expected = vector_page(512, 1); - expected.extend(vector_page(512, 2).into_iter().map(|byte| 255 - byte)); - expected.extend_from_slice(&vector_page(512, 3)); - assert_eq!(std::fs::read(&restored).unwrap(), expected); - - // This crate's writer reproduces the reference successor bytes exactly. - let next: Vec = vector_page(512, 2) - .into_iter() - .map(|byte| 255 - byte) - .collect(); - let mut encoder = crate::codec::Encoder::new_block(Vec::new()); - encoder - .encode_header(ltx::Header { - version: ltx::VERSION, - flags: 0, - page_size: 512, - commit: 3, - min_txid: crate::Txid(2), - max_txid: crate::Txid(2), - timestamp: 1_700_000_000_000, - pre_apply_checksum: delta_pre, - wal_offset: 0, - wal_size: 0, - wal_salt1: 0, - wal_salt2: 0, - node_id: 0, - }) - .unwrap(); - encoder - .encode_page(ltx::PageHeader { pgno: 2, flags: 0 }, &next) - .unwrap(); - encoder.close(delta_post).unwrap(); - assert_eq!(encoder.into_writer(), delta_bytes); -} - -#[test] -fn external_snapshots_decode_restore_and_match_our_writer() { - for (name, page_size, commit, block) in EXTERNAL_VECTORS { - let bytes = std::fs::read(vector_path(name)).unwrap(); - let summary = crate::internal::inspect_ltx(&bytes).unwrap(); - assert_eq!(summary.page_size, *page_size, "{name}"); - assert_eq!(summary.commit, *commit, "{name}"); - assert_eq!((summary.min_txid, summary.max_txid), (1, 1), "{name}"); - assert_eq!(summary.pages, *commit, "{name}"); - assert_eq!(summary.size_bytes, bytes.len() as u64, "{name}"); - assert_eq!(summary.blake3, *blake3::hash(&bytes).as_bytes(), "{name}"); - - // The upstream file restores the exact image the pages encode. - let temp = tempfile::TempDir::new().unwrap(); - let segment_path = temp.path().join(name); - std::fs::write(&segment_path, &bytes).unwrap(); - let info = SegmentInfo { - min_txid: summary.min_txid, - max_txid: summary.max_txid, - page_size: summary.page_size, - database_pages: summary.commit, - pre_checksum: 0, - post_checksum: { - let mut checksum = CHECKSUM_FLAG; - for pgno in 1..=*commit { - checksum = - CHECKSUM_FLAG | (checksum ^ page_sum(pgno, &vector_page(*page_size, pgno))); - } - checksum - }, - size_bytes: summary.size_bytes, - blake3: summary.blake3, - }; - let position = Position { - txid: summary.max_txid, - checksum: info.post_checksum, - }; - let post_checksum = info.post_checksum; - let plan = VerifiedPlan::new( - &[LocalSegment::new(segment_path, info)], - position, - Limits::default(), - ) - .unwrap(); - let restored = temp.path().join("restored.sqlite"); - assert_eq!(restore_exact(&plan, &restored).unwrap(), position, "{name}"); - let mut expected = Vec::new(); - for pgno in 1..=*commit { - expected.extend_from_slice(&vector_page(*page_size, pgno)); - } - assert_eq!(std::fs::read(&restored).unwrap(), expected, "{name}"); - - // A sized-block vector must be byte-identical to what this crate writes - // for the same header and pages, so the two writers cannot drift. - if *block { - let pages: Vec<(u32, Vec)> = (1..=*commit) - .map(|pgno| (pgno, vector_page(*page_size, pgno))) - .collect(); - let mut encoder = crate::codec::Encoder::new_block(Vec::new()); - encoder - .encode_header(ltx::Header { - version: ltx::VERSION, - flags: 0, - page_size: *page_size, - commit: *commit, - min_txid: crate::Txid(1), - max_txid: crate::Txid(1), - timestamp: 1_700_000_000_000, - pre_apply_checksum: 0, - wal_offset: 0, - wal_size: 0, - wal_salt1: 0, - wal_salt2: 0, - node_id: 0, - }) - .unwrap(); - for (pgno, data) in pages { - encoder - .encode_page(ltx::PageHeader { pgno, flags: 0 }, &data) - .unwrap(); - } - encoder.close(post_checksum).unwrap(); - assert_eq!(encoder.into_writer(), bytes, "{name}"); - } - } -} - -fn crc(bytes: &[u8]) -> u64 { - let mut sum = !0u64; - for byte in bytes { - sum ^= u64::from(*byte); - for _ in 0..8 { - sum = (sum >> 1) - ^ if sum & 1 != 0 { - 0xd800_0000_0000_0000 - } else { - 0 - }; - } - } - !sum -} - -fn page_sum(pgno: u32, data: &[u8]) -> u64 { - let mut bytes = pgno.to_be_bytes().to_vec(); - bytes.extend_from_slice(data); - CHECKSUM_FLAG | crc(&bytes) -} - -fn varint(out: &mut Vec, mut n: u64) { - while n >= 128 { - out.push(n as u8 | 128); - n >>= 7; - } - out.push(n as u8); -} - -// Deliberately permits invalid order, checksums, and index offsets so tests can -// produce malformed files with otherwise valid outer CRC/digest expectations. -fn fixture( - pages: &[(u32, Vec)], - commit: u32, - post: u64, - index_bias: u64, - legacy: bool, -) -> Vec { - use std::io::Write; - let mut bytes = vec![0; 100]; - bytes[0..4].copy_from_slice(b"LTX1"); - bytes[8..12].copy_from_slice(&512u32.to_be_bytes()); - bytes[12..16].copy_from_slice(&commit.to_be_bytes()); - bytes[16..24].copy_from_slice(&1u64.to_be_bytes()); - bytes[24..32].copy_from_slice(&1u64.to_be_bytes()); - let mut hashed = bytes.clone(); - let mut index = Vec::new(); - for (pgno, data) in pages { - let offset = bytes.len() as u64; - let mut header = pgno.to_be_bytes().to_vec(); - header.extend_from_slice(&(if legacy { 0u16 } else { 1u16 }).to_be_bytes()); - let payload = if legacy { - let info = lz4_flex::frame::FrameInfo::new().content_checksum(true); - let mut encoder = lz4_flex::frame::FrameEncoder::with_frame_info(info, Vec::new()); - encoder.write_all(data).unwrap(); - encoder.finish().unwrap() - } else { - // 512 literals: 15 in the token, then 255 + 242 extension bytes. - let mut payload = vec![0xf0, 255, 242]; - payload.extend_from_slice(data); - header.extend_from_slice(&(payload.len() as u32).to_be_bytes()); - payload - }; - bytes.extend_from_slice(&header); - bytes.extend_from_slice(&payload); - hashed.extend_from_slice(&header); - hashed.extend_from_slice(data); - varint(&mut index, u64::from(*pgno)); - varint(&mut index, offset + index_bias); - varint(&mut index, bytes.len() as u64 - offset); - } - index.push(0); - let mut tail = vec![0; 6]; - tail.extend_from_slice(&index); - tail.extend_from_slice(&(index.len() as u64).to_be_bytes()); - tail.extend_from_slice(&post.to_be_bytes()); - bytes.extend_from_slice(&tail); - hashed.extend_from_slice(&tail); - bytes.extend_from_slice(&(CHECKSUM_FLAG | crc(&hashed)).to_be_bytes()); - bytes -} - -#[test] -fn independent_crc_and_both_ltx_page_encodings_match() { - assert_eq!(crc(b"123456789"), 0xb909_56c7_75a4_1001); - let data = vec![0x39; 512]; - for legacy in [false, true] { - let bytes = fixture(&[(1, data.clone())], 1, page_sum(1, &data), 0, legacy); - let (file, decoded) = ltx::decode_file_with_pages(&bytes).unwrap(); - assert_eq!(decoded, vec![(1, data.clone())]); - assert_eq!(file.trailer.post_apply_checksum, page_sum(1, &data)); - } -} - -#[test] -fn valid_outer_checksums_do_not_hide_bad_page_order_or_index() { - let one = vec![1; 512]; - let two = vec![2; 512]; - for (pages, commit, bias) in [ - (vec![(1, one.clone()), (1, one.clone())], 2, 0), - (vec![(2, two.clone()), (1, one.clone())], 2, 0), - (vec![(1, one.clone())], 2, 0), - (vec![(1, one.clone())], 1, 1), - (vec![(2, two)], 1, 0), - ] { - let sum = pages.iter().fold(CHECKSUM_FLAG, |sum, (pgno, data)| { - CHECKSUM_FLAG | (sum ^ page_sum(*pgno, data)) - }); - assert!(decode_file(&fixture(&pages, commit, sum, bias, false)).is_err()); - } -} - -#[test] -fn footer_checks_the_original_varint_encoding_and_exact_length() { - let data = vec![8; 512]; - let mut bytes = fixture(&[(1, data.clone())], 1, page_sum(1, &data), 0, false); - let sentinel = bytes.len() - 25; - assert_eq!(bytes[sentinel], 0); - // Nonminimal varints are accepted by the existing format. Both their - // encoded length and their original bytes contribute to footer validation. - bytes.splice(sentinel..=sentinel, [0x80, 0]); - let size_offset = bytes.len() - 24; - let old_size = u64::from_be_bytes(bytes[size_offset..size_offset + 8].try_into().unwrap()); - for correct_size in [true, false] { - let size = old_size + u64::from(correct_size); - bytes[size_offset..size_offset + 8].copy_from_slice(&size.to_be_bytes()); - let mut hashed = bytes[..110].to_vec(); - hashed.extend_from_slice(&data); - hashed.extend_from_slice(&bytes[625..bytes.len() - 8]); - let len = bytes.len(); - bytes[len - 8..].copy_from_slice(&(CHECKSUM_FLAG | crc(&hashed)).to_be_bytes()); - let decoded = decode_file(&bytes); - if correct_size { - decoded.unwrap(); - } else { - assert!(matches!(decoded, Err(CrabError::LTXCorrupted))); - } - } -} - -#[test] -fn every_truncated_prefix_is_rejected_without_panicking() { - let data = vec![8; 512]; - let bytes = fixture(&[(1, data.clone())], 1, page_sum(1, &data), 0, false); - for end in 0..bytes.len() { - assert!(decode_file(&bytes[..end]).is_err()); - } -} - -#[test] -fn exact_restore_rejects_checksum_disabled_file() { - let temp = tempfile::TempDir::new().unwrap(); - let data = vec![0; 512]; - let mut bytes = fixture(&[(1, data.clone())], 1, 0, 0, false); - bytes[4..8].copy_from_slice(&2u32.to_be_bytes()); - // Recompute the file CRC with the decompressed page, as mandated by LTX. - let mut hashed = bytes[..110].to_vec(); - hashed.extend_from_slice(&data); - hashed.extend_from_slice(&bytes[625..bytes.len() - 8]); - let len = bytes.len(); - bytes[len - 8..].copy_from_slice(&(CHECKSUM_FLAG | crc(&hashed)).to_be_bytes()); - let file = decode_file(&bytes).unwrap(); - let info = SegmentInfo::from_decoded(&bytes, &file); - let path = temp.path().join("unchecked.ltx"); - std::fs::write(&path, bytes).unwrap(); - let result = VerifiedPlan::new( - &[LocalSegment::new(path, info)], - Position { - txid: 1, - checksum: 0, - }, - Limits::default(), - ); - assert!(matches!(result, Err(CrabError::LTXCorrupted))); -} - -#[test] -fn captured_positions_match_full_database_crc_oracle() { - let temp = tempfile::TempDir::new().unwrap(); - let mut db = crate::Db::open(&temp.path().join("source.sqlite"), Limits::default()).unwrap(); - let mut segments = Vec::new(); - for round in 0..8 { - db.transaction(|tx| { - tx.execute( - "CREATE TABLE IF NOT EXISTS t (id INTEGER PRIMARY KEY, data BLOB)", - [], - )?; - tx.execute("INSERT INTO t(data) VALUES (randomblob(9000))", [])?; - tx.execute("UPDATE t SET data = randomblob(5000) WHERE id % 2 = 0", [])?; - Ok(()) - }) - .unwrap(); - let batch = db.capture().unwrap(); - segments.extend(batch.segments); - let plan = VerifiedPlan::new(&segments, batch.position, Limits::default()).unwrap(); - let path = temp.path().join(format!("restored-{round}.sqlite")); - crate::restore_exact(&plan, &path).unwrap(); - let image = std::fs::read(path).unwrap(); - let sum = image - .as_chunks::<4096>() - .0 - .iter() - .enumerate() - .fold(CHECKSUM_FLAG, |sum, (i, page)| { - CHECKSUM_FLAG | (sum ^ page_sum(i as u32 + 1, page)) - }); - assert_eq!(batch.position.checksum, sum); - } -} - -#[test] -fn altered_delta_predecessor_or_post_state_is_rejected_with_valid_file_crc() { - let temp = tempfile::TempDir::new().unwrap(); - let before = vec![1; 512]; - let after = vec![2; 512]; - let first = fixture(&[(1, before.clone())], 1, page_sum(1, &before), 0, false); - let select = |name: &str, bytes: Vec| { - let decoded = decode_file(&bytes).unwrap(); - let info = SegmentInfo::from_decoded(&bytes, &decoded); - let path = temp.path().join(name); - std::fs::write(&path, bytes).unwrap(); - LocalSegment::new(path, info) - }; - let first = select("first.ltx", first); - for (i, bad_pre) in [true, false].into_iter().enumerate() { - let post = page_sum(1, &after) ^ if bad_pre { 0 } else { 1 }; - let mut bytes = fixture(&[(1, after.clone())], 1, post, 0, false); - bytes[16..24].copy_from_slice(&2u64.to_be_bytes()); - bytes[24..32].copy_from_slice(&2u64.to_be_bytes()); - let pre = page_sum(1, &before) ^ u64::from(bad_pre); - bytes[40..48].copy_from_slice(&pre.to_be_bytes()); - let mut hashed = bytes[..110].to_vec(); - hashed.extend_from_slice(&after); - hashed.extend_from_slice(&bytes[625..bytes.len() - 8]); - let len = bytes.len(); - bytes[len - 8..].copy_from_slice(&(CHECKSUM_FLAG | crc(&hashed)).to_be_bytes()); - let delta = select(&format!("delta-{i}.ltx"), bytes); - let target = delta.info().position(); - let result = VerifiedPlan::new(&[first.clone(), delta], target, Limits::default()); - assert!(matches!(result, Err(CrabError::ChecksumMismatch))); - } -} - -#[test] -fn compressor_round_trips_all_sqlite_page_sizes_and_patterns() { - let mut compressor = crate::lz4_block::Compressor::default(); - let mut random = 0x6a09e667f3bcc908u64; - for size in [512, 1024, 2048, 4096, 8192, 16384, 32768, 65536] { - for pattern in 0..8 { - let page: Vec = (0..size) - .map(|i| { - random ^= random << 13; - random ^= random >> 7; - random ^= random << 17; - match pattern { - 0 => 0, - 1 => 255, - 2 => i as u8, - 3 => (i % 11) as u8, - _ => random as u8, - } - }) - .collect(); - let compressed = compressor.compress(&page).unwrap(); - let restored = lz4_flex::block::decompress(&compressed, size).unwrap(); - assert_eq!(restored, page); - } - } -} diff --git a/crates/crab-ltx/src/hex.rs b/crates/crab-ltx/src/hex.rs deleted file mode 100644 index fc06c5e04..000000000 --- a/crates/crab-ltx/src/hex.rs +++ /dev/null @@ -1,12 +0,0 @@ -//! Lowercase hex encoding for LTX layouts and root documents. - -/// Encodes bytes as lowercase hex, the only form LTX readers accept. -pub(crate) fn encode_hex(bytes: &[u8]) -> String { - const TABLE: &[u8; 16] = b"0123456789abcdef"; - let mut encoded = String::with_capacity(bytes.len() * 2); - for byte in bytes { - encoded.push(TABLE[(byte >> 4) as usize] as char); - encoded.push(TABLE[(byte & 0x0f) as usize] as char); - } - encoded -} diff --git a/crates/crab-ltx/src/host.rs b/crates/crab-ltx/src/host.rs deleted file mode 100644 index cd3b22b60..000000000 --- a/crates/crab-ltx/src/host.rs +++ /dev/null @@ -1,248 +0,0 @@ -//! Bounded synchronous filesystem operations; execution scheduling belongs to the caller. - -use std::fs::File; -use std::io::{self, Read, Write}; -use std::path::Path; -use std::time::{Duration, Instant}; - -pub(crate) struct LtxHost { - pub facilities: crate::Host, - pub max_database_bytes: u64, - pub max_file_bytes: u64, -} - -pub(crate) struct HostFile { - file: Box, - limit: u64, - read_offset: u64, - sequential_write_bytes: Option, -} - -impl HostFile { - pub fn write_all(&mut self, bytes: &[u8]) -> io::Result<()> { - let Some(written) = self.sequential_write_bytes else { - check_size( - self.file.file_len()?.saturating_add(bytes.len() as u64), - self.limit, - )?; - return self.file.write_all(bytes); - }; - let end = written - .checked_add(bytes.len() as u64) - .ok_or_else(|| io::Error::other("LTX local file byte limit exceeded"))?; - check_size(end, self.limit)?; - self.file.write_all(bytes)?; - self.sequential_write_bytes = Some(end); - Ok(()) - } - - pub fn read_exact_at(&mut self, offset: u64, len: usize) -> io::Result> { - check_size(len as u64, self.limit)?; - check_size(self.file.file_len()?, self.limit)?; - let bytes = self.file.read_exact_at(offset, len)?; - if bytes.len() != len { - return Err(io::ErrorKind::UnexpectedEof.into()); - } - Ok(bytes) - } - - #[cfg(feature = "replica")] - pub fn write_all_at(&mut self, offset: u64, bytes: &[u8]) -> io::Result<()> { - let end = offset - .checked_add(bytes.len() as u64) - .ok_or_else(|| io::Error::other("local file offset overflow"))?; - check_size(end, self.limit)?; - self.file.write_all_at(offset, bytes) - } - - pub fn sync_all(&mut self) -> io::Result<()> { - self.file.sync_all() - } - pub fn file_len(&mut self) -> io::Result { - let len = self.file.file_len()?; - check_size(len, self.limit)?; - Ok(len) - } - - #[cfg(feature = "replica")] - pub fn set_len(&mut self, len: u64) -> io::Result<()> { - check_size(len, self.limit)?; - self.file.set_len(len) - } -} - -impl Read for HostFile { - fn read(&mut self, bytes: &mut [u8]) -> io::Result { - let remaining = self.file_len()?.saturating_sub(self.read_offset); - let length = - usize::try_from(remaining.min(bytes.len() as u64)).map_err(io::Error::other)?; - if length == 0 { - return Ok(0); - } - let read = self.file.read_exact_at(self.read_offset, length)?; - bytes[..length].copy_from_slice(&read); - self.read_offset += length as u64; - Ok(length) - } -} - -impl Write for HostFile { - fn write(&mut self, bytes: &[u8]) -> io::Result { - self.write_all(bytes)?; - Ok(bytes.len()) - } - - fn flush(&mut self) -> io::Result<()> { - Ok(()) - } -} - -pub(crate) struct HostMetadata { - pub len: u64, -} - -impl LtxHost { - pub fn check_database_size(&self, size: u64) -> crate::Result<()> { - if size > self.max_database_bytes { - return Err(crate::CrabError::Limit(crate::LimitKind::DatabaseBytes)); - } - Ok(()) - } - pub fn read(&self, path: &Path) -> io::Result> { - let mut file = self.open(path)?; - let len = file.file_len()?; - file.read_exact_at(0, usize::try_from(len).map_err(io::Error::other)?) - } - pub fn open(&self, path: &Path) -> io::Result { - Ok(HostFile { - file: self.facilities.filesystem.open(path)?, - limit: self.max_file_bytes, - read_offset: 0, - sequential_write_bytes: None, - }) - } - #[cfg(feature = "replica")] - pub fn open_rw(&self, path: &Path) -> io::Result { - Ok(HostFile { - file: self.facilities.filesystem.open_rw(path)?, - limit: self.max_file_bytes, - read_offset: 0, - sequential_write_bytes: None, - }) - } - pub fn create(&self, path: &Path) -> io::Result { - Ok(HostFile { - file: self.facilities.filesystem.create(path)?, - limit: self.max_file_bytes, - read_offset: 0, - sequential_write_bytes: Some(0), - }) - } - pub fn metadata(&self, path: &Path) -> io::Result { - Ok(HostMetadata { - len: self.facilities.filesystem.file_len(path)?, - }) - } - - pub fn create_dir_all(&self, path: &Path) -> io::Result<()> { - self.facilities.filesystem.create_dir_all(path) - } - pub fn remove_file(&self, path: &Path) -> io::Result<()> { - self.facilities.filesystem.remove_file(path) - } - pub fn rename(&self, from: &Path, to: &Path) -> io::Result<()> { - // A fresh session owns this directory. Directory fsync seals the new name. - self.facilities.filesystem.rename(from, to) - } - pub fn rename_uncommitted(&self, from: &Path, to: &Path) -> io::Result<()> { - self.facilities.filesystem.rename_uncommitted(from, to) - } - pub fn now_unix_millis(&self) -> i64 { - self.facilities.clock.unix_millis() - } - pub fn now_monotonic(&self) -> Instant { - self.facilities.clock.monotonic() - } - pub fn file_age(&self, path: &Path) -> io::Result { - self.facilities.clock.file_age(path) - } -} - -fn check_size(size: u64, limit: u64) -> io::Result<()> { - if size > limit { - return Err(io::Error::other("LTX local file byte limit exceeded")); - } - Ok(()) -} - -pub(crate) fn sync_parent(path: &Path) -> io::Result<()> { - let parent = path - .parent() - .filter(|p| !p.as_os_str().is_empty()) - .unwrap_or(Path::new(".")); - File::open(parent)?.sync_all() -} - -#[cfg(test)] -mod tests { - use std::sync::{ - Arc, - atomic::{AtomicUsize, Ordering}, - }; - - use super::*; - - struct CountingFile { - inner: File, - file_len_calls: Arc, - } - - impl crate::environment::FileIo for CountingFile { - fn write_all(&mut self, bytes: &[u8]) -> io::Result<()> { - crate::environment::FileIo::write_all(&mut self.inner, bytes) - } - - fn write_all_at(&mut self, offset: u64, bytes: &[u8]) -> io::Result<()> { - crate::environment::FileIo::write_all_at(&mut self.inner, offset, bytes) - } - - fn read_exact_at(&mut self, offset: u64, len: usize) -> io::Result> { - crate::environment::FileIo::read_exact_at(&mut self.inner, offset, len) - } - - fn sync_all(&mut self) -> io::Result<()> { - crate::environment::FileIo::sync_all(&mut self.inner) - } - - fn file_len(&self) -> io::Result { - self.file_len_calls.fetch_add(1, Ordering::Relaxed); - crate::environment::FileIo::file_len(&self.inner) - } - - fn set_len(&mut self, len: u64) -> io::Result<()> { - crate::environment::FileIo::set_len(&mut self.inner, len) - } - } - - #[test] - fn created_output_tracks_its_extent_without_metadata_queries() { - let file_len_calls = Arc::new(AtomicUsize::new(0)); - let mut file = HostFile { - file: Box::new(CountingFile { - inner: tempfile::tempfile().unwrap(), - file_len_calls: Arc::clone(&file_len_calls), - }), - limit: 4, - read_offset: 0, - sequential_write_bytes: Some(0), - }; - - file.write_all(&[1, 2]).unwrap(); - file.write_all(&[3, 4]).unwrap(); - assert_eq!( - file.write_all(&[5]).unwrap_err().kind(), - io::ErrorKind::Other - ); - assert_eq!(file_len_calls.load(Ordering::Relaxed), 0); - } -} diff --git a/crates/crab-ltx/src/internal.rs b/crates/crab-ltx/src/internal.rs deleted file mode 100644 index f77ec2a7b..000000000 --- a/crates/crab-ltx/src/internal.rs +++ /dev/null @@ -1,85 +0,0 @@ -//! Unstable inspection surface for external fuzzers, auditors, and tools. -//! -//! Nothing here carries a compatibility guarantee: it exists so a verifier can -//! exercise the same decoders production uses without depending on private -//! modules. Production callers use the typed APIs instead. The shapes follow -//! Celld's `internal` module for the same purpose. - -use crate::Result; - -/// Summary of one decoded LTX stream. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub struct InspectedLtx { - /// Page size the file was encoded with. - pub page_size: u32, - /// Database page count the commit published. - pub commit: u32, - /// First transaction id the file carries. - pub min_txid: u64, - /// Last transaction id the file carries. - pub max_txid: u64, - /// Rolling database checksum the file expects before it applies. - pub pre_apply_checksum: u64, - /// Rolling database checksum the file publishes after it applies. - pub post_apply_checksum: u64, - /// Number of page frames the file carries. - pub pages: u32, - /// Exact encoded length in bytes. - pub size_bytes: u64, - /// BLAKE3 digest of the encoded bytes. - pub blake3: [u8; 32], -} - -/// Decodes one complete LTX stream and reports what it carries. -/// -/// Every structural rule the production decoder enforces applies: checksum -/// trailer, page order, page coverage, index agreement, and the rolling -/// database checksum for snapshots. -pub fn inspect_ltx(bytes: &[u8]) -> Result { - crate::ltx::inspect_bytes(bytes) -} - -/// Decodes and shape-checks one Cell root document. -#[cfg(feature = "replica")] -pub fn inspect_root(bytes: &[u8]) -> Result { - crate::replica::root::inspect_root(bytes) -} - -/// Decodes one root descriptor page and reports its descriptor count. -#[cfg(feature = "replica")] -pub fn inspect_segment_page(bytes: &[u8]) -> Result { - crate::replica::root::inspect_segment_page(bytes) -} - -/// Shape-checks one encoded directory node. -/// -/// See [`crate::replica::directory::inspect_node`]: extent membership and the -/// lock-page rule need a root graph and are not checked here. -#[cfg(feature = "replica")] -pub fn inspect_directory_node(bytes: &[u8]) -> Result<()> { - crate::replica::directory::inspect_node(bytes) -} - -/// Decodes one checked Cell bundle and reports its row count. -#[cfg(feature = "replica")] -pub fn inspect_bundle(bytes: &[u8], limits: crate::Limits) -> Result { - let bundle = crate::bundle::Bundle::decode(bytes.to_vec(), limits)?; - Ok(bundle.rows().len()) -} - -/// Decodes one authenticated node frame and reports its body length. -#[cfg(feature = "replica")] -pub fn inspect_node_frame(bytes: &[u8], limits: crate::Limits) -> Result { - let frame = crate::inspect_node_frame(bytes.to_vec().into(), limits)?; - Ok(frame.body().len()) -} - -/// Reports the LTX file-size bound for a page count, for external admission tools. -pub fn cut_upper_bound(page_size: u32, pages: u64) -> Result { - crate::ltx::cut_upper_bound(page_size, pages) -} - -/// Reports the LTX frame bound for one page, for external admission tools. -pub fn frame_upper_bound(page_size: u32) -> Result { - crate::ltx::frame_payload_upper_bound(page_size) -} diff --git a/crates/crab-ltx/src/lib.rs b/crates/crab-ltx/src/lib.rs deleted file mode 100644 index 38e208996..000000000 --- a/crates/crab-ltx/src/lib.rs +++ /dev/null @@ -1,125 +0,0 @@ -// Contains adapted Celld lib.rs source at the revision in UPSTREAM.md. -// Apache-2.0; modified by Crab contributors. See LICENSE. - -//! Local SQLite WAL capture and exact, checksum-verified LTX recovery. -//! -//! Local capture is synchronous; use a dedicated database thread or blocking -//! executor. The optional `replica` feature adds Cell-root publication, -//! authenticated bundles, compaction, and sparse paged SQL. Leases and HTTP -//! policy remain caller-owned. - -#![deny(missing_docs)] -// A panic in a filter process or FUSE path corrupts a worktree, so production -// builds deny unwrap, expect, panic, todo, and unimplemented; test builds keep -// them available. -#![cfg_attr( - not(test), - deny( - clippy::unwrap_used, - clippy::expect_used, - clippy::panic, - clippy::todo, - clippy::unimplemented - ) -)] -#![doc = include_str!("../README.md")] - -pub mod capture; -#[cfg(feature = "replica")] -mod cell_layout; -mod codec; -mod commit; -pub mod db; -pub mod environment; -pub mod error; -#[cfg(feature = "replica")] -mod hex; -mod host; -/// Unstable inspection surface for external fuzzers, auditors, and tools. -/// -/// Nothing here carries a compatibility guarantee; production callers use the -/// typed APIs instead. -#[doc(hidden)] -pub mod internal; -#[cfg(feature = "replica")] -pub use cell_layout::{CellObjectKind, CellStorageLayout}; -#[cfg(feature = "replica")] -pub use environment::{DirectoryCacheStats, ScratchMonitor}; -pub use environment::{DiskBudget, DiskBudgetAdmission, DiskReservation, Host}; -#[cfg(feature = "replica")] -pub use environment::{ - HostResourceAdmission, HostResourceKind, HostResourcePermit, LtxPhase, LtxReadOrigin, - LtxRequestOutcome, LtxTelemetry, -}; -mod ltx; -mod lz4_block; -mod pages; -pub mod recovery; -#[cfg(feature = "replica")] -mod resume; -pub mod types; -mod wal; - -#[cfg(feature = "replica")] -pub mod bundle; -#[cfg(feature = "replica")] -mod node_frame; -#[cfg(feature = "replica")] -mod paged; -#[cfg(feature = "replica")] -mod paged_io; -#[cfg(feature = "replica")] -mod replica; -#[cfg(feature = "replica")] -pub use paged_io::with_paged_io_deadline; -#[cfg(feature = "replica")] -mod writable_vfs; -#[cfg(feature = "replica")] -pub use node_frame::{NodeFrameScope, VerifiedNodeFrame, encode_node_frame, inspect_node_frame}; -#[cfg(feature = "replica")] -pub use replica::{ - CellPagedDatabase, CellReplica, CellWritableDatabase, PreparedRoot, PublicationCost, - ReadOnlyRoot, RecoveryOverlay, RootObjectRef, RootRef, VerifiedRoot, -}; -#[cfg(feature = "replica")] -pub use writable_vfs::Hydration; - -#[cfg(all(test, feature = "replica"))] -mod format_tests; - -pub use capture::CheckpointMode; -pub use db::{Db, MANAGED_CONNECTION_PAGE_CACHE_BYTES, MANAGED_SQLITE_CONNECTIONS}; -pub use error::{CrabError, FailureClass, LimitKind, QueryError, Result, TransactionError}; -pub use recovery::{VerifiedPlan, compact_exact, restore_exact}; -pub use rusqlite; -pub use types::{CaptureBatch, CaptureTiming, Limits, LocalSegment, Position, SegmentInfo}; - -use host::{HostFile, LtxHost}; -use types::{CHECKSUM_FLAG, Checksum, Pos, Txid}; - -const META_DIR_SUFFIX: &str = "-crab-ltx"; -const CHECKPOINT_MODE_PASSIVE: &str = "PASSIVE"; -const CHECKPOINT_MODE_TRUNCATE: &str = "TRUNCATE"; -const WAL_HEADER_SIZE: usize = 32; -const WAL_FRAME_HEADER_SIZE: usize = 24; - -fn ltx_file_path(root: &str, level: u32, min: Txid, max: Txid) -> String { - format!("{root}/ltx/{level}/{min}-{max}.ltx") -} - -// Derived from Celld's lib.rs at the revision in UPSTREAM.md (Apache-2.0). -// Only validated WAL headers and pages enter this private checksum routine. -fn wal_checksum(big_endian: bool, mut s0: u32, mut s1: u32, bytes: &[u8]) -> (u32, u32) { - for chunk in bytes.as_chunks::<8>().0 { - let a = [chunk[0], chunk[1], chunk[2], chunk[3]]; - let b = [chunk[4], chunk[5], chunk[6], chunk[7]]; - let (a, b) = if big_endian { - (u32::from_be_bytes(a), u32::from_be_bytes(b)) - } else { - (u32::from_le_bytes(a), u32::from_le_bytes(b)) - }; - s0 = s0.wrapping_add(a).wrapping_add(s1); - s1 = s1.wrapping_add(b).wrapping_add(s0); - } - (s0, s1) -} diff --git a/crates/crab-ltx/src/ltx.rs b/crates/crab-ltx/src/ltx.rs deleted file mode 100644 index e01a4a842..000000000 --- a/crates/crab-ltx/src/ltx.rs +++ /dev/null @@ -1,435 +0,0 @@ -// Derived from denoland/celld, commit 10cb1303dac710dcb3b557e318e08c855261f68b. -// Apache-2.0; see LICENSE and UPSTREAM.md. Modified by Crab contributors. - -use crate::error::{CrabError, Result}; -use crate::{CHECKSUM_FLAG, Checksum, Txid}; - -pub const MAGIC: &[u8; 4] = b"LTX1"; -pub const VERSION: i32 = 3; -pub const HEADER_SIZE: usize = 100; -pub const PAGE_HEADER_SIZE: usize = 6; -pub const TRAILER_SIZE: usize = 16; -pub const CHECKSUM_SIZE: usize = 8; - -pub const HEADER_FLAG_NO_CHECKSUM: u32 = 1 << 1; -pub const HEADER_FLAG_MASK: u32 = HEADER_FLAG_NO_CHECKSUM; - -pub const PAGE_HEADER_FLAG_SIZE: u16 = 1 << 0; -pub const PAGE_HEADER_FLAG_MASK: u16 = PAGE_HEADER_FLAG_SIZE; - -pub const PENDING_BYTE: i64 = 0x4000_0000; - -fn corrupt(msg: impl Into) -> CrabError { - // Wrap a format error as LTXCorrupted, matching litestream's classification - // of malformed LTX content (litestream.go ErrLTXCorrupted). - let _ = msg; - CrabError::LTXCorrupted -} - -pub fn lock_pgno(page_size: u32) -> u32 { - if page_size == 0 { - return 0; - } - (PENDING_BYTE / page_size as i64) as u32 + 1 -} - -#[derive(Clone)] -pub struct Crc64 { - digest: crc_fast::Digest, -} - -impl Default for Crc64 { - fn default() -> Self { - Self::new() - } -} - -impl Crc64 { - pub fn new() -> Self { - Self { - digest: crc_fast::Digest::new(crc_fast::CrcAlgorithm::Crc64GoIso), - } - } - - pub fn update(&mut self, data: &[u8]) { - self.digest.update(data); - } - - pub fn sum64(&self) -> u64 { - self.digest.finalize() - } -} - -pub fn checksum_page(pgno: u32, data: &[u8]) -> Checksum { - let mut h = Crc64::new(); - h.update(&pgno.to_be_bytes()); - h.update(data); - CHECKSUM_FLAG | h.sum64() -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] -pub struct Header { - pub version: i32, - pub flags: u32, - pub page_size: u32, - pub commit: u32, - pub min_txid: Txid, - pub max_txid: Txid, - pub timestamp: i64, - pub pre_apply_checksum: Checksum, - pub wal_offset: i64, - pub wal_size: i64, - pub wal_salt1: u32, - pub wal_salt2: u32, - pub node_id: u64, -} - -impl Header { - pub fn is_snapshot(&self) -> bool { - self.min_txid == Txid(1) - } - - pub fn no_checksum(&self) -> bool { - self.flags & HEADER_FLAG_NO_CHECKSUM != 0 - } - - pub fn parse(b: &[u8]) -> Result

{ - if b.len() < HEADER_SIZE { - return Err(corrupt("short header")); - } - if &b[0..4] != MAGIC { - return Err(corrupt("bad magic")); - } - Ok(Header { - version: VERSION, - flags: u32_be(&b[4..]), - page_size: u32_be(&b[8..]), - commit: u32_be(&b[12..]), - min_txid: Txid(u64_be(&b[16..])), - max_txid: Txid(u64_be(&b[24..])), - timestamp: u64_be(&b[32..]) as i64, - pre_apply_checksum: u64_be(&b[40..]), - wal_offset: u64_be(&b[48..]) as i64, - wal_size: u64_be(&b[56..]) as i64, - wal_salt1: u32_be(&b[64..]), - wal_salt2: u32_be(&b[68..]), - node_id: u64_be(&b[72..]), - }) - } - - pub fn marshal(&self) -> [u8; HEADER_SIZE] { - let mut b = [0u8; HEADER_SIZE]; - b[0..4].copy_from_slice(MAGIC); - b[4..8].copy_from_slice(&self.flags.to_be_bytes()); - b[8..12].copy_from_slice(&self.page_size.to_be_bytes()); - b[12..16].copy_from_slice(&self.commit.to_be_bytes()); - b[16..24].copy_from_slice(&self.min_txid.0.to_be_bytes()); - b[24..32].copy_from_slice(&self.max_txid.0.to_be_bytes()); - b[32..40].copy_from_slice(&(self.timestamp as u64).to_be_bytes()); - b[40..48].copy_from_slice(&self.pre_apply_checksum.to_be_bytes()); - b[48..56].copy_from_slice(&(self.wal_offset as u64).to_be_bytes()); - b[56..64].copy_from_slice(&(self.wal_size as u64).to_be_bytes()); - b[64..68].copy_from_slice(&self.wal_salt1.to_be_bytes()); - b[68..72].copy_from_slice(&self.wal_salt2.to_be_bytes()); - b[72..80].copy_from_slice(&self.node_id.to_be_bytes()); - b - } - - pub fn validate(&self) -> Result<()> { - if self.version != VERSION { - return Err(corrupt("invalid version")); - } - if self.flags != (self.flags & HEADER_FLAG_MASK) { - return Err(corrupt("invalid flags")); - } - if !is_valid_page_size(self.page_size) { - return Err(corrupt("invalid page size")); - } - if self.min_txid == Txid(0) { - return Err(corrupt("minimum transaction id required")); - } - if self.max_txid == Txid(0) { - return Err(corrupt("maximum transaction id required")); - } - if self.min_txid > self.max_txid { - return Err(corrupt("transaction ids out of order")); - } - if self.wal_offset < 0 { - return Err(corrupt("wal offset cannot be negative")); - } - if self.wal_size < 0 { - return Err(corrupt("wal size cannot be negative")); - } - if (self.wal_salt1 != 0 || self.wal_salt2 != 0) && self.wal_offset == 0 { - return Err(corrupt("wal offset required if salt exists")); - } - if self.wal_offset == 0 && self.wal_size != 0 { - return Err(corrupt("wal offset required if wal size exists")); - } - if self.is_snapshot() { - if self.pre_apply_checksum != 0 { - return Err(corrupt("pre-apply checksum must be zero on snapshots")); - } - } else if self.no_checksum() { - if self.pre_apply_checksum != 0 { - return Err(corrupt("pre-apply checksum not allowed")); - } - } else { - if self.pre_apply_checksum == 0 { - return Err(corrupt("pre-apply checksum required on non-snapshot files")); - } - if self.pre_apply_checksum & CHECKSUM_FLAG == 0 { - return Err(corrupt("invalid pre-apply checksum format")); - } - } - Ok(()) - } -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] -pub struct PageHeader { - pub pgno: u32, - pub flags: u16, -} - -impl PageHeader { - pub fn is_zero(&self) -> bool { - self.pgno == 0 && self.flags == 0 - } - - pub fn parse(b: &[u8]) -> Result { - if b.len() < PAGE_HEADER_SIZE { - return Err(corrupt("short page header")); - } - Ok(PageHeader { - pgno: u32_be(&b[0..]), - flags: u16_be(&b[4..]), - }) - } - - pub fn marshal(&self) -> [u8; PAGE_HEADER_SIZE] { - let mut b = [0u8; PAGE_HEADER_SIZE]; - b[0..4].copy_from_slice(&self.pgno.to_be_bytes()); - b[4..6].copy_from_slice(&self.flags.to_be_bytes()); - b - } - - pub fn validate(&self) -> Result<()> { - if self.pgno == 0 { - return Err(corrupt("page number required")); - } - if self.flags != (self.flags & PAGE_HEADER_FLAG_MASK) { - return Err(corrupt("invalid page header flags")); - } - Ok(()) - } -} - -#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] -pub struct Trailer { - pub post_apply_checksum: Checksum, - pub file_checksum: Checksum, -} - -impl Trailer { - pub fn validate(&self, header: Header) -> Result<()> { - if header.no_checksum() { - if self.post_apply_checksum != 0 { - return Err(corrupt("post-apply checksum not allowed")); - } - } else if self.post_apply_checksum == 0 || self.post_apply_checksum & CHECKSUM_FLAG == 0 { - return Err(corrupt("invalid post-apply checksum")); - } - - if self.file_checksum == 0 || self.file_checksum & CHECKSUM_FLAG == 0 { - return Err(corrupt("invalid file checksum")); - } - Ok(()) - } - - pub fn parse(b: &[u8]) -> Result { - if b.len() < TRAILER_SIZE { - return Err(corrupt("short trailer")); - } - Ok(Trailer { - post_apply_checksum: u64_be(&b[0..]), - file_checksum: u64_be(&b[8..]), - }) - } - - pub fn marshal(&self) -> [u8; TRAILER_SIZE] { - let mut b = [0u8; TRAILER_SIZE]; - b[0..8].copy_from_slice(&self.post_apply_checksum.to_be_bytes()); - b[8..16].copy_from_slice(&self.file_checksum.to_be_bytes()); - b - } -} - -pub fn is_valid_page_size(sz: u32) -> bool { - let mut i = 512u32; - while i <= 65536 { - if sz == i { - return true; - } - i *= 2; - } - false -} - -/// Worst-case file-index bytes one page adds: three uvarints. -/// -/// A page number stays below five bytes and both `u64` offsets below ten, so -/// thirty bytes cover one entry and the trailing zero marker. -const INDEX_ENTRY_UPPER_BOUND: u64 = 40; - -/// Worst-case encoded bytes of one page frame. -/// -/// Covers the fixed page header, the compressed-length field, and the -/// compressor's worst-case output. An actual encoder never exceeds it because -/// it compresses every page with the same bound the codec reports. -pub(crate) fn frame_payload_upper_bound(page_size: u32) -> Result { - let compressed = crate::lz4_block::compress_bound(page_size as usize) as u64; - compressed - .checked_add((PAGE_HEADER_SIZE + 4) as u64) - .ok_or(CrabError::Limit(crate::LimitKind::LtxFileBytes)) -} - -/// Upper bound of the LTX file size that encodes `pages` pages. -/// -/// Callers use it to admit a cut before writing: a commit whose bound exceeds -/// the configured limit is refused instead of being captured and then failing. -pub(crate) fn cut_upper_bound(page_size: u32, pages: u64) -> Result { - pages - .checked_mul( - frame_payload_upper_bound(page_size)? - .checked_add(INDEX_ENTRY_UPPER_BOUND) - .ok_or(CrabError::Limit(crate::LimitKind::LtxFileBytes))?, - ) - .and_then(|bytes| bytes.checked_add((HEADER_SIZE + TRAILER_SIZE) as u64)) - .ok_or(CrabError::Limit(crate::LimitKind::LtxFileBytes)) -} - -#[derive(Debug, Clone)] -pub struct DecodedFile { - pub header: Header, - pub trailer: Trailer, -} - -type DecodedPages = Vec<(u32, Vec)>; - -pub(crate) fn inspect_reader(reader: impl std::io::Read) -> Result<(DecodedFile, u64, [u8; 32])> { - let (file, _, size, digest) = decode_reader_inner(reader, false)?; - Ok((file, size, digest)) -} - -/// Decodes one complete LTX stream and summarizes what it carries. -/// -/// Page bodies are never retained. Scratch includes one page, its compressed -/// bytes, a bounded footer buffer, and an observed entry per page. -pub(crate) fn inspect_bytes(bytes: &[u8]) -> Result { - let mut decoder = crate::codec::Decoder::new(std::io::Cursor::new(bytes)); - decoder.decode_header()?; - let page_size = decoder.header.page_size; - let mut data = vec![0; page_size as usize]; - let mut pages = 0_u32; - while decoder.decode_page(&mut data)?.is_some() { - pages = pages - .checked_add(1) - .ok_or(CrabError::Limit(crate::LimitKind::LtxPageIndexBytes))?; - } - decoder.close()?; - let (size_bytes, blake3) = decoder.artifact()?; - Ok(crate::internal::InspectedLtx { - page_size, - commit: decoder.header.commit, - min_txid: decoder.header.min_txid.0, - max_txid: decoder.header.max_txid.0, - pre_apply_checksum: decoder.header.pre_apply_checksum, - post_apply_checksum: decoder.trailer.post_apply_checksum, - pages, - size_bytes, - blake3, - }) -} - -/// Inspects a replica-sized LTX stream without retaining its body. -/// -/// The page index is small metadata (one entry per page); page bodies are -/// decoded into the caller-provided decoder scratch and are never accumulated. -#[cfg(feature = "replica")] -pub(crate) fn inspect_reader_with_index( - reader: impl std::io::Read, -) -> Result<(DecodedFile, u64, [u8; 32], Vec)> { - let mut decoder = crate::codec::Decoder::new_with_index(reader); - decoder.decode_header()?; - let mut data = vec![0; decoder.header.page_size as usize]; - while decoder.decode_page(&mut data)?.is_some() {} - decoder.close()?; - let (size, digest) = decoder.artifact()?; - Ok(( - DecodedFile { - header: decoder.header, - trailer: decoder.trailer, - }, - size, - digest, - decoder.into_replica_index()?, - )) -} - -#[cfg(feature = "replica")] -pub(crate) fn inspect_bytes_with_index( - bytes: &[u8], -) -> Result<(DecodedFile, u64, [u8; 32], Vec)> { - inspect_reader_with_index(std::io::Cursor::new(bytes)) -} - -#[cfg_attr(any(not(test), all(test, not(feature = "replica"))), expect(dead_code))] -pub(crate) fn decode_file_with_pages(bytes: &[u8]) -> Result<(DecodedFile, DecodedPages)> { - decode_file_inner(bytes, true) -} - -fn decode_file_inner(bytes: &[u8], retain_pages: bool) -> Result<(DecodedFile, DecodedPages)> { - let (file, pages, _, _) = decode_reader_inner(std::io::Cursor::new(bytes), retain_pages)?; - Ok((file, pages)) -} - -fn decode_reader_inner( - reader: impl std::io::Read, - retain_pages: bool, -) -> Result<(DecodedFile, DecodedPages, u64, [u8; 32])> { - let mut decoder = crate::codec::Decoder::new(reader); - decoder.decode_header()?; - let header = decoder.header; - let mut pages = Vec::new(); - let mut data = vec![0; header.page_size as usize]; - - while let Some(page) = decoder.decode_page(&mut data)? { - if retain_pages { - pages.push((page.pgno, data.clone())); - } - } - decoder.close()?; - - let (size, digest) = decoder.artifact()?; - - Ok(( - DecodedFile { - header, - trailer: decoder.trailer, - }, - pages, - size, - digest, - )) -} - -fn u16_be(b: &[u8]) -> u16 { - u16::from_be_bytes([b[0], b[1]]) -} -fn u32_be(b: &[u8]) -> u32 { - u32::from_be_bytes([b[0], b[1], b[2], b[3]]) -} -fn u64_be(b: &[u8]) -> u64 { - u64::from_be_bytes([b[0], b[1], b[2], b[3], b[4], b[5], b[6], b[7]]) -} diff --git a/crates/crab-ltx/src/lz4_block.rs b/crates/crab-ltx/src/lz4_block.rs deleted file mode 100644 index 0e5ef4d75..000000000 --- a/crates/crab-ltx/src/lz4_block.rs +++ /dev/null @@ -1,203 +0,0 @@ -// Derived from denoland/celld, commit 10cb1303dac710dcb3b557e318e08c855261f68b. -// Apache-2.0; see LICENSE and UPSTREAM.md. Modified by Crab contributors. - -const MIN_MATCH: usize = 4; -const WINDOW_LOG: usize = 16; -const WINDOW_SIZE: usize = 1 << WINDOW_LOG; -const WINDOW_MASK: usize = WINDOW_SIZE - 1; -const HASH_LOG: usize = 16; -const HASH_SIZE: usize = 1 << HASH_LOG; -const MATCH_FIND_LIMIT: usize = 10 + MIN_MATCH; - -pub(crate) struct Compressor { - table: Vec, - in_use: Vec, -} - -impl Default for Compressor { - fn default() -> Self { - Self { - table: vec![0; HASH_SIZE], - in_use: vec![0; HASH_SIZE / 32], - } - } -} - -impl Compressor { - pub(crate) fn compress(&mut self, src: &[u8]) -> crate::Result> { - self.in_use.fill(0); - - let mut dst = vec![0; compress_bound(src.len())]; - let mut si = 0usize; - let mut di = 0usize; - let mut anchor = 0usize; - let sn = src.len().saturating_sub(MATCH_FIND_LIMIT); - - if src.len() > MATCH_FIND_LIMIT { - while si < sn { - let matched = read_u64_le(src, si)?; - let mut hash = block_hash(matched); - let hash2 = block_hash(matched >> 8); - - let mut reference = self.get(hash, si); - let reference2 = self.get(hash2, si + 1); - self.put(hash, si); - self.put(hash2, si + 1); - - let mut offset = si as isize - reference; - if !matches_at(src, matched as u32, reference, offset) { - hash = block_hash(matched >> 16); - let reference3 = self.get(hash, si + 2); - - si += 1; - reference = reference2; - offset = si as isize - reference; - if !matches_at(src, (matched >> 8) as u32, reference, offset) { - si += 1; - reference = reference3; - offset = si as isize - reference; - self.put(hash, si); - if !matches_at(src, (matched >> 16) as u32, reference, offset) { - si += 2 + ((si - anchor) >> 7); - continue; - } - } - } - - let offset = offset as usize; - let mut literal_len = si - anchor; - let mut match_len = MIN_MATCH; - - let mut target = si as isize - offset as isize - 1; - while literal_len > 0 && target >= 0 && src[si - 1] == src[target as usize] { - si -= 1; - target -= 1; - literal_len -= 1; - match_len += 1; - } - - let match_base = si + MIN_MATCH; - si += match_len; - while si + 8 <= sn { - let diff = read_u64_le(src, si)? ^ read_u64_le(src, si - offset)?; - if diff == 0 { - si += 8; - } else { - si += diff.trailing_zeros() as usize >> 3; - break; - } - } - match_len = si - match_base; - - let token_offset = di; - dst[token_offset] = match_len.min(0x0f) as u8; - di += 1; - if literal_len < 0x0f { - dst[token_offset] |= (literal_len << 4) as u8; - } else { - dst[token_offset] |= 0xf0; - di = write_length(&mut dst, di, literal_len - 0x0f); - } - - dst[di..di + literal_len].copy_from_slice(&src[anchor..anchor + literal_len]); - di += literal_len; - dst[di..di + 2].copy_from_slice(&(offset as u16).to_le_bytes()); - di += 2; - anchor = si; - - if match_len >= 0x0f { - di = write_length(&mut dst, di, match_len - 0x0f); - } - if si >= sn { - break; - } - - hash = block_hash(read_u64_le(src, si - 2)?); - self.put(hash, si - 2); - } - } - - let mut literal_len = src.len() - anchor; - if literal_len < 0x0f { - dst[di] = (literal_len << 4) as u8; - } else { - dst[di] = 0xf0; - di += 1; - while literal_len >= 0xff + 0x0f { - dst[di] = 0xff; - di += 1; - literal_len -= 0xff; - } - dst[di] = (literal_len - 0x0f) as u8; - } - di += 1; - di += copy_into(&mut dst[di..], &src[anchor..]); - dst.truncate(di); - Ok(dst) - } - - fn get(&self, hash: u32, si: usize) -> isize { - let hash = hash as usize & (HASH_SIZE - 1); - let mut pos = 0isize; - if self.in_use[hash / 32] & (1 << (hash % 32)) != 0 { - pos = self.table[hash] as isize; - } - pos += (si & !WINDOW_MASK) as isize; - if pos >= si as isize { - pos -= WINDOW_SIZE as isize; - } - pos - } - - fn put(&mut self, hash: u32, si: usize) { - let hash = hash as usize & (HASH_SIZE - 1); - self.table[hash] = si as u16; - self.in_use[hash / 32] |= 1 << (hash % 32); - } -} - -pub(crate) fn compress_bound(n: usize) -> usize { - n + n / 255 + 16 -} - -fn block_hash(value: u64) -> u32 { - const PRIME_6_BYTES: u64 = 227_718_039_650_203; - ((value << 16).wrapping_mul(PRIME_6_BYTES) >> (64 - HASH_LOG)) as u32 -} - -fn read_u64_le(src: &[u8], offset: usize) -> crate::Result { - let bytes = src - .get(offset..offset + 8) - .ok_or(crate::CrabError::LTXCorrupted)?; - Ok(u64::from_le_bytes( - bytes - .try_into() - .map_err(|_| crate::CrabError::LTXCorrupted)?, - )) -} - -fn matches_at(src: &[u8], value: u32, reference: isize, offset: isize) -> bool { - offset > 0 - && offset < WINDOW_SIZE as isize - && reference >= 0 - && src - .get(reference as usize..reference as usize + 4) - .is_some_and(|bytes| { - value == u32::from_le_bytes([bytes[0], bytes[1], bytes[2], bytes[3]]) - }) -} - -fn write_length(dst: &mut [u8], mut offset: usize, mut len: usize) -> usize { - while len >= 0xff { - dst[offset] = 0xff; - offset += 1; - len -= 0xff; - } - dst[offset] = len as u8; - offset + 1 -} - -fn copy_into(dst: &mut [u8], src: &[u8]) -> usize { - dst[..src.len()].copy_from_slice(src); - src.len() -} diff --git a/crates/crab-ltx/src/node_frame.rs b/crates/crab-ltx/src/node_frame.rs deleted file mode 100644 index 01007beaa..000000000 --- a/crates/crab-ltx/src/node_frame.rs +++ /dev/null @@ -1,228 +0,0 @@ -//! Verified binary envelopes for multiplexed follower replication. - -use bytes::Bytes; - -use crate::{CrabError, Limits, Result, SegmentInfo}; - -const MAGIC: &[u8; 4] = b"CNL1"; -const VERSION: u16 = 1; -const HEADER_BYTES: usize = 240; - -/// Immutable routing and ordering fields authenticated by a node-log frame. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct NodeFrameScope { - /// Session of the leader that published the frame. - pub leader_session: [u8; 16], - /// Node-log epoch the frame was captured under. - pub log_epoch: u64, - /// Position of the frame in the node log. - pub node_sequence: u64, - /// Application the frame belongs to. - pub application: [u8; 16], - /// Cell the frame belongs to. - pub cell: [u8; 32], - /// Incarnation that captured the frame. - pub incarnation: [u8; 16], - /// Cell epoch the segment was captured under. - pub cell_epoch: u64, - /// Root commit sequence the segment publishes. - pub commit_sequence: u64, -} - -/// One canonical node-log frame whose complete LTX body has been verified. -/// -/// Construction is restricted to [`encode_node_frame`] and -/// [`inspect_node_frame`], so callers cannot attach trusted metadata to an -/// unchecked body. -#[derive(Clone, Debug)] -pub struct VerifiedNodeFrame { - scope: NodeFrameScope, - segment: SegmentInfo, - body: Bytes, - encoded: Bytes, -} - -impl VerifiedNodeFrame { - /// Returns the routing fields the frame was authenticated with. - #[must_use] - pub const fn scope(&self) -> NodeFrameScope { - self.scope - } - - /// Returns the manifest expectations of the embedded segment. - #[must_use] - pub const fn segment(&self) -> &SegmentInfo { - &self.segment - } - - /// Returns the verified LTX body. - #[must_use] - pub fn body(&self) -> &Bytes { - &self.body - } - - /// Returns the encoded frame bytes as received. - #[must_use] - pub fn encoded(&self) -> &Bytes { - &self.encoded - } - - /// Returns the digest followers use to distinguish an exact retry from a - /// conflicting duplicate sequence. - #[must_use] - pub fn digest(&self) -> [u8; 32] { - *blake3::hash(&self.encoded).as_bytes() - } -} - -/// Verifies an LTX body and encodes its canonical node-log frame. -pub fn encode_node_frame( - scope: NodeFrameScope, - segment: SegmentInfo, - body: Bytes, - limits: Limits, -) -> Result { - validate_scope(scope)?; - validate_body(&body, &segment, limits)?; - - let capacity = HEADER_BYTES - .checked_add(body.len()) - .ok_or(CrabError::Limit(crate::LimitKind::NodeFrameBytes))?; - let mut encoded = Vec::with_capacity(capacity); - encoded.extend_from_slice(MAGIC); - encoded.extend_from_slice(&VERSION.to_le_bytes()); - encoded.extend_from_slice(&(HEADER_BYTES as u16).to_le_bytes()); - encoded.extend_from_slice(&scope.leader_session); - push_u64(&mut encoded, scope.log_epoch); - push_u64(&mut encoded, scope.node_sequence); - encoded.extend_from_slice(&scope.application); - encoded.extend_from_slice(&scope.cell); - encoded.extend_from_slice(&scope.incarnation); - push_u64(&mut encoded, scope.cell_epoch); - push_u64(&mut encoded, scope.commit_sequence); - push_segment(&mut encoded, &segment); - push_u64(&mut encoded, body.len() as u64); - encoded.extend_from_slice(blake3::hash(&body).as_bytes()); - debug_assert_eq!(encoded.len(), HEADER_BYTES); - encoded.extend_from_slice(&body); - inspect_node_frame(Bytes::from(encoded), limits) -} - -/// Decodes a canonical frame and verifies all declared LTX metadata. -pub fn inspect_node_frame(encoded: Bytes, limits: Limits) -> Result { - if encoded.len() < HEADER_BYTES || &encoded[..4] != MAGIC { - return Err(CrabError::LTXCorrupted); - } - let mut cursor = 4; - if take_u16(&encoded, &mut cursor)? != VERSION - || usize::from(take_u16(&encoded, &mut cursor)?) != HEADER_BYTES - { - return Err(CrabError::LTXCorrupted); - } - let scope = NodeFrameScope { - leader_session: take_array(&encoded, &mut cursor)?, - log_epoch: take_u64(&encoded, &mut cursor)?, - node_sequence: take_u64(&encoded, &mut cursor)?, - application: take_array(&encoded, &mut cursor)?, - cell: take_array(&encoded, &mut cursor)?, - incarnation: take_array(&encoded, &mut cursor)?, - cell_epoch: take_u64(&encoded, &mut cursor)?, - commit_sequence: take_u64(&encoded, &mut cursor)?, - }; - let segment = take_segment(&encoded, &mut cursor)?; - let body_len = take_u64(&encoded, &mut cursor)?; - let body_digest: [u8; 32] = take_array(&encoded, &mut cursor)?; - if cursor != HEADER_BYTES - || body_len != segment.size_bytes - || body_len > limits.max_capture_bytes - || body_len > usize::MAX as u64 - || encoded.len() != HEADER_BYTES.saturating_add(body_len as usize) - { - return Err(CrabError::Limit(crate::LimitKind::NodeFrameBytes)); - } - validate_scope(scope)?; - let body = encoded.slice(HEADER_BYTES..); - if body_digest != segment.blake3 || body_digest != *blake3::hash(&body).as_bytes() { - return Err(CrabError::ChecksumMismatch); - } - validate_body(&body, &segment, limits)?; - Ok(VerifiedNodeFrame { - scope, - segment, - body, - encoded, - }) -} - -fn validate_scope(scope: NodeFrameScope) -> Result<()> { - if scope.leader_session.iter().all(|byte| *byte == 0) - || scope.application.iter().all(|byte| *byte == 0) - || scope.cell.iter().all(|byte| *byte == 0) - || scope.incarnation.iter().all(|byte| *byte == 0) - || scope.log_epoch == 0 - || scope.node_sequence == 0 - || scope.cell_epoch == 0 - || scope.commit_sequence == 0 - { - return Err(CrabError::InvalidState("invalid node frame scope")); - } - Ok(()) -} - -fn validate_body(body: &[u8], segment: &SegmentInfo, limits: Limits) -> Result<()> { - if body.len() as u64 > limits.max_capture_bytes { - return Err(CrabError::Limit(crate::LimitKind::NodeFrameBody)); - } - crate::recovery::verify_segment(body, segment, limits) -} - -fn push_u64(bytes: &mut Vec, value: u64) { - bytes.extend_from_slice(&value.to_le_bytes()); -} - -fn push_segment(bytes: &mut Vec, segment: &SegmentInfo) { - push_u64(bytes, segment.min_txid); - push_u64(bytes, segment.max_txid); - bytes.extend_from_slice(&segment.page_size.to_le_bytes()); - bytes.extend_from_slice(&segment.database_pages.to_le_bytes()); - push_u64(bytes, segment.pre_checksum); - push_u64(bytes, segment.post_checksum); - push_u64(bytes, segment.size_bytes); - bytes.extend_from_slice(&segment.blake3); -} - -fn take_segment(bytes: &[u8], cursor: &mut usize) -> Result { - Ok(SegmentInfo { - min_txid: take_u64(bytes, cursor)?, - max_txid: take_u64(bytes, cursor)?, - page_size: take_u32(bytes, cursor)?, - database_pages: take_u32(bytes, cursor)?, - pre_checksum: take_u64(bytes, cursor)?, - post_checksum: take_u64(bytes, cursor)?, - size_bytes: take_u64(bytes, cursor)?, - blake3: take_array(bytes, cursor)?, - }) -} - -fn take_u16(bytes: &[u8], cursor: &mut usize) -> Result { - Ok(u16::from_le_bytes(take_array(bytes, cursor)?)) -} - -fn take_u32(bytes: &[u8], cursor: &mut usize) -> Result { - Ok(u32::from_le_bytes(take_array(bytes, cursor)?)) -} - -fn take_u64(bytes: &[u8], cursor: &mut usize) -> Result { - Ok(u64::from_le_bytes(take_array(bytes, cursor)?)) -} - -fn take_array(bytes: &[u8], cursor: &mut usize) -> Result<[u8; N]> { - let end = cursor.checked_add(N).ok_or(CrabError::LTXCorrupted)?; - let value = bytes - .get(*cursor..end) - .ok_or(CrabError::LTXCorrupted)? - .try_into() - .map_err(|_| CrabError::LTXCorrupted)?; - *cursor = end; - Ok(value) -} diff --git a/crates/crab-ltx/src/paged.rs b/crates/crab-ltx/src/paged.rs deleted file mode 100644 index 72d9ab0ce..000000000 --- a/crates/crab-ltx/src/paged.rs +++ /dev/null @@ -1,182 +0,0 @@ -//! Celld-inspired pinned page maps; authenticated indexes replace unchecked tails. - -use crate::{CrabError, Result}; - -pub(crate) const ENTRY_BYTES: usize = 60; -const FRAME_PREFIX: usize = crate::ltx::PAGE_HEADER_SIZE + 4; - -/// Worst-case bytes of one replica-indexed page frame. -/// -/// Shares the local cut bound's frame math so a page admitted by one path can -/// never be rejected by the other. -pub(crate) fn maximum_frame_bytes(page_size: u32) -> Result { - u32::try_from(crate::ltx::frame_payload_upper_bound(page_size)?) - .map_err(|_| CrabError::LTXCorrupted) -} - -pub(crate) struct IndexEntry { - pub page: u32, - pub offset: u64, - pub size: u64, - pub hash: [u8; 32], - pub checksum: u64, -} - -struct IndexEntries<'a> { - chunks: std::slice::Iter<'a, [u8; ENTRY_BYTES]>, -} - -impl Iterator for IndexEntries<'_> { - type Item = Result; - - fn next(&mut self) -> Option { - self.chunks.next().map(|entry| decode_index_entry(entry)) - } -} - -pub(crate) fn decode_index_entry(entry: &[u8]) -> Result { - Ok(IndexEntry { - page: u32::from_be_bytes(array(entry.get(..4).ok_or(CrabError::LTXCorrupted)?)?), - offset: u64::from_be_bytes(array(entry.get(4..12).ok_or(CrabError::LTXCorrupted)?)?), - size: u64::from_be_bytes(array(entry.get(12..20).ok_or(CrabError::LTXCorrupted)?)?), - hash: array(entry.get(20..52).ok_or(CrabError::LTXCorrupted)?)?, - checksum: u64::from_be_bytes(array(entry.get(52..60).ok_or(CrabError::LTXCorrupted)?)?), - }) -} - -pub(crate) struct ValidatedIndexEntries<'a> { - entries: IndexEntries<'a>, - validator: IndexValidator, -} - -pub(crate) struct IndexValidator { - info: crate::SegmentInfo, - previous_page: u32, - previous_end: u64, -} - -impl IndexValidator { - pub(crate) fn new(info: &crate::SegmentInfo) -> Self { - Self { - info: info.clone(), - previous_page: 0, - previous_end: crate::ltx::HEADER_SIZE as u64, - } - } - - pub(crate) fn validate(&mut self, entry: IndexEntry) -> Result { - let lock = crate::ltx::lock_pgno(self.info.page_size); - let max_frame = u64::from(maximum_frame_bytes(self.info.page_size)?); - let end = entry - .offset - .checked_add(entry.size) - .ok_or(CrabError::LTXCorrupted)?; - let footer = crate::ltx::PAGE_HEADER_SIZE + 8 + crate::ltx::TRAILER_SIZE + 1; - if entry.page <= self.previous_page - || entry.page > self.info.database_pages - || entry.page == lock - || entry.offset != self.previous_end - || !(FRAME_PREFIX as u64..=max_frame).contains(&entry.size) - || end > self.info.size_bytes.saturating_sub(footer as u64) - { - return Err(CrabError::LTXCorrupted); - } - self.previous_page = entry.page; - self.previous_end = end; - Ok(entry) - } -} - -impl Iterator for ValidatedIndexEntries<'_> { - type Item = Result; - - fn next(&mut self) -> Option { - self.entries - .next() - .map(|entry| self.validator.validate(entry?)) - } -} - -pub(crate) fn decode_index(bytes: &[u8]) -> Result> { - index_entries(bytes)?.collect() -} - -fn index_entries(bytes: &[u8]) -> Result> { - let (entries, remainder) = bytes.as_chunks::(); - if !remainder.is_empty() { - return Err(CrabError::LTXCorrupted); - } - Ok(IndexEntries { - chunks: entries.iter(), - }) -} - -pub(crate) fn validated_index_entries<'a>( - bytes: &'a [u8], - info: &'a crate::SegmentInfo, -) -> Result> { - Ok(ValidatedIndexEntries { - entries: index_entries(bytes)?, - validator: IndexValidator::new(info), - }) -} - -/// Encodes the fixed-size authenticated index collected by the streaming LTX -/// decoder. The decoder has already verified the embedded variable-length -/// index and trailer; this pass only changes its representation for paged SQL. -pub(crate) fn encode_index_from_pages(pages: &[crate::codec::EncodedPage]) -> Result> { - if pages.is_empty() { - return Err(CrabError::LTXCorrupted); - } - let mut index = Vec::with_capacity( - pages - .len() - .checked_mul(ENTRY_BYTES) - .ok_or(CrabError::Limit(crate::LimitKind::LtxPageIndexBytes))?, - ); - for page in pages { - append_index_page(&mut index, page)?; - } - Ok(index) -} - -pub(crate) fn append_index_page( - index: &mut Vec, - page: &crate::codec::EncodedPage, -) -> Result<()> { - if page.frame_hash == [0; 32] || page.size == 0 { - return Err(CrabError::LTXCorrupted); - } - index.extend_from_slice(&page.page.to_be_bytes()); - index.extend_from_slice(&page.offset.to_be_bytes()); - index.extend_from_slice(&page.size.to_be_bytes()); - index.extend_from_slice(&page.frame_hash); - index.extend_from_slice(&page.checksum.to_be_bytes()); - Ok(()) -} - -fn array(bytes: &[u8]) -> Result<[u8; N]> { - bytes.try_into().map_err(|_| CrabError::LTXCorrupted) -} - -pub(crate) fn decode_frame(frame: &[u8], page_size: u32, pgno: u32) -> Result> { - let prefix = frame.get(..FRAME_PREFIX).ok_or(CrabError::LTXCorrupted)?; - let header = crate::ltx::PageHeader::parse(&prefix[..crate::ltx::PAGE_HEADER_SIZE])?; - header.validate()?; - let size = - u32::from_be_bytes(array(&prefix[crate::ltx::PAGE_HEADER_SIZE..FRAME_PREFIX])?) as usize; - if header.pgno != pgno - || header.flags != crate::ltx::PAGE_HEADER_FLAG_SIZE - || size != frame.len() - FRAME_PREFIX - || size > crate::lz4_block::compress_bound(page_size as usize) - { - return Err(CrabError::LTXCorrupted); - } - let mut data = vec![0; page_size as usize]; - let n = lz4_flex::block::decompress_into(&frame[FRAME_PREFIX..], &mut data) - .map_err(|e| CrabError::Other(Box::new(e)))?; - if n != data.len() { - return Err(CrabError::LTXCorrupted); - } - Ok(data) -} diff --git a/crates/crab-ltx/src/paged_io.rs b/crates/crab-ltx/src/paged_io.rs deleted file mode 100644 index d3bd37b39..000000000 --- a/crates/crab-ltx/src/paged_io.rs +++ /dev/null @@ -1,454 +0,0 @@ -//! Shared, bounded bridge between blocking SQLite calls and asynchronous stores. - -use crate::{CellWritableDatabase, CrabError, Result}; -use std::{ - cell::Cell, - collections::{HashMap, VecDeque}, - sync::{Arc, Mutex, Weak, mpsc}, - time::{Duration, Instant}, -}; - -pub(crate) type Pages = Vec<(u32, Vec)>; -pub(crate) type DriverSlot = Arc>>; -const DEFAULT_DEADLINE: Duration = Duration::from_secs(30); - -thread_local! { - static DEADLINE: Cell> = const { Cell::new(None) }; - static ORIGIN: Cell = const { Cell::new(crate::LtxReadOrigin::Sparse) }; -} - -struct DeadlineGuard(Option); - -impl Drop for DeadlineGuard { - fn drop(&mut self) { - DEADLINE.set(self.0); - } -} - -struct OriginGuard(crate::LtxReadOrigin); - -impl Drop for OriginGuard { - fn drop(&mut self) { - ORIGIN.set(self.0); - } -} - -/// Applies one absolute deadline to sparse page faults on the current thread. -/// -/// Call this around SQLite work on its owning thread. Nested scopes retain the -/// earliest deadline. The deadline stops the synchronous VFS wait but does not -/// cancel a provider request that has already been accepted by the I/O driver. -pub fn with_paged_io_deadline(deadline: Instant, operation: impl FnOnce() -> T) -> T { - let previous = DEADLINE.get(); - DEADLINE.set(Some( - previous.map_or(deadline, |current| current.min(deadline)), - )); - let _guard = DeadlineGuard(previous); - operation() -} - -pub(crate) fn with_paged_io_origin( - origin: crate::LtxReadOrigin, - operation: impl FnOnce() -> T, -) -> T { - let previous = ORIGIN.replace(origin); - let _guard = OriginGuard(previous); - operation() -} - -fn deadline() -> Instant { - DEADLINE - .get() - .unwrap_or_else(|| Instant::now() + DEFAULT_DEADLINE) -} - -#[derive(Clone)] -pub(crate) enum Database { - Cell(CellWritableDatabase), - Snapshot(crate::CellPagedDatabase), -} - -impl Database { - pub(crate) fn read_only(&self) -> bool { - matches!(self, Self::Snapshot(_)) - } - - pub(crate) fn host(&self) -> crate::Host { - match self { - Self::Cell(database) => database.host(), - Self::Snapshot(database) => database.host(), - } - } - - pub(crate) fn limits(&self) -> crate::Limits { - match self { - Self::Cell(database) => database.limits(), - Self::Snapshot(database) => database.limits(), - } - } - - pub(crate) fn page_size(&self) -> u32 { - match self { - Self::Cell(database) => database.page_size(), - Self::Snapshot(database) => database.page_size(), - } - } - - pub(crate) fn page_count(&self) -> u32 { - match self { - Self::Cell(database) => database.page_count(), - Self::Snapshot(database) => database.page_count(), - } - } - - pub(crate) fn position(&self) -> crate::Position { - match self { - Self::Cell(database) => database.position(), - Self::Snapshot(database) => database.position(), - } - } - - pub(crate) fn checksums(&self) -> Result { - match self { - Self::Cell(database) => Ok(database.checksums()), - Self::Snapshot(_) => Err(CrabError::InvalidState("snapshot has no capture index")), - } - } - - async fn read_run( - &self, - first: u32, - max_pages: u32, - origin: crate::LtxReadOrigin, - ) -> Result { - match self { - Self::Cell(database) => database.read_run(first, max_pages, origin).await, - Self::Snapshot(database) => database.read_run(first, max_pages, origin).await, - } - } -} - -struct Request { - database: Database, - view: u64, - page: u32, - origin: crate::LtxReadOrigin, - deadline: Instant, - reply: mpsc::SyncSender>>, -} - -#[derive(Default)] -struct Cache { - pages: HashMap<(u64, u32), Vec>, - order: VecDeque<(u64, u32)>, - bytes: usize, -} - -impl Cache { - fn missing_prefix(&self, view: u64, first: u32, count: u32) -> u32 { - // The first page is missing. A later page may already be prefetched; - // stop before it so demand reads and hydration do not fetch it twice. - (1..count) - .find(|offset| { - first - .checked_add(*offset) - .is_some_and(|page| self.pages.contains_key(&(view, page))) - }) - .unwrap_or(count) - } - - fn insert(&mut self, view: u64, pages: Pages) { - for (page, bytes) in pages { - let key = (view, page); - if self.pages.contains_key(&key) { - continue; - } - while self.bytes + bytes.len() > 8 << 20 { - let Some(old) = self.order.pop_front() else { - break; - }; - if let Some(bytes) = self.pages.remove(&old) { - self.bytes -= bytes.len(); - } - } - self.bytes += bytes.len(); - self.pages.insert(key, bytes); - self.order.push_back(key); - } - } -} - -pub(crate) struct Driver { - sender: Option>, - cache: Arc>, - worker: Option>, -} - -impl Driver { - fn new(host: &crate::Host) -> Result> { - let (sender, mut receiver) = tokio::sync::mpsc::channel::(256); - let (ready, started) = mpsc::sync_channel(1); - let cache = Arc::new(Mutex::new(Cache::default())); - let shared_cache = cache.clone(); - let worker = host.executor.start_worker(Box::new(move || { - let runtime = match tokio::runtime::Builder::new_current_thread() - .enable_all() - .build() - { - Ok(runtime) => runtime, - Err(error) => { - let _ = ready.send(Err(error)); - return; - } - }; - if ready.send(Ok(())).is_err() { - return; - } - // The runtime must keep polling while SQL is idle: provider clients - // retain pooled connections whose drivers were spawned here. - runtime.block_on(async move { - let mut jobs = tokio::task::JoinSet::new(); - let mut closed = false; - while !closed || !jobs.is_empty() { - tokio::select! { - request = receiver.recv(), if !closed && jobs.len() < 32 => { - match request { - Some(request) => { - let cache = shared_cache.clone(); - jobs.spawn(async move { - let result = fetch(&request, &cache).await; - let _ = request.reply.send(result); - }); - } - None => closed = true, - } - } - _ = jobs.join_next(), if !jobs.is_empty() => {} - } - } - }); - }))?; - let driver = Arc::new(Self { - sender: Some(sender), - cache, - worker: Some(worker), - }); - started - .recv() - .map_err(|e| CrabError::Other(Box::new(e)))??; - Ok(driver) - } -} - -async fn fetch(request: &Request, cache: &Mutex) -> Result> { - let deadline = tokio::time::Instant::from_std(request.deadline); - let count = cache - .lock() - .map_err(|_| CrabError::InvalidState("paged cache poisoned"))? - .missing_prefix(request.view, request.page, 64); - let pages = tokio::time::timeout_at( - deadline, - request - .database - .read_run(request.page, count, request.origin), - ) - .await - .map_err(|_| CrabError::Deadline)??; - let mut cache = cache - .lock() - .map_err(|_| CrabError::InvalidState("paged cache poisoned"))?; - cache.insert(request.view, pages); - cache - .pages - .get(&(request.view, request.page)) - .cloned() - .ok_or(CrabError::LTXCorrupted) -} - -impl Drop for Driver { - fn drop(&mut self) { - self.sender.take(); - if let Some(worker) = self.worker.take() { - let _ = worker.join(); - } - } -} - -pub(crate) struct Io { - driver: Arc, - database: Database, - view: u64, - gate: Mutex<()>, -} - -impl Io { - pub(crate) async fn hydration_pages(&self, first: u32, count: u32) -> Result { - self.database - .host() - .observe_ltx_logical_read(crate::LtxReadOrigin::Hydrating); - let missing = { - let cache = self - .driver - .cache - .lock() - .map_err(|_| CrabError::InvalidState("paged cache poisoned"))?; - let mut cached = Vec::new(); - for offset in 0..count { - let page = first.checked_add(offset).ok_or(CrabError::LTXCorrupted)?; - let Some(bytes) = cache.pages.get(&(self.view, page)) else { - break; - }; - cached.push((page, bytes.clone())); - } - if !cached.is_empty() { - return Ok(cached); - } - cache.missing_prefix(self.view, first, count) - }; - self.database - .read_run(first, missing, crate::LtxReadOrigin::Hydrating) - .await - } - - pub(crate) fn new(database: Database) -> Result { - let host = database.host(); - let mut slot = host - .paged_driver - .lock() - .map_err(|_| CrabError::InvalidState("paged driver slot poisoned"))?; - let driver = match slot.upgrade() { - Some(driver) => driver, - None => { - let driver = Driver::new(&host)?; - *slot = Arc::downgrade(&driver); - driver - } - }; - static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); - Ok(Self { - driver, - database, - view: NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed), - gate: Mutex::new(()), - }) - } - - pub(crate) fn page(&self, page: u32) -> Result> { - let origin = ORIGIN.get(); - self.database.host().observe_ltx_logical_read(origin); - let _gate = self - .gate - .lock() - .map_err(|_| CrabError::InvalidState("paged request gate poisoned"))?; - if let Some(bytes) = self - .driver - .cache - .lock() - .map_err(|_| CrabError::InvalidState("paged cache poisoned"))? - .pages - .get(&(self.view, page)) - .cloned() - { - return Ok(bytes); - } - let (reply, response) = mpsc::sync_channel(1); - let deadline = deadline(); - let request = Request { - database: self.database.clone(), - view: self.view, - page, - origin, - deadline, - reply, - }; - self.driver - .sender - .as_ref() - .ok_or(CrabError::InvalidState("paged I/O closed"))? - .try_send(request) - .map_err(|error| match error { - tokio::sync::mpsc::error::TrySendError::Full(_) => { - CrabError::Limit(crate::LimitKind::PagedRequestQueue) - } - tokio::sync::mpsc::error::TrySendError::Closed(_) => { - CrabError::InvalidState("paged I/O closed") - } - })?; - receive(response, deadline) - } -} - -fn receive(response: mpsc::Receiver>>, deadline: Instant) -> Result> { - match response.recv_timeout(deadline.saturating_duration_since(Instant::now())) { - Ok(result) => result, - Err(mpsc::RecvTimeoutError::Timeout) => Err(CrabError::Deadline), - Err(error @ mpsc::RecvTimeoutError::Disconnected) => Err(CrabError::Other(Box::new(error))), - } -} - -pub(crate) fn default_slot() -> DriverSlot { - static SLOT: std::sync::OnceLock = std::sync::OnceLock::new(); - SLOT.get_or_init(|| Arc::new(Mutex::new(Weak::new()))) - .clone() -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn scoped_deadline_keeps_earliest_value_and_restores_the_thread() { - let outer = Instant::now() + Duration::from_secs(2); - let later = outer + Duration::from_secs(1); - let earlier = outer - Duration::from_secs(1); - with_paged_io_deadline(outer, || { - assert_eq!(deadline(), outer); - with_paged_io_deadline(later, || assert_eq!(deadline(), outer)); - with_paged_io_deadline(earlier, || assert_eq!(deadline(), earlier)); - assert_eq!(deadline(), outer); - }); - assert!(deadline() >= Instant::now() + Duration::from_secs(29)); - } - - #[test] - fn scoped_origin_restores_sparse_default() { - assert_eq!(ORIGIN.get(), crate::LtxReadOrigin::Sparse); - with_paged_io_origin(crate::LtxReadOrigin::Hydrating, || { - assert_eq!(ORIGIN.get(), crate::LtxReadOrigin::Hydrating); - }); - assert_eq!(ORIGIN.get(), crate::LtxReadOrigin::Sparse); - } - - #[test] - fn blocking_page_wait_stops_at_its_deadline() { - let (reply, response) = mpsc::sync_channel(1); - let started = Instant::now(); - let result = receive(response, started + Duration::from_millis(20)); - drop(reply); - assert!(matches!(result, Err(CrabError::Deadline))); - assert!(started.elapsed() < Duration::from_secs(1)); - } - - #[test] - fn cache_bounds_payload_and_isolates_pinned_views() { - let mut cache = Cache::default(); - for view in 0..16 { - cache.insert(view, vec![(1, vec![view as u8; 1 << 20])]); - } - assert_eq!(cache.bytes, 8 << 20); - assert_eq!(cache.pages.len(), 8); - assert_eq!(cache.order.len(), 8); - for view in 0..16 { - match cache.pages.get(&(view, 1)) { - Some(bytes) => { - assert!(view >= 8); - assert!(bytes.iter().all(|byte| *byte == view as u8)); - } - None => assert!(view < 8), - } - } - cache.insert(15, vec![(1, vec![99; 1 << 20])]); - assert_eq!(cache.bytes, 8 << 20); - assert_eq!(cache.pages[&(15, 1)][0], 15); - } -} diff --git a/crates/crab-ltx/src/pages.rs b/crates/crab-ltx/src/pages.rs deleted file mode 100644 index b96136271..000000000 --- a/crates/crab-ltx/src/pages.rs +++ /dev/null @@ -1,732 +0,0 @@ -#[cfg(feature = "replica")] -use std::io::{BufReader, Read as _}; -use std::{collections::HashMap, sync::Arc}; - -use crate::{CHECKSUM_FLAG, CrabError, Result, ltx}; - -#[cfg(feature = "replica")] -const CHECKSUM_READ_BYTES: usize = 64 * 1024; -/// Buffered page-checksum writes keep a dense copy off the syscall path. -#[cfg(feature = "replica")] -const DENSE_WRITE_BYTES: usize = 64 * 1024; -const CHECKSUM_BLOCK_BYTES: usize = 4096; - -#[derive(Clone)] -enum ChecksumBase { - Memory(Arc>), - #[cfg(feature = "replica")] - File(Arc), -} - -#[cfg(feature = "replica")] -#[derive(Clone)] -struct FileChecksumBase { - host: crate::Host, - path: std::path::PathBuf, - limit: u64, -} - -#[cfg(feature = "replica")] -impl FileChecksumBase { - fn open(&self) -> Result { - crate::LtxHost { - facilities: self.host.clone(), - max_database_bytes: self.limit, - max_file_bytes: self.limit, - } - .open_rw(&self.path) - .map_err(Into::into) - } -} - -/// Transactional page-checksum view used by WAL capture. -/// -/// Clones retain an immutable base and copy only the uncommitted overlay. Cell -/// activations use a local fixed-width file as that base, keeping resident -/// memory proportional to pages changed by the current cut. -#[derive(Clone)] -pub(crate) struct PageChecksums { - base: ChecksumBase, - base_count: u32, - count: u32, - changes: HashMap, - checksum: u64, -} - -pub(crate) struct PageChecksumApply<'a> { - target: &'a mut PageChecksums, - page_size: u32, - commit: u32, - previous_count: u32, - previous: u32, - next_required: u32, - base_file: Option, -} - -struct ChecksumReader { - file: crate::HostFile, - first: u32, - bytes: Vec, -} - -impl ChecksumReader { - fn value(&mut self, page: u32, count: u32) -> Result { - // Capture supplies increasing pages. One local window amortizes nearby - // overwrites without retaining a database-sized checksum cache. - let per_block = (CHECKSUM_BLOCK_BYTES / 8) as u32; - let first = (page - 1) / per_block * per_block; - if self.bytes.is_empty() || self.first != first { - let length = (count - first).min(per_block) as usize * 8; - self.bytes = self.file.read_exact_at(u64::from(first) * 8, length)?; - self.first = first; - } - let offset = (page - 1 - self.first) as usize * 8; - Ok(u64::from_be_bytes( - self.bytes[offset..offset + 8] - .try_into() - .map_err(|_| CrabError::LTXCorrupted)?, - )) - } -} - -impl Default for PageChecksums { - fn default() -> Self { - Self { - base: ChecksumBase::Memory(Arc::new(Vec::new())), - base_count: 0, - count: 0, - changes: HashMap::new(), - checksum: CHECKSUM_FLAG, - } - } -} - -impl PageChecksums { - #[cfg(feature = "replica")] - pub(crate) fn from_file( - host: crate::LtxHost, - path: &std::path::Path, - page_size: u32, - count: u32, - checksum: u64, - ) -> Result { - if !ltx::is_valid_page_size(page_size) || count == 0 || checksum & CHECKSUM_FLAG == 0 { - return Err(CrabError::LTXCorrupted); - } - if host.metadata(path)?.len != u64::from(count) * 8 { - return Err(CrabError::LTXCorrupted); - } - Ok(Self { - base: ChecksumBase::File(Arc::new(FileChecksumBase { - host: host.facilities, - path: path.to_owned(), - limit: host.max_file_bytes, - })), - base_count: count, - count, - changes: HashMap::new(), - checksum, - }) - } - - #[cfg_attr(not(test), expect(dead_code))] - pub fn apply( - &mut self, - page_size: u32, - commit: u32, - pages: &[(u32, Vec)], - limit: u64, - ) -> Result<()> { - self.apply_iter(page_size, commit, pages.iter().cloned().map(Ok), limit) - } - - pub(crate) fn apply_iter( - &mut self, - page_size: u32, - commit: u32, - pages: impl Iterator)>>, - limit: u64, - ) -> Result<()> { - let mut apply = self.begin_apply(page_size, commit, limit)?; - for page in pages { - let (pgno, data) = page?; - apply.page(pgno, &data)?; - } - apply.finish() - } - - pub(crate) fn begin_apply( - &mut self, - page_size: u32, - commit: u32, - limit: u64, - ) -> Result> { - if !ltx::is_valid_page_size(page_size) || commit == 0 { - return Err(CrabError::LTXCorrupted); - } - if u64::from(page_size) * u64::from(commit) > limit { - return Err(CrabError::Limit(crate::LimitKind::DatabaseBytes)); - } - - let previous_count = self.count; - let mut base_file = { - #[cfg(feature = "replica")] - { - match &self.base { - ChecksumBase::File(base) => Some(ChecksumReader { - file: base.open()?, - first: 0, - bytes: Vec::new(), - }), - ChecksumBase::Memory(_) => None, - } - } - #[cfg(not(feature = "replica"))] - { - None - } - }; - if commit < previous_count { - #[cfg(feature = "replica")] - if matches!(self.base, ChecksumBase::File(_)) { - self.remove_file_suffix(commit, previous_count, base_file.as_mut())?; - } else { - for page in commit + 1..=previous_count { - let old = self.value(page, base_file.as_mut())?; - self.checksum = CHECKSUM_FLAG | (self.checksum ^ old); - } - } - #[cfg(not(feature = "replica"))] - for page in commit + 1..=previous_count { - let old = self.value(page, base_file.as_mut())?; - self.checksum = CHECKSUM_FLAG | (self.checksum ^ old); - } - self.changes.retain(|page, _| *page <= commit); - } - - Ok(PageChecksumApply { - target: self, - page_size, - commit, - previous_count, - previous: 0, - next_required: previous_count.saturating_add(1), - base_file, - }) - } - - /// Adopts a sealed candidate; a failure requires fencing the capture session. - pub(crate) fn commit(&mut self, candidate: Self) -> Result<()> { - // Retire the predecessor only after sealing. Its Arc otherwise forces - // a complete memory-base copy even when only one checksum changed. - *self = candidate; - // A sealed overlay is consumed once. Retaining its empty allocation - // makes every later candidate clone the largest historical table. - let changes = std::mem::take(&mut self.changes); - match &mut self.base { - ChecksumBase::Memory(base) => { - let dense = Arc::make_mut(base); - dense.truncate(self.base_count as usize); - dense.resize(self.count as usize, 0); - for (page, checksum) in changes { - if page <= self.count { - dense[page as usize - 1] = checksum; - } - } - // A large truncation must release historical capacity. Small - // changes retain it so the next append need not copy the base. - if dense.capacity() > dense.len().saturating_mul(2) { - dense.shrink_to_fit(); - } - } - #[cfg(feature = "replica")] - ChecksumBase::File(base) => { - let mut file = base.open()?; - let length = u64::from(self.count) * 8; - if length > u64::from(self.base_count) * 8 { - file.set_len(length)?; - } - let mut changes = changes - .into_iter() - .filter(|(page, _)| *page <= self.count) - .collect::>(); - changes.sort_unstable_by_key(|(page, _)| *page); - let mut output = Vec::with_capacity(changes.len().min(DENSE_WRITE_BYTES / 8) * 8); - let mut start = 0; - for (page, checksum) in changes { - let offset = u64::from(page - 1) * 8; - if !output.is_empty() - && (offset != start + output.len() as u64 - || output.len() == DENSE_WRITE_BYTES) - { - file.write_all_at(start, &output)?; - output.clear(); - } - if output.is_empty() { - start = offset; - } - output.extend_from_slice(&checksum.to_be_bytes()); - } - if !output.is_empty() { - file.write_all_at(start, &output)?; - } - // This base is active-session scratch. A clean handoff writes - // and syncs a fresh dense sidecar; a crash cannot reopen this - // session directory or use its mutable base as authority. - file.set_len(length)?; - } - } - self.base_count = self.count; - Ok(()) - } - - pub fn checksum(&self) -> u64 { - self.checksum - } - - /// Returns the database page count this index describes. - #[cfg(feature = "replica")] - pub(crate) fn count(&self) -> u32 { - self.count - } - - /// Verifies the checksum-bearing pages of a clean local database. - #[cfg(feature = "replica")] - pub(crate) fn verify_database( - &self, - host: &crate::LtxHost, - path: &std::path::Path, - page_size: u32, - ) -> Result<()> { - let mut database = host.open(path)?; - let mut base_file = match &self.base { - ChecksumBase::File(base) => base.open()?, - ChecksumBase::Memory(_) => { - return Err(CrabError::InvalidState( - "resume requires a file-backed checksum index", - )); - } - }; - let mut fold = CHECKSUM_FLAG; - let pages_per_read = (CHECKSUM_READ_BYTES / page_size as usize).max(1) as u64; - let mut first = 1u64; - while first <= u64::from(self.count) { - let pages = (u64::from(self.count) - first + 1).min(pages_per_read); - let checksums = base_file.read_exact_at((first - 1) * 8, pages as usize * 8)?; - let image = database.read_exact_at( - (first - 1) * u64::from(page_size), - pages as usize * page_size as usize, - )?; - for (index, (stored, bytes)) in checksums - .as_chunks::<8>() - .0 - .iter() - .zip(image.chunks_exact(page_size as usize)) - .enumerate() - { - let page = (first + index as u64) as u32; - let expected = u64::from_be_bytes(*stored); - if page == ltx::lock_pgno(page_size) { - if expected != 0 { - return Err(CrabError::ChecksumMismatch); - } - continue; - } - if expected != ltx::checksum_page(page, bytes) { - return Err(CrabError::ChecksumMismatch); - } - fold = CHECKSUM_FLAG | (fold ^ expected); - } - first += pages; - } - if fold != self.checksum { - return Err(CrabError::ChecksumMismatch); - } - Ok(()) - } - - /// Writes the dense per-page checksum list, proving it folds to this index. - /// - /// The fold is the same aggregate the capture maintains, so a base file that - /// no longer matches it is refused instead of copied into a continuation - /// that a later open would trust. - #[cfg(feature = "replica")] - pub(crate) fn write_dense(&self, sink: &mut dyn crate::environment::FileIo) -> Result<()> { - let mut base_file = match &self.base { - ChecksumBase::File(base) => { - Some(BufReader::with_capacity(DENSE_WRITE_BYTES, base.open()?)) - } - ChecksumBase::Memory(_) => None, - }; - let mut fold = CHECKSUM_FLAG; - let mut output = Vec::with_capacity(DENSE_WRITE_BYTES); - for page in 1..=self.count { - let checksum = if page <= self.base_count - && let Some(file) = base_file.as_mut() - { - // Consume the base even when an overlay replaces this entry, - // so later pages retain their exact position in the sidecar. - let mut bytes = [0; 8]; - file.read_exact(&mut bytes)?; - let checksum = self - .changes - .get(&page) - .copied() - .unwrap_or(u64::from_be_bytes(bytes)); - if checksum != 0 && checksum & CHECKSUM_FLAG == 0 { - return Err(CrabError::LTXCorrupted); - } - checksum - } else { - self.value(page, None)? - }; - fold = CHECKSUM_FLAG | (fold ^ checksum); - output.extend_from_slice(&checksum.to_be_bytes()); - if output.len() >= DENSE_WRITE_BYTES { - sink.write_all(&output)?; - output.clear(); - } - } - if !output.is_empty() { - sink.write_all(&output)?; - } - if fold != self.checksum { - return Err(CrabError::ChecksumMismatch); - } - Ok(()) - } - - #[cfg(feature = "replica")] - fn remove_file_suffix( - &mut self, - commit: u32, - previous_count: u32, - file: Option<&mut ChecksumReader>, - ) -> Result<()> { - let file = file.ok_or(CrabError::InvalidState("checksum file not open"))?; - let mut page = commit.checked_add(1).ok_or(CrabError::LTXCorrupted)?; - let file_end = previous_count.min(self.base_count); - while page <= file_end { - let count = usize::try_from(file_end - page + 1) - .map_err(|_| CrabError::LTXCorrupted)? - .min(CHECKSUM_READ_BYTES / 8); - let bytes = file - .file - .read_exact_at(u64::from(page - 1) * 8, count * 8)?; - for checksum in bytes.as_chunks::<8>().0 { - let stored = u64::from_be_bytes(*checksum); - let old = self.changes.get(&page).copied().unwrap_or(stored); - if old != 0 && old & CHECKSUM_FLAG == 0 { - return Err(CrabError::LTXCorrupted); - } - self.checksum = CHECKSUM_FLAG | (self.checksum ^ old); - page = page.checked_add(1).ok_or(CrabError::LTXCorrupted)?; - } - } - while page <= previous_count { - let old = self.changes.get(&page).copied().unwrap_or(0); - if old != 0 && old & CHECKSUM_FLAG == 0 { - return Err(CrabError::LTXCorrupted); - } - self.checksum = CHECKSUM_FLAG | (self.checksum ^ old); - page = page.checked_add(1).ok_or(CrabError::LTXCorrupted)?; - } - Ok(()) - } - - fn value(&self, page: u32, file: Option<&mut ChecksumReader>) -> Result { - if page == 0 || page > self.count { - return Err(CrabError::LTXCorrupted); - } - if let Some(checksum) = self.changes.get(&page) { - return Ok(*checksum); - } - if page > self.base_count { - return Ok(0); - } - let checksum = if let Some(file) = file { - file.value(page, self.base_count)? - } else { - match &self.base { - ChecksumBase::Memory(pages) => *pages - .get(page as usize - 1) - .ok_or(CrabError::LTXCorrupted)?, - #[cfg(feature = "replica")] - ChecksumBase::File(_) => { - return Err(CrabError::InvalidState("checksum file not open")); - } - } - }; - if checksum != 0 && checksum & CHECKSUM_FLAG == 0 { - return Err(CrabError::LTXCorrupted); - } - Ok(checksum) - } -} - -impl PageChecksumApply<'_> { - pub(crate) fn page(&mut self, pgno: u32, data: &[u8]) -> Result<()> { - let lock = ltx::lock_pgno(self.page_size); - if pgno <= self.previous - || pgno > self.commit - || pgno == lock - || data.len() != self.page_size as usize - { - return Err(CrabError::LTXCorrupted); - } - while self.next_required == lock { - self.next_required = self - .next_required - .checked_add(1) - .ok_or(CrabError::LTXCorrupted)?; - } - if pgno > self.previous_count { - if pgno != self.next_required { - return Err(CrabError::LTXCorrupted); - } - self.next_required = self - .next_required - .checked_add(1) - .ok_or(CrabError::LTXCorrupted)?; - } - let old = if pgno <= self.previous_count { - self.target.value(pgno, self.base_file.as_mut())? - } else { - 0 - }; - let checksum = ltx::checksum_page(pgno, data); - self.target.checksum = CHECKSUM_FLAG | (self.target.checksum ^ old ^ checksum); - self.target.changes.insert(pgno, checksum); - self.previous = pgno; - Ok(()) - } - - pub(crate) fn finish(mut self) -> Result<()> { - let lock = ltx::lock_pgno(self.page_size); - while self.next_required == lock { - self.next_required = self - .next_required - .checked_add(1) - .ok_or(CrabError::LTXCorrupted)?; - } - if self.commit > self.previous_count && self.next_required <= self.commit { - return Err(CrabError::LTXCorrupted); - } - self.target.count = self.commit; - Ok(()) - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn fixed_size_memory_updates_reuse_the_owned_base() { - let allocation = |index: &PageChecksums| match &index.base { - ChecksumBase::Memory(base) => base.as_ptr(), - #[cfg(feature = "replica")] - ChecksumBase::File(_) => panic!("expected a memory base"), - }; - for count in [1024, 32_768] { - let mut owner = PageChecksums::default(); - owner - .apply_iter( - 512, - count, - (1..=count).map(|n| Ok((n, vec![n as u8; 512]))), - 32 << 20, - ) - .unwrap(); - owner.commit(owner.clone()).unwrap(); - assert_eq!(owner.changes.capacity(), 0, "merged {count}-page overlay"); - let before = allocation(&owner); - let expected_before = owner.checksum(); - let mut candidate = owner.clone(); - assert_eq!(candidate.changes.capacity(), 0, "next candidate overlay"); - candidate - .apply(512, count, &[(7, vec![91; 512])], 32 << 20) - .unwrap(); - assert_eq!(owner.checksum(), expected_before); - // Sealing retires the predecessor before merging the isolated overlay. - owner.commit(candidate).unwrap(); - assert_eq!(owner.changes.capacity(), 0, "merged one-page overlay"); - assert_eq!(allocation(&owner), before, "{count} pages"); - assert_eq!( - owner.value(7, None).unwrap(), - ltx::checksum_page(7, &[91; 512]) - ); - } - } - - #[test] - fn memory_candidates_preserve_snapshots_through_truncation_and_regrowth() { - for page_size in [512, 4096] { - let original = (1..=1025) - .map(|number| (number, vec![number as u8; page_size as usize])) - .collect::>(); - let mut owner = PageChecksums::default(); - owner.apply(page_size, 1025, &original, 8 << 20).unwrap(); - owner.commit(owner.clone()).unwrap(); - let snapshot = owner.clone(); - let mut candidate = owner.clone(); - let changed = (2, vec![90; page_size as usize]); - candidate - .apply(page_size, 100, std::slice::from_ref(&changed), 8 << 20) - .unwrap(); - owner.commit(candidate).unwrap(); - assert_eq!( - snapshot.value(2, None).unwrap(), - ltx::checksum_page(2, &original[1].1) - ); - - let mut candidate = owner.clone(); - let regrown = (101..=1027) - .map(|number| (number, vec![91; page_size as usize])) - .collect::>(); - candidate.apply(page_size, 1027, ®rown, 8 << 20).unwrap(); - owner.commit(candidate).unwrap(); - let expected = original[..100] - .iter() - .map(|page| if page.0 == 2 { &changed } else { page }) - .chain(®rown) - .fold(CHECKSUM_FLAG, |sum, (number, bytes)| { - CHECKSUM_FLAG | (sum ^ ltx::checksum_page(*number, bytes)) - }); - assert_eq!(owner.checksum(), expected); - assert_eq!(snapshot.checksum(), checksum(&original)); - } - } - - #[cfg(feature = "replica")] - fn page(number: u32, byte: u8) -> (u32, Vec) { - (number, vec![byte; 4096]) - } - - fn checksum(pages: &[(u32, Vec)]) -> u64 { - pages.iter().fold(CHECKSUM_FLAG, |sum, (number, bytes)| { - CHECKSUM_FLAG | (sum ^ ltx::checksum_page(*number, bytes)) - }) - } - - #[cfg(feature = "replica")] - #[test] - fn file_backed_overlay_matches_full_scan_across_update_truncate_and_regrowth() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("checksums"); - let mut pages = vec![page(1, 1), page(2, 2), page(3, 3), page(4, 4)]; - let initial_checksum = checksum(&pages); - let bytes: Vec = pages - .iter() - .flat_map(|(number, bytes)| ltx::checksum_page(*number, bytes).to_be_bytes()) - .collect(); - std::fs::write(&path, bytes).unwrap(); - let host = crate::LtxHost { - facilities: crate::Host::default(), - max_database_bytes: 1 << 20, - max_file_bytes: 1 << 20, - }; - let mut index = PageChecksums::from_file(host, &path, 4096, 4, initial_checksum).unwrap(); - assert!(matches!(index.base, ChecksumBase::File(_))); - assert!(index.changes.is_empty()); - - pages[1] = page(2, 9); - pages.truncate(3); - index.apply(4096, 3, &[pages[1].clone()], 1 << 20).unwrap(); - assert_eq!(index.changes.len(), 1); - assert_eq!(index.checksum(), checksum(&pages)); - index.commit(index.clone()).unwrap(); - assert!(index.changes.is_empty()); - assert_eq!(index.changes.capacity(), 0, "merged file overlay"); - assert_eq!( - std::fs::read(&path).unwrap(), - pages - .iter() - .flat_map(|(number, bytes)| ltx::checksum_page(*number, bytes).to_be_bytes()) - .collect::>() - ); - - pages[0] = page(1, 6); - pages.push(page(4, 7)); - pages.push(page(5, 8)); - index - .apply( - 4096, - 5, - &[pages[0].clone(), pages[3].clone(), pages[4].clone()], - 1 << 20, - ) - .unwrap(); - assert_eq!(index.checksum(), checksum(&pages)); - index.commit(index.clone()).unwrap(); - assert_eq!(index.changes.capacity(), 0, "merged regrown overlay"); - assert_eq!( - std::fs::read(path).unwrap(), - pages - .iter() - .flat_map(|(number, bytes)| ltx::checksum_page(*number, bytes).to_be_bytes()) - .collect::>() - ); - } - - #[cfg(feature = "replica")] - #[test] - fn file_backed_truncation_reduces_multiple_checksum_chunks_with_overlay_updates() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("checksums"); - let page_size = 512; - let count = 20_000; - let pages = (1..=count) - .map(|page| (page, vec![page as u8; page_size as usize])) - .collect::>(); - let initial_checksum = checksum(&pages); - let bytes = pages - .iter() - .flat_map(|(number, bytes)| ltx::checksum_page(*number, bytes).to_be_bytes()) - .collect::>(); - std::fs::write(&path, bytes).unwrap(); - let facilities = crate::Host::default(); - let host = crate::LtxHost { - facilities: facilities.clone(), - max_database_bytes: 32 << 20, - max_file_bytes: 32 << 20, - }; - let mut index = - PageChecksums::from_file(host, &path, page_size, count, initial_checksum).unwrap(); - index - .apply( - page_size, - count, - &[ - (101, vec![91; page_size as usize]), - (9_000, vec![92; page_size as usize]), - (19_999, vec![93; page_size as usize]), - ], - 32 << 20, - ) - .unwrap(); - - let dense_path = directory.path().join("dense"); - let mut dense = facilities.filesystem.create(&dense_path).unwrap(); - index.write_dense(dense.as_mut()).unwrap(); - drop(dense); - let expected = pages - .iter() - .map(|(number, bytes)| { - let replacement = match number { - 101 => Some(vec![91; page_size as usize]), - 9_000 => Some(vec![92; page_size as usize]), - 19_999 => Some(vec![93; page_size as usize]), - _ => None, - }; - ltx::checksum_page(*number, replacement.as_ref().unwrap_or(bytes)) - }) - .flat_map(u64::to_be_bytes) - .collect::>(); - assert_eq!(std::fs::read(dense_path).unwrap(), expected); - - index.apply(page_size, 100, &[], 32 << 20).unwrap(); - - assert_eq!(index.checksum(), checksum(&pages[..100])); - } -} diff --git a/crates/crab-ltx/src/recovery.rs b/crates/crab-ltx/src/recovery.rs deleted file mode 100644 index f9eece798..000000000 --- a/crates/crab-ltx/src/recovery.rs +++ /dev/null @@ -1,473 +0,0 @@ -//! Recovery planning: scratch budgets, verified plans, and materialized plans. -use std::io::{BufReader, BufWriter, Read, Write}; -use std::path::{Path, PathBuf}; -use std::sync::Arc; - -use crate::{ - CrabError, Limits, LocalSegment, Position, Result, SegmentInfo, ltx, pages::PageChecksums, -}; - -#[cfg(feature = "replica")] -pub(crate) fn full_job_scratch_bytes(page_size: u32, database_pages: u32) -> Result { - const HEADROOM: u64 = 64 << 20; - u64::from(page_size) - .checked_mul(u64::from(database_pages)) - .and_then(|bytes| bytes.checked_mul(2)) - .and_then(|bytes| bytes.checked_add(HEADROOM)) - .ok_or(CrabError::Limit(crate::LimitKind::ScratchDiskBytes)) -} - -/// A fully verified, explicit snapshot-plus-deltas plan ending at an exact position. -/// -/// Construction reads only named files and owns their exact verified database -/// image, preventing later path replacement from changing the plan. A remote -/// manifest's authenticity, repository identity, epoch, and object selection -/// remain the caller's job. -pub struct VerifiedPlan { - pub(crate) infos: Vec, - materialized: MaterializedPlan, - limits: Limits, -} - -pub(crate) struct MaterializedPlan { - pub(crate) image: Vec, - pub(crate) checksums: PageChecksums, - pub(crate) page_size: u32, - pub(crate) database_pages: u32, - pub(crate) position: Position, - timestamp: i64, -} - -#[derive(Default)] -struct MaterializationState { - image: Vec, - page: Vec, - checksums: PageChecksums, - position: Position, - page_size: u32, - timestamp: i64, -} - -impl MaterializationState { - fn apply(&mut self, bytes: &[u8], info: &SegmentInfo, limits: Limits) -> Result<()> { - if bytes.len() as u64 > limits.max_file_bytes { - return Err(CrabError::Limit(crate::LimitKind::LtxBytes)); - } - if bytes.len() as u64 != info.size_bytes { - return Err(CrabError::ChecksumMismatch); - } - let mut decoder = crate::codec::Decoder::new(bytes); - decoder.decode_header()?; - let header = decoder.header; - validate_header(&header, limits)?; - if header.min_txid.0 - != self - .position - .txid - .checked_add(1) - .ok_or(CrabError::TxNotAvailable)? - { - return Err(CrabError::TxNotAvailable); - } - if header.pre_apply_checksum != self.position.checksum { - return Err(CrabError::ChecksumMismatch); - } - if self.page_size != 0 && self.page_size != header.page_size { - return Err(CrabError::LTXCorrupted); - } - self.page_size = header.page_size; - self.timestamp = header.timestamp; - let image_len = usize::try_from(header.commit) - .ok() - .and_then(|pages| pages.checked_mul(header.page_size as usize)) - .ok_or(CrabError::Limit(crate::LimitKind::DatabaseBytes))?; - self.image.resize(image_len, 0); - let mut checksums = self.checksums.begin_apply( - header.page_size, - header.commit, - limits.max_database_bytes, - )?; - self.page.resize(header.page_size as usize, 0); - while let Some(page) = decoder.decode_page(&mut self.page)? { - checksums.page(page.pgno, &self.page)?; - let offset = (page.pgno as usize - 1) - .checked_mul(header.page_size as usize) - .ok_or(CrabError::Limit(crate::LimitKind::DatabaseBytes))?; - let end = offset - .checked_add(header.page_size as usize) - .ok_or(CrabError::Limit(crate::LimitKind::DatabaseBytes))?; - self.image[offset..end].copy_from_slice(&self.page); - } - checksums.finish()?; - decoder.close()?; - let (size, digest) = decoder.artifact()?; - let file = ltx::DecodedFile { - header, - trailer: decoder.trailer, - }; - if self.checksums.checksum() != file.trailer.post_apply_checksum { - return Err(CrabError::ChecksumMismatch); - } - if size != bytes.len() as u64 || SegmentInfo::from_inspected(&file, size, digest) != *info { - return Err(CrabError::ChecksumMismatch); - } - self.position = Position { - txid: header.max_txid.0, - checksum: file.trailer.post_apply_checksum, - }; - Ok(()) - } - - fn finish(self, target: Position) -> Result { - if self.position != target || self.page_size == 0 { - return Err(CrabError::ChecksumMismatch); - } - let database_pages = u32::try_from(self.image.len() / self.page_size as usize) - .map_err(|_| CrabError::Limit(crate::LimitKind::DatabasePages))?; - Ok(MaterializedPlan { - image: self.image, - checksums: self.checksums, - page_size: self.page_size, - database_pages, - position: self.position, - timestamp: self.timestamp, - }) - } -} - -impl VerifiedPlan { - /// Verifies expected sizes/digests, headers, page ordering, checksums, and continuity. - /// - /// The first file must be a full snapshot. Subsequent files must start at - /// the previous maximum TXID plus one; gaps, overlaps and implicit latest - /// selection are rejected. Every resulting database state is checksummed. - pub fn new(segments: &[LocalSegment], target: Position, limits: Limits) -> Result { - crate::Host::default().verify(segments, target, limits) - } - - pub(crate) fn with_host( - segments: &[LocalSegment], - target: Position, - limits: Limits, - host: &crate::Host, - ) -> Result { - let limits = limits.validate()?; - if segments.is_empty() { - return Err(CrabError::TxNotAvailable); - } - if segments.len() > limits.max_segments { - return Err(CrabError::Limit(crate::LimitKind::PlanSegments)); - } - let mut infos = Vec::with_capacity(segments.len()); - let mut materialization = MaterializationState::default(); - let mut total = 0u64; - for segment in segments { - total = total - .checked_add(segment.info().size_bytes) - .ok_or(CrabError::Limit(crate::LimitKind::PlanBytes))?; - if total > limits.max_plan_bytes || segment.info().size_bytes > limits.max_file_bytes { - return Err(CrabError::Limit(crate::LimitKind::PlanBytes)); - } - let bytes = host.read(segment.path(), segment.info().size_bytes)?; - materialization.apply(&bytes, segment.info(), limits)?; - infos.push(segment.info().clone()); - } - let materialized = materialization.finish(target)?; - Ok(Self { - infos, - materialized, - limits, - }) - } - - /// Returns the position the recovered database ends at. - #[must_use] - pub fn position(&self) -> Position { - self.materialized.position - } - - pub(crate) fn materialize(&self) -> &MaterializedPlan { - &self.materialized - } -} - -fn validate_header(header: <x::Header, limits: Limits) -> Result<()> { - header.validate()?; - if header.no_checksum() { - return Err(CrabError::LTXCorrupted); - } - if u64::from(header.commit) * u64::from(header.page_size) > limits.max_database_bytes { - return Err(CrabError::Limit(crate::LimitKind::DatabaseBytes)); - } - Ok(()) -} - -#[cfg(feature = "replica")] -pub(crate) fn verify_segment(bytes: &[u8], info: &SegmentInfo, limits: Limits) -> Result<()> { - if bytes.len() as u64 > limits.max_file_bytes { - return Err(CrabError::Limit(crate::LimitKind::LtxBytes)); - } - // Buffer callers know their exact bytes up front; preserve permanent - // length/digest rejection before the decoder can report a short read. - if bytes.len() as u64 != info.size_bytes || *blake3::hash(bytes).as_bytes() != info.blake3 { - return Err(CrabError::ChecksumMismatch); - } - verify_segment_reader(bytes, info, limits) -} - -#[cfg(feature = "replica")] -pub(crate) fn verify_segment_reader( - reader: impl Read, - info: &SegmentInfo, - limits: Limits, -) -> Result<()> { - if info.size_bytes > limits.max_file_bytes { - return Err(CrabError::Limit(crate::LimitKind::LtxBytes)); - } - // Read one extra byte to reject trailing data without trusting a stream's - // length. Retain page scratch and indexes, not the complete compressed body. - let mut decoder = crate::codec::Decoder::new(reader.take(info.size_bytes.saturating_add(1))); - decoder.decode_header()?; - validate_header(&decoder.header, limits)?; - let mut page = vec![0; decoder.header.page_size as usize]; - while decoder.decode_page(&mut page)?.is_some() {} - decoder.close()?; - let (size, digest) = decoder.artifact()?; - let decoded = ltx::DecodedFile { - header: decoder.header, - trailer: decoder.trailer, - }; - if SegmentInfo::from_inspected(&decoded, size, digest) != *info { - return Err(CrabError::ChecksumMismatch); - } - Ok(()) -} - -/// Restores exactly the verified target into a new, atomically installed SQLite file. -/// -/// Never overwrites a destination; no WAL, local database, or bucket listing is -/// consulted. Caller must prevent concurrent use of the destination and sidecars. -pub fn restore_exact(plan: &VerifiedPlan, destination: &Path) -> Result { - crate::Host::default().restore(plan, destination) -} - -/// Compacts exactly the verified snapshot chain into a new self-contained LTX snapshot. -/// -/// Output is checked against the original target before installation. Input -/// deletion, remote publication, retention, and partial-range compaction are not performed. -pub fn compact_exact(plan: &VerifiedPlan, destination: &Path) -> Result { - crate::Host::default().compact(plan, destination) -} - -pub(crate) fn compact_to_file( - host: &crate::Host, - plan: &VerifiedPlan, - destination: &Path, -) -> Result { - let first = plan.infos.first().ok_or(CrabError::TxNotAvailable)?; - let last = plan.infos.last().ok_or(CrabError::TxNotAvailable)?; - if plan.infos.len() > plan.limits.max_segments { - return Err(CrabError::Limit(crate::LimitKind::CompactionInputs)); - } - let materialized = plan.materialize(); - if materialized.checksums.checksum() != materialized.position.checksum { - return Err(CrabError::ChecksumMismatch); - } - let (mut scratch, file) = CompactionScratch::create(host, destination)?; - let writer = BoundedFileWriter { - file, - limit: plan.limits.max_file_bytes, - written: 0, - digest: blake3::Hasher::new(), - }; - let mut encoder = crate::codec::Encoder::new_block(BufWriter::with_capacity(1 << 20, writer)); - encoder.encode_header(ltx::Header { - version: ltx::VERSION, - page_size: materialized.page_size, - commit: materialized.database_pages, - min_txid: crate::Txid(first.min_txid), - max_txid: crate::Txid(last.max_txid), - timestamp: materialized.timestamp, - pre_apply_checksum: first.pre_checksum, - ..ltx::Header::default() - })?; - let page_size = materialized.page_size as usize; - let lock_page = ltx::lock_pgno(materialized.page_size); - for page in 1..=materialized.database_pages { - if page == lock_page { - continue; - } - let start = (page as usize - 1) - .checked_mul(page_size) - .ok_or(CrabError::Limit(crate::LimitKind::DatabaseBytes))?; - let end = start - .checked_add(page_size) - .ok_or(CrabError::Limit(crate::LimitKind::DatabaseBytes))?; - let data = materialized - .image - .get(start..end) - .ok_or(CrabError::LTXCorrupted)?; - encoder.encode_page( - ltx::PageHeader { - pgno: page, - flags: 0, - }, - data, - )?; - } - encoder.close(materialized.position.checksum)?; - let encoded = ltx::DecodedFile { - header: encoder.header, - trailer: encoder.trailer, - }; - let writer = encoder.into_writer(); - let writer = writer.into_inner().map_err(|error| error.into_error())?; - let (mut file, size, digest) = writer.finish(); - file.sync_all()?; - drop(file); - - let ltx_host = crate::LtxHost { - facilities: host.clone(), - max_database_bytes: plan.limits.max_database_bytes, - max_file_bytes: plan.limits.max_file_bytes, - }; - let output = BufReader::with_capacity(1 << 20, ltx_host.open(&scratch.path)?); - let (stored_size, stored_digest) = digest_reader(output)?; - if stored_size != size || stored_digest != digest { - return Err(CrabError::ChecksumMismatch); - } - let info = SegmentInfo::from_inspected(&encoded, size, digest); - if info.min_txid != first.min_txid - || info.max_txid != last.max_txid - || info.pre_checksum != first.pre_checksum - || info.position() != materialized.position - || info.database_pages != last.database_pages - || info.page_size != first.page_size - { - return Err(CrabError::ChecksumMismatch); - } - host.filesystem - .persist_file_new(&scratch.path, destination)?; - scratch.installed = true; - Ok(LocalSegment::new(destination.to_owned(), info)) -} - -struct BoundedFileWriter { - file: Box, - limit: u64, - written: u64, - digest: blake3::Hasher, -} - -impl Write for BoundedFileWriter { - fn write(&mut self, bytes: &[u8]) -> std::io::Result { - let end = self - .written - .checked_add(bytes.len() as u64) - .ok_or_else(|| std::io::Error::other("compacted file byte limit exceeded"))?; - if end > self.limit { - return Err(std::io::Error::other("compacted file byte limit exceeded")); - } - self.file.write_all(bytes)?; - self.digest.update(bytes); - self.written = end; - Ok(bytes.len()) - } - fn flush(&mut self) -> std::io::Result<()> { - Ok(()) - } -} - -impl BoundedFileWriter { - fn finish(self) -> (Box, u64, [u8; 32]) { - (self.file, self.written, *self.digest.finalize().as_bytes()) - } -} - -fn digest_reader(mut reader: impl Read) -> Result<(u64, [u8; 32])> { - let mut digest = blake3::Hasher::new(); - let mut size = 0u64; - let mut buffer = [0; 64 << 10]; - loop { - let read = reader.read(&mut buffer)?; - if read == 0 { - break; - } - size = size - .checked_add(read as u64) - .ok_or(CrabError::Limit(crate::LimitKind::LtxBytes))?; - digest.update(&buffer[..read]); - } - Ok((size, *digest.finalize().as_bytes())) -} - -struct CompactionScratch { - filesystem: Arc, - path: PathBuf, - installed: bool, -} - -impl CompactionScratch { - fn create( - host: &crate::Host, - destination: &Path, - ) -> Result<(Self, Box)> { - let parent = destination - .parent() - .filter(|path| !path.as_os_str().is_empty()) - .unwrap_or(Path::new(".")); - let filename = destination - .file_name() - .ok_or(CrabError::InvalidState("missing compaction filename"))?; - static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); - for _ in 0..16 { - let mut scratch_name = filename.to_owned(); - scratch_name.push(format!( - ".tmp-crab-ltx-compaction-{}-{}", - std::process::id(), - NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed) - )); - let path = parent.join(scratch_name); - match host.filesystem.create(&path) { - Ok(file) => { - return Ok(( - Self { - filesystem: host.filesystem.clone(), - path, - installed: false, - }, - file, - )); - } - Err(error) if error.kind() == std::io::ErrorKind::AlreadyExists => continue, - Err(error) => return Err(error.into()), - } - } - Err(std::io::Error::new( - std::io::ErrorKind::AlreadyExists, - "compaction scratch namespace exhausted", - ) - .into()) - } -} - -impl Drop for CompactionScratch { - fn drop(&mut self) { - if !self.installed { - let _ = self.filesystem.remove_file(&self.path); - } - } -} - -pub(crate) fn reject_sidecars(path: &Path, host: &crate::Host) -> Result<()> { - for suffix in ["-wal", "-shm", "-journal"] { - let mut sidecar = path.as_os_str().to_owned(); - sidecar.push(suffix); - if host.filesystem.exists(Path::new(&sidecar))? { - return Err(CrabError::InvalidState( - "restore destination has SQLite sidecars", - )); - } - } - Ok(()) -} diff --git a/crates/crab-ltx/src/replica.rs b/crates/crab-ltx/src/replica.rs deleted file mode 100644 index 4fb46dcba..000000000 --- a/crates/crab-ltx/src/replica.rs +++ /dev/null @@ -1,962 +0,0 @@ -//! Immutable Cell-scoped LTX roots prepared independently of ownership CAS. - -use std::{ - collections::BTreeMap, - io, - path::{Path, PathBuf}, - sync::{Arc, Mutex}, -}; - -use crate::{CellObjectKind, CellStorageLayout}; -use bytes::Bytes; -use futures_util::{StreamExt as _, TryStreamExt as _, stream}; - -use crate::{CaptureBatch, CrabError, Host, Limits, Position, Result}; - -mod cache; -mod compaction; -pub(crate) mod directory; -mod merge; -mod prepare; -mod read_only; -mod restore; -pub(crate) mod root; -mod upload; -mod verify; - -use directory::{DirectoryEntry, DirectorySpan, DirectoryTree, ObjectExtent}; -pub use read_only::ReadOnlyRoot; -use root::{ - RootDocument, SegmentDescriptor, decode_root, decode_segment_page, encode_root, - encode_segment_page, -}; - -const ROOT_BYTES: u64 = 32 << 10; -const SEGMENT_PAGE_BYTES: u64 = 64 << 10; -const MAX_SEGMENTS: usize = 4096; -const SEGMENTS_PER_PAGE: usize = 96; -const MAX_SEGMENT_PAGES: usize = 64; -const COMPACTION_FANOUT: usize = 8; -const MAX_COMPACTION_INPUTS: usize = 128; -// These buffers schedule immutable reads; Host I/O permits remain the shared -// admission boundary across roots, restores, and concurrent Cells. -const OBJECT_FETCH_CONCURRENCY: usize = 8; -pub(super) const OBJECT_UPLOAD_CONCURRENCY: usize = 8; -// A capture body upload can retain four multipart chunks. Bound each -// multi-segment transfer cohort here; Host permits cap aggregate cohorts. -pub(super) const SEGMENT_TRANSFER_CONCURRENCY: usize = 4; -pub(super) const RESTORE_IN_FLIGHT_WINDOWS: usize = 8; -pub(super) const RESTORE_WINDOW_BYTES: u32 = 1 << 20; - -/// An immutable Cell root identity suitable for publication in control state. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct RootRef { - /// Cell this root belongs to. - pub cell: [u8; 32], - /// Incarnation that produced the root. - pub incarnation: [u8; 16], - /// Digest of the published root record. - pub digest: [u8; 32], - /// Position the root publishes. - pub position: Position, - /// Root commit sequence. - pub commit_sequence: u64, -} - -/// One immutable object authenticated as part of an exact Cell root. -#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] -pub struct RootObjectRef { - /// Digest of the immutable object. - pub digest: [u8; 32], - /// Kind of immutable object. - pub kind: CellObjectKind, -} - -/// Immutable publication cost a replica path paid. -/// -/// One prepared root uploads a segment body and index, the changed directory -/// nodes, and the root document, so a caller that wants the object-store cost -/// of one command reads this ledger instead of inferring it from the database -/// size. -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub struct PublicationCost { - /// Immutable objects uploaded. - pub objects: u64, - /// Sum of immutable object bytes uploaded. - pub bytes: u64, -} - -#[derive(Default)] -pub(super) struct PublicationLedger { - objects: std::sync::atomic::AtomicU64, - bytes: std::sync::atomic::AtomicU64, -} - -impl PublicationLedger { - pub(super) fn record(&self, bytes: u64) { - self.objects - .fetch_add(1, std::sync::atomic::Ordering::Relaxed); - self.bytes - .fetch_add(bytes, std::sync::atomic::Ordering::Relaxed); - } - - /// Returns the accumulated cost and resets the ledger. - fn take(&self) -> PublicationCost { - PublicationCost { - objects: self.objects.swap(0, std::sync::atomic::Ordering::AcqRel), - bytes: self.bytes.swap(0, std::sync::atomic::Ordering::AcqRel), - } - } - - /// Returns the accumulated cost without resetting the ledger. - fn snapshot(&self) -> PublicationCost { - PublicationCost { - objects: self.objects.load(std::sync::atomic::Ordering::Relaxed), - bytes: self.bytes.load(std::sync::atomic::Ordering::Relaxed), - } - } -} - -/// A fully uploaded immutable root proposal. -/// -/// Construction is private so callers cannot publish a root before all of its -/// dependencies have passed verification and immutable upload. -#[derive(Clone)] -pub struct PreparedRoot { - predecessor: Option, - verified: VerifiedRoot, -} - -/// A verified recovery bundle pinned to one exact published predecessor. -/// -/// The bundle validates every embedded LTX file at construction. A -/// [`CellReplica`] additionally verifies Cell scope, chain continuity, and the -/// declared final position before it uploads a successor root. -pub struct RecoveryOverlay { - predecessor: RootRef, - bundle: crate::bundle::Bundle, - final_position: Position, - final_commit_sequence: u64, - _disk_reservation: Option, - _bundle_lease: Option>, -} - -impl RecoveryOverlay { - /// Describes the overlay that supersedes `predecessor`; the caller adds - /// the disk reservation and bundle lease that keep it readable. - #[must_use] - pub fn new( - predecessor: RootRef, - bundle: crate::bundle::Bundle, - final_position: Position, - final_commit_sequence: u64, - ) -> Self { - Self { - predecessor, - bundle, - final_position, - final_commit_sequence, - _disk_reservation: None, - _bundle_lease: None, - } - } - - /// Keeps temporary recovery storage admitted until this overlay is dropped. - #[must_use] - pub fn with_disk_reservation(mut self, reservation: crate::DiskReservation) -> Self { - self._disk_reservation = Some(reservation); - self - } - - /// Keeps a server-owned verified artifact alive while this overlay is used. - #[must_use] - pub fn with_bundle_lease(mut self, lease: Arc) -> Self { - self._bundle_lease = Some(lease); - self - } - - /// Transfers an unleased verified bundle out after publication. - pub fn into_bundle(self) -> crate::Result { - if self._disk_reservation.is_some() || self._bundle_lease.is_some() { - return Err(crate::CrabError::InvalidState( - "recovery overlay still owns bundle resources", - )); - } - Ok(self.bundle) - } - - /// Returns the root this overlay supersedes. - #[must_use] - pub const fn predecessor(&self) -> RootRef { - self.predecessor - } - - /// Returns the position the overlay publishes. - #[must_use] - pub const fn final_position(&self) -> Position { - self.final_position - } - - /// Returns the commit sequence the overlay publishes. - #[must_use] - pub const fn final_commit_sequence(&self) -> u64 { - self.final_commit_sequence - } - - /// Returns the verified recovery bundle. - #[must_use] - pub fn bundle(&self) -> &crate::bundle::Bundle { - &self.bundle - } -} - -impl PreparedRoot { - /// Returns the exact root that was prepared. - #[must_use] - pub fn root(&self) -> RootRef { - self.verified.root - } - - /// Returns the root this preparation supersedes, if any. - #[must_use] - pub fn predecessor(&self) -> Option { - self.predecessor - } - - /// Returns the verified metadata for the prepared root. - #[must_use] - pub fn verified(&self) -> &VerifiedRoot { - &self.verified - } -} - -/// Metadata and dependency graph verified from one exact immutable root. -#[derive(Clone)] -pub struct VerifiedRoot { - root: RootRef, - page_size: u32, - database_pages: u32, - schema: u32, - segment_count: usize, - directory_height: u32, - pages: CellPagedDatabase, -} - -impl VerifiedRoot { - /// Returns the exact immutable root. - #[must_use] - pub fn root(&self) -> RootRef { - self.root - } - - /// Returns the SQLite page size. - #[must_use] - pub fn page_size(&self) -> u32 { - self.page_size - } - - /// Returns the database page count. - #[must_use] - pub fn database_pages(&self) -> u32 { - self.database_pages - } - - /// Returns the schema version. - #[must_use] - pub fn schema(&self) -> u32 { - self.schema - } - - /// Returns the number of segments the root references. - #[must_use] - pub fn segment_count(&self) -> usize { - self.segment_count - } - - /// Returns the height of the root's directory tree. - #[must_use] - pub fn directory_height(&self) -> u32 { - self.directory_height - } - - /// Returns a pinned, lazy page reader for this exact root. - #[must_use] - pub fn paged(&self) -> CellPagedDatabase { - self.pages.clone() - } - - /// Streams this exact root into a new local SQLite file. - /// - /// The destination and its SQLite sidecars must not exist. Every page is - /// authenticated before an atomically installed result becomes visible. - pub async fn restore(&self, destination: &Path) -> Result { - self.pages - .replica - .host - .observe_ltx_logical_read(crate::LtxReadOrigin::Cold); - restore::run(&self.pages, destination).await - } - - /// Opens this exact root as an authenticated, sparse read-only SQLite view. - /// - /// The destination and its sidecars must not exist. The view removes its - /// empty placeholder when the last owner drops; page bodies use a bounded cache. - /// Call on a SQLite worker: opening can block on authenticated page faults. - pub fn open_read_only(&self, destination: &Path) -> Result { - ReadOnlyRoot::open(self, destination) - } -} - -/// Lazy authenticated page access through one immutable Cell root. -#[derive(Clone)] -pub struct CellPagedDatabase { - replica: CellReplica, - directory_digest: [u8; 32], - directory_height: u32, - extents: Arc>, - page_size: u32, - database_pages: u32, - position: Position, -} - -pub(super) struct FetchedSpan { - span: DirectorySpan, - frames: Bytes, -} - -impl FetchedSpan { - pub(super) fn try_for_each_page( - self, - page_size: u32, - mut operation: impl FnMut(u32, Vec) -> Result<()>, - ) -> Result<()> { - for entry in self.span.entries { - let offset = usize::try_from(entry.offset - self.span.start) - .map_err(|_| CrabError::LTXCorrupted)?; - let frame_end = offset - .checked_add(entry.length as usize) - .ok_or(CrabError::LTXCorrupted)?; - let frame = self - .frames - .get(offset..frame_end) - .ok_or(CrabError::LTXCorrupted)?; - if *blake3::hash(frame).as_bytes() != entry.frame_hash { - return Err(CrabError::ChecksumMismatch); - } - let bytes = crate::paged::decode_frame(frame, page_size, entry.page)?; - if crate::ltx::checksum_page(entry.page, &bytes) != entry.checksum { - return Err(CrabError::ChecksumMismatch); - } - operation(entry.page, bytes)?; - } - Ok(()) - } -} - -/// Exact immutable Cell root prepared for writable sparse activation. -/// -/// Preparation loads authenticated directory checksums, never LTX page bodies. -/// The resulting value can cross into the Cell's assigned SQLite worker. -#[derive(Clone)] -pub struct CellWritableDatabase { - database: CellPagedDatabase, - checksums: crate::pages::PageChecksums, - destination: std::path::PathBuf, -} - -impl CellPagedDatabase { - pub(crate) fn host(&self) -> Host { - self.replica.host.clone() - } - - pub(crate) fn limits(&self) -> Limits { - self.replica.limits - } - - /// Returns the position this root publishes. - #[must_use] - pub fn position(&self) -> Position { - self.position - } - - /// Returns the SQLite page size. - #[must_use] - pub fn page_size(&self) -> u32 { - self.page_size - } - - /// Returns the database page count. - #[must_use] - pub fn page_count(&self) -> u32 { - self.database_pages - } - - /// Streams the authenticated checksum index to this activation's local disk. - /// - /// The destination must be fresh and must later be passed unchanged to - /// `CellWritableDatabase::open_writable`. - /// Requires a running Tokio runtime, which must remain alive while canceled - /// activation work releases its file and host admission. - pub async fn prepare_writable( - mut self, - destination: &std::path::Path, - ) -> Result { - let host = self.replica.host.for_dirty().await?; - self.replica = self.replica.with_host(host); - let checksums = directory::load_checksums( - directory::Verification { - layout: &self.replica.layout, - cell: &self.replica.cell, - incarnation: &self.replica.incarnation, - page_size: self.page_size, - database_pages: self.database_pages, - extents: &self.extents, - host: &self.replica.host, - origin: crate::LtxReadOrigin::Cold, - }, - self.directory_digest, - self.directory_height, - destination, - self.replica.limits, - ) - .await?; - let host = self.replica.host.clone().without_dirty(); - self.replica = self.replica.with_host(host); - Ok(CellWritableDatabase { - database: self, - checksums, - destination: destination.to_owned(), - }) - } - - /// Reads one page by verifying every radix node and the selected LTX frame. - pub async fn read_page(&self, page: u32) -> Result> { - self.replica - .host - .observe_ltx_logical_read(crate::LtxReadOrigin::Sparse); - self.read_page_with_origin(page, crate::LtxReadOrigin::Sparse) - .await - } - - async fn read_page_with_origin( - &self, - page: u32, - origin: crate::LtxReadOrigin, - ) -> Result> { - if page == crate::ltx::lock_pgno(self.page_size) && page <= self.database_pages { - return Ok(vec![0; self.page_size as usize]); - } - let directory_started = self.replica.host.now_monotonic(); - let entry = directory::lookup( - directory::Verification { - layout: &self.replica.layout, - cell: &self.replica.cell, - incarnation: &self.replica.incarnation, - page_size: self.page_size, - database_pages: self.database_pages, - extents: &self.extents, - host: &self.replica.host, - origin, - }, - self.directory_digest, - self.directory_height, - page, - ) - .await; - self.replica.host.observe_ltx_phase( - crate::LtxPhase::Directory, - directory_started, - entry.is_ok(), - ); - let entry = entry?; - let extent = self - .extents - .get(&entry.object) - .ok_or(CrabError::LTXCorrupted)?; - let end = entry - .offset - .checked_add(u64::from(entry.length)) - .ok_or(CrabError::LTXCorrupted)?; - let path = self.replica.layout.incarnation_object_path( - &self.replica.cell, - &self.replica.incarnation, - &entry.object, - extent.kind, - ); - let _permit = self.replica.host.io_permit().await?; - let fetch_started = self.replica.host.now_monotonic(); - let frame = self - .replica - .layout - .store() - .range_get(&path, entry.offset..end) - .await; - self.replica.host.observe_ltx_phase( - crate::LtxPhase::FrameFetch, - fetch_started, - frame.is_ok(), - ); - self.replica.host.observe_ltx_origin_request( - origin, - frame.is_ok(), - frame.as_ref().map_or(0, |bytes| bytes.len()), - ); - let frame = frame?; - if frame.len() != entry.length as usize - || *blake3::hash(&frame).as_bytes() != entry.frame_hash - { - return Err(CrabError::ChecksumMismatch); - } - let bytes = crate::paged::decode_frame(&frame, self.page_size, page)?; - if crate::ltx::checksum_page(page, &bytes) != entry.checksum { - return Err(CrabError::ChecksumMismatch); - } - Ok(bytes) - } - - pub(crate) async fn read_run( - &self, - first: u32, - max_pages: u32, - origin: crate::LtxReadOrigin, - ) -> Result)>> { - if max_pages == 0 || first == 0 || first > self.database_pages { - return Ok(Vec::new()); - } - let lock = crate::ltx::lock_pgno(self.page_size); - if first == lock { - return Ok(vec![( - first, - self.read_page_with_origin(first, origin).await?, - )]); - } - let count = max_pages - .min(RESTORE_WINDOW_BYTES / self.page_size) - .min(self.database_pages - first + 1); - let span = self - .lookup_spans(first, count, origin) - .await? - .into_iter() - .next() - .ok_or(CrabError::LTXCorrupted)?; - let fetched = self.fetch_span(span, origin).await?; - let mut output = Vec::new(); - fetched.try_for_each_page(self.page_size, |page, bytes| { - output.push((page, bytes)); - Ok(()) - })?; - Ok(output) - } - - async fn read_restore_window(&self, first: u32, count: u32) -> Result> { - let spans = self - .lookup_spans(first, count, crate::LtxReadOrigin::Cold) - .await?; - let runs = stream::iter( - spans - .into_iter() - .map(|span| self.fetch_span(span, crate::LtxReadOrigin::Cold)), - ) - .buffered(OBJECT_FETCH_CONCURRENCY) - .try_collect() - .await?; - Ok(runs) - } - - async fn lookup_spans( - &self, - first: u32, - count: u32, - origin: crate::LtxReadOrigin, - ) -> Result> { - let started = self.replica.host.now_monotonic(); - let result = directory::lookup_spans( - directory::Verification { - layout: &self.replica.layout, - cell: &self.replica.cell, - incarnation: &self.replica.incarnation, - page_size: self.page_size, - database_pages: self.database_pages, - extents: &self.extents, - host: &self.replica.host, - origin, - }, - self.directory_digest, - self.directory_height, - first, - count, - ) - .await; - self.replica - .host - .observe_ltx_phase(crate::LtxPhase::Directory, started, result.is_ok()); - result - } - - async fn fetch_span( - &self, - span: DirectorySpan, - origin: crate::LtxReadOrigin, - ) -> Result { - let extent = self - .extents - .get(&span.object) - .ok_or(CrabError::LTXCorrupted)?; - let path = self.replica.layout.incarnation_object_path( - &self.replica.cell, - &self.replica.incarnation, - &span.object, - extent.kind, - ); - let _permit = self.replica.host.io_permit().await?; - let started = self.replica.host.now_monotonic(); - let frames = self - .replica - .layout - .store() - .range_get(&path, span.start..span.end) - .await; - self.replica - .host - .observe_ltx_phase(crate::LtxPhase::FrameFetch, started, frames.is_ok()); - self.replica.host.observe_ltx_origin_request( - origin, - frames.is_ok(), - frames.as_ref().map_or(0, |bytes| bytes.len()), - ); - let frames = frames?; - if frames.len() as u64 != span.end - span.start { - return Err(CrabError::ChecksumMismatch); - } - Ok(FetchedSpan { span, frames }) - } -} - -impl CellWritableDatabase { - pub(crate) fn host(&self) -> Host { - self.database.replica.host.clone() - } - - pub(crate) fn limits(&self) -> Limits { - self.database.replica.limits - } - - pub(crate) fn checksums(&self) -> crate::pages::PageChecksums { - self.checksums.clone() - } - - /// Returns the position this activation reads. - #[must_use] - pub fn position(&self) -> Position { - self.database.position() - } - - /// Returns the SQLite page size. - #[must_use] - pub fn page_size(&self) -> u32 { - self.database.page_size() - } - - /// Returns the database page count. - #[must_use] - pub fn page_count(&self) -> u32 { - self.database.page_count() - } - - pub(crate) async fn read_run( - &self, - first: u32, - max_pages: u32, - origin: crate::LtxReadOrigin, - ) -> Result)>> { - self.database.read_run(first, max_pages, origin).await - } - - /// Opens a fresh sparse SQLite file pinned to this exact Cell root. - pub fn open_writable(self, destination: &std::path::Path) -> Result { - if destination != self.destination { - return Err(CrabError::InvalidState( - "writable destination differs from prepared destination", - )); - } - crate::Db::open_cell_paged(self, destination) - } -} - -/// Immutable LTX object graph for one Cell incarnation. -/// -/// Ownership, command acknowledgement and the `control.json` CAS belong to the -/// runtime. This type never writes a mutable pointer or lists object storage. -#[derive(Clone)] -pub struct CellReplica { - layout: CellStorageLayout, - cell: [u8; 32], - incarnation: [u8; 16], - limits: Limits, - host: Host, - cost: Arc, -} - -impl CellReplica { - /// Binds all immutable operations to one Cell incarnation. - pub fn new( - layout: CellStorageLayout, - cell: [u8; 32], - incarnation: [u8; 16], - limits: Limits, - ) -> Result { - if layout.store().staging_write_prefix().is_some() { - return Err(CrabError::InvalidState("staged Cell replica store")); - } - Ok(Self { - layout, - cell, - incarnation, - limits: limits.validate()?, - host: Host::default(), - cost: Arc::new(PublicationLedger::default()), - }) - } - - /// Returns the immutable admission limits selected for this replica. - #[must_use] - pub const fn limits(&self) -> Limits { - self.limits - } - - /// Returns the cumulative immutable publication cost this replica paid. - /// - /// The ledger covers every object the replica uploaded: segment bodies and - /// indexes, directory nodes, root documents, segment pages, bundle bodies, - /// and compaction outputs. Callers sampling per-command cost use - /// [`Self::take_publication_cost`] instead. - #[must_use] - pub fn publication_cost(&self) -> PublicationCost { - self.cost.snapshot() - } - - /// Returns the publication cost accumulated since the last call and resets it. - /// - /// One Cell has a single publisher at a time, so the reset is safe there; - /// a caller that runs concurrent prepares must use - /// [`Self::publication_cost`] deltas instead. - #[must_use] - pub fn take_publication_cost(&self) -> PublicationCost { - self.cost.take() - } - - /// Selects the caller's bounded I/O and blocking execution facilities. - #[must_use] - pub fn with_host(mut self, host: Host) -> Self { - self.host = host; - self - } - - /// Exclusively creates a fresh local database using this replica's host and limits. - /// - /// The destination and SQLite sidecars must not exist. A failed open leaves - /// its artifacts quarantined for caller-owned inspection and cleanup. - pub fn open_new(&self, destination: &std::path::Path) -> Result { - crate::recovery::reject_sidecars(destination, &self.host)?; - let mut file = self.host.filesystem.create(destination)?; - file.sync_all()?; - self.host.filesystem.sync_parent(destination)?; - drop(file); - crate::Db::open_with_host(destination, self.limits, self.host.clone()) - } - - /// Moves a resumable database onto a fresh path and continues its capture. - /// - /// The caller owns the proof that the record it read still matches the - /// authoritative control: this path reads no origin object, so a foreign or - /// stale file would otherwise be served as though it held this replica's - /// root. A database that is not cleanly checkpointed is refused, and the - /// caller must fall back to restoring the exact root. - pub fn open_resumed( - &self, - source: &std::path::Path, - destination: &std::path::Path, - ) -> Result { - let host = self.host.clone().without_dirty(); - crate::resume::move_resumed(source, destination, &host)?; - crate::Db::open_resumed_with_host(destination, self.limits, host) - } - - /// Removes a database and its resume sidecars that this replica refused. - pub fn discard_resumed(&self, database: &std::path::Path) -> Result<()> { - let host = self.host.clone().without_dirty(); - crate::resume::discard_resumed(database, &host) - } -} - -fn scheduled_compaction_range( - descriptors: &[SegmentDescriptor], - level: u8, - max_file_bytes: u64, -) -> Option> { - let source_level = level.checked_sub(1)?; - let mut start = 0; - while start < descriptors.len() { - if descriptors[start].level() != source_level { - start += 1; - continue; - } - let mut end = start; - let mut bytes = 0_u64; - while end < descriptors.len() - && descriptors[end].level() == source_level - && end - start < MAX_COMPACTION_INPUTS - && bytes - .checked_add(descriptors[end].info.size_bytes) - .is_some_and(|next| next <= max_file_bytes) - { - bytes += descriptors[end].info.size_bytes; - end += 1; - } - if end - start >= COMPACTION_FANOUT { - return Some(start..end); - } - start = end.max(start + 1); - } - None -} - -struct LoadedGraph { - aggregate: directory::Aggregate, - document: RootDocument, - descriptors: Vec, -} - -fn compaction_scratch_bytes(graph: &LoadedGraph, range: std::ops::Range) -> Result { - let selected = graph - .descriptors - .get(range) - .filter(|descriptors| !descriptors.is_empty()) - .ok_or(CrabError::TxNotAvailable)?; - let indexes = selected.iter().try_fold(0_u64, |total, descriptor| { - total - .checked_add(descriptor.index_length) - .ok_or(CrabError::Limit(crate::LimitKind::ScratchDiskBytes)) - })?; - let inputs = selected.iter().try_fold(indexes, |total, descriptor| { - total - .checked_add(descriptor.info.size_bytes) - .ok_or(CrabError::Limit(crate::LimitKind::ScratchDiskBytes)) - })?; - let endpoint = selected.last().ok_or(CrabError::TxNotAvailable)?; - let pages = - (indexes / crate::paged::ENTRY_BYTES as u64).min(u64::from(endpoint.info.database_pages)); - // Five coexisting files: selected indexes/bodies, encoded LTX, its temporary - // varint index, and the fixed-width sidecar. Include worst-case compression - // and both index copies; logical database size is not a range-work bound. - let ltx = crate::ltx::cut_upper_bound(endpoint.info.page_size, pages)?; - pages - .checked_mul(30 + crate::paged::ENTRY_BYTES as u64) - .and_then(|output_indexes| output_indexes.checked_add(ltx)) - .and_then(|outputs| inputs.checked_add(outputs)) - .and_then(|total| total.checked_add(64 << 10)) - .ok_or(CrabError::Limit(crate::LimitKind::ScratchDiskBytes)) -} - -struct AppendInput { - info: crate::SegmentInfo, - location: BodyLocation, - index: Bytes, - body: AppendBody, -} - -enum AppendBody { - Native(Arc), - Bundle, -} - -#[derive(Clone, Copy)] -enum BodyLocation { - Native, - Bundle { digest: [u8; 32], offset: u64 }, -} - -struct PreparedSegment { - descriptor: SegmentDescriptor, - index: Bytes, - body: AppendBody, -} - -struct DirectoryInput { - descriptor: SegmentDescriptor, - index: Bytes, -} - -fn object_extents(descriptors: &[SegmentDescriptor]) -> Result> { - let mut extents = BTreeMap::new(); - for descriptor in descriptors { - let (digest, offset, length, kind) = descriptor.object_extent(); - let end = offset.checked_add(length).ok_or(CrabError::LTXCorrupted)?; - match extents.entry(digest) { - std::collections::btree_map::Entry::Vacant(entry) => { - entry.insert(ObjectExtent { - kind, - ranges: std::iter::once(offset..end).collect(), - }); - } - std::collections::btree_map::Entry::Occupied(mut entry) => { - let extent = entry.get_mut(); - if extent.kind != kind { - return Err(CrabError::LTXCorrupted); - } - if !extent - .ranges - .iter() - .any(|range| range.start == offset && range.end == end) - { - extent.ranges.push(offset..end); - } - } - } - } - for extent in extents.values_mut() { - extent.ranges.sort_by_key(|range| range.start); - if extent - .ranges - .windows(2) - .any(|ranges| ranges[0].end > ranges[1].start) - { - return Err(CrabError::LTXCorrupted); - } - } - Ok(extents) -} - -fn directory_changes( - prepared: &[DirectoryInput], - base_pages: u32, -) -> Result<(BTreeMap, u32)> { - let mut changes = BTreeMap::new(); - let mut retain_through = base_pages; - for prepared in prepared { - let info = &prepared.descriptor.info; - let index = &prepared.index; - // Once a cut truncates a page, later growth must provide a new frame; - // retaining its old locator would resurrect bytes from before truncation. - retain_through = retain_through.min(info.database_pages); - changes.retain(|page, _| *page <= info.database_pages); - for entry in crate::paged::decode_index(index)? { - changes.insert( - entry.page, - DirectoryEntry { - page: entry.page, - object: prepared.descriptor.object_digest(), - offset: prepared - .descriptor - .offset() - .checked_add(entry.offset) - .ok_or(CrabError::LTXCorrupted)?, - length: u32::try_from(entry.size).map_err(|_| CrabError::LTXCorrupted)?, - frame_hash: entry.hash, - checksum: entry.checksum, - }, - ); - } - } - Ok((changes, retain_through)) -} diff --git a/crates/crab-ltx/src/replica/cache.rs b/crates/crab-ltx/src/replica/cache.rs deleted file mode 100644 index 910149163..000000000 --- a/crates/crab-ltx/src/replica/cache.rs +++ /dev/null @@ -1,133 +0,0 @@ -use std::{ - collections::{HashMap, VecDeque}, - sync::{Arc, Mutex, OnceLock}, -}; - -use crate::{CellObjectKind, CellStorageLayout, CrabError, Result}; - -// Root metadata and directory nodes each have an independent process-wide cap. -const CACHE_BYTES: usize = 8 << 20; - -#[derive(Clone, Debug, PartialEq, Eq, Hash)] -struct Key { - store: u64, - path: String, - digest: [u8; 32], -} - -#[derive(Default)] -struct Cache { - objects: HashMap>, - order: VecDeque, - bytes: usize, -} - -impl Cache { - fn get(&self, key: &Key) -> Option> { - self.objects.get(key).cloned() - } - - fn insert(&mut self, key: Key, bytes: Arc<[u8]>) -> Arc<[u8]> { - if let Some(existing) = self.objects.get(&key) { - return Arc::clone(existing); - } - while self.bytes.saturating_add(bytes.len()) > CACHE_BYTES { - let Some(old) = self.order.pop_front() else { - break; - }; - if let Some(removed) = self.objects.remove(&old) { - self.bytes -= removed.len(); - } - } - self.bytes += bytes.len(); - self.order.push_back(key.clone()); - self.objects.insert(key, Arc::clone(&bytes)); - bytes - } -} - -fn band(kind: CellObjectKind) -> Result<&'static Mutex> { - static ROOTS: OnceLock> = OnceLock::new(); - static DIRECTORIES: OnceLock> = OnceLock::new(); - match kind { - CellObjectKind::Root => Ok(ROOTS.get_or_init(|| Mutex::new(Cache::default()))), - CellObjectKind::Directory => Ok(DIRECTORIES.get_or_init(|| Mutex::new(Cache::default()))), - _ => Err(CrabError::InvalidState("unsupported immutable cache kind")), - } -} - -fn key( - layout: &CellStorageLayout, - cell: &[u8; 32], - incarnation: &[u8; 16], - digest: [u8; 32], - kind: CellObjectKind, -) -> Key { - Key { - store: layout.immutable_cache_identity(), - path: layout - .incarnation_object_path(cell, incarnation, &digest, kind) - .to_string(), - digest, - } -} - -pub(super) fn get( - layout: &CellStorageLayout, - cell: &[u8; 32], - incarnation: &[u8; 16], - digest: [u8; 32], - kind: CellObjectKind, -) -> Result>> { - band(kind)? - .lock() - .map_err(|_| CrabError::InvalidState("immutable object cache poisoned")) - .map(|cache| cache.get(&key(layout, cell, incarnation, digest, kind))) -} - -// Callers insert only digest-verified bytes after a successful immutable upload -// or authenticated read; backup inventory bypasses this process cache. -pub(super) fn insert( - layout: &CellStorageLayout, - cell: &[u8; 32], - incarnation: &[u8; 16], - digest: [u8; 32], - kind: CellObjectKind, - bytes: Arc<[u8]>, -) -> Result> { - Ok(band(kind)? - .lock() - .map_err(|_| CrabError::InvalidState("immutable object cache poisoned"))? - .insert(key(layout, cell, incarnation, digest, kind), bytes)) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn cache_is_byte_bounded_and_store_isolated() { - let mut cache = Cache::default(); - for store in 0..10 { - let key = Key { - store, - path: "same/path.root".into(), - digest: [1; 32], - }; - cache.insert(key, vec![store as u8; 1 << 20].into()); - } - assert_eq!(cache.bytes, CACHE_BYTES); - assert_eq!(cache.objects.len(), 8); - assert!(!cache.objects.keys().any(|key| key.store < 2)); - for store in 2..10 { - assert_eq!( - cache.objects[&Key { - store, - path: "same/path.root".into(), - digest: [1; 32], - }][0], - store as u8 - ); - } - } -} diff --git a/crates/crab-ltx/src/replica/compaction.rs b/crates/crab-ltx/src/replica/compaction.rs deleted file mode 100644 index 68ad8bff7..000000000 --- a/crates/crab-ltx/src/replica/compaction.rs +++ /dev/null @@ -1,302 +0,0 @@ -use std::{ - io, - ops::Range, - path::{Path, PathBuf}, - sync::Arc, -}; - -use futures_util::{StreamExt as _, TryStreamExt as _, stream}; - -use super::merge::LocatorMerge; -use super::{ - CellReplica, DirectoryEntry, LoadedGraph, PreparedRoot, RootRef, SEGMENT_TRANSFER_CONCURRENCY, - SegmentDescriptor, directory, -}; -use crate::{CellObjectKind, CrabError, Result, SegmentInfo, Txid, environment::FileIo}; - -mod output; -mod source; - -pub(super) mod scratch; -use scratch::ScratchFiles; - -use output::*; -use scratch::upload; -pub(super) use scratch::upload_source; -use source::*; - -const INDEX_READ_BYTES: u64 = 60 * 8_192; -const FRAME_READ_BYTES: u64 = 1 << 20; - -pub(super) async fn prepare( - replica: &CellReplica, - base: &RootRef, - graph: LoadedGraph, - range: Range, - level: u8, - scratch_directory: &Path, -) -> Result { - let selected = graph - .descriptors - .get(range.clone()) - .filter(|segments| !segments.is_empty()) - .ok_or(CrabError::TxNotAvailable)?; - if !(1..=9).contains(&level) - || (level == 9 && (range.start != 0 || range.end != graph.descriptors.len())) - || (level < 9 && selected.iter().any(|segment| segment.level() >= level)) - { - return Err(CrabError::InvalidState("invalid compaction level or range")); - } - - let host = replica.host.clone(); - let directory = scratch_directory.to_owned(); - let runtime = tokio::runtime::Handle::try_current().map_err(io::Error::other)?; - let (cleaned, cleanup) = tokio::sync::oneshot::channel(); - let result = async { - let files = replica - .host - .run(move || { - let mut scratch = ScratchFiles::new(host, &directory, runtime, cleaned); - Ok::<_, CrabError>(CompactionFiles { - original_indexes: scratch.create("source-indexes")?, - original_bodies: scratch.create("source-bodies")?, - compacted_ltx: scratch.create("compacted-ltx")?, - codec_index: scratch.create("codec-index")?, - compacted_index: scratch.create("compacted-index")?, - scratch: Arc::new(scratch), - }) - }) - .await??; - prepare_root(replica, base, graph, range, level, files).await - } - .await; - // Ordinary completion includes cleanup. Cancellation leaves cleanup owned - // by the last dispatched file job instead of unlinking its inputs early. - let _ = cleanup.await; - result -} - -struct CompactionFiles { - original_indexes: PathBuf, - original_bodies: PathBuf, - compacted_ltx: PathBuf, - codec_index: PathBuf, - compacted_index: PathBuf, - scratch: Arc, -} - -async fn prepare_root( - replica: &CellReplica, - base: &RootRef, - graph: LoadedGraph, - range: Range, - level: u8, - files: CompactionFiles, -) -> Result { - let selected = &graph.descriptors[range.clone()]; - // The authenticated streams have separate scratch files. Both must finish - // before the merge, but neither depends on the other's transfer. - let (spooled, body_inputs) = futures_util::future::join( - spool_indexes(replica, selected, &files.scratch, &files.original_indexes), - spool_selected_bodies(replica, selected, &files.scratch, &files.original_bodies), - ) - .await; - let spooled = spooled?; - let body_inputs = body_inputs?; - let artifacts = write_compacted(replica, spooled, &body_inputs, &files).await?; - - let first = selected.first().ok_or(CrabError::TxNotAvailable)?; - let last = selected.last().ok_or(CrabError::TxNotAvailable)?; - let info = SegmentInfo { - min_txid: first.info.min_txid, - max_txid: last.info.max_txid, - page_size: first.info.page_size, - database_pages: last.info.database_pages, - pre_checksum: first.info.pre_checksum, - post_checksum: last.info.post_checksum, - size_bytes: artifacts.ltx.length, - blake3: artifacts.ltx.digest, - }; - let descriptor = - SegmentDescriptor::native(info, artifacts.index.digest, artifacts.index.length) - .with_level(level); - descriptor.validate_published(replica.limits)?; - - // These immutable objects are unreachable until the final root is returned, - // so either upload may finish first without publishing a partial compaction. - let (body_upload, index_upload) = futures_util::future::join( - upload( - replica, - &files.scratch, - &files.compacted_ltx, - artifacts.ltx.length, - &descriptor.info.blake3, - CellObjectKind::Ltx, - ), - upload( - replica, - &files.scratch, - &files.compacted_index, - artifacts.index.length, - &descriptor.index_digest, - CellObjectKind::Index, - ), - ) - .await; - body_upload?; - index_upload?; - - let mut descriptors = graph.descriptors.clone(); - descriptors.splice(range.clone(), [descriptor.clone()]); - replica.validate_chain(&descriptors, base.position)?; - - let compacted_source = files.scratch.open(&files.compacted_index).await?; - let compacted_input = SpoolInput { - descriptor, - start: 0, - length: artifacts.index.length, - }; - let entries = - MergedEntries::open(&replica.host, compacted_source, vec![compacted_input]).await?; - let endpoint = descriptors.last().ok_or(CrabError::LTXCorrupted)?; - let page_size = endpoint.info.page_size; - let database_pages = endpoint.info.database_pages; - let directory = directory::relocate_and_upload( - replica, - &graph, - &descriptors, - selected, - entries.stream(replica.host.clone()), - ) - .await?; - replica - .finish_root( - Some(base), - descriptors, - base.position, - base.commit_sequence, - graph.document.schema, - page_size, - database_pages, - directory, - ) - .await -} - -struct CompactedArtifacts { - ltx: Artifact, - index: Artifact, -} - -struct Artifact { - digest: [u8; 32], - length: u64, -} - -#[derive(Clone)] -struct SpoolInput { - descriptor: SegmentDescriptor, - start: u64, - length: u64, -} - -#[derive(Clone)] -struct BodySpoolInput { - descriptor: SegmentDescriptor, - start: u64, -} - -struct LocalBodyRange { - start: u64, - length: usize, -} - -struct MergedEntries { - source: Box, - cursors: Vec, - merge: LocatorMerge, -} - -impl MergedEntries { - async fn open( - host: &crate::Host, - mut source: Box, - inputs: Vec, - ) -> Result { - host.run(move || { - const TOTAL_BUFFER_ENTRIES: usize = 16_384; - let buffer_entries = TOTAL_BUFFER_ENTRIES - .checked_div(inputs.len()) - .filter(|entries| *entries > 0) - .ok_or(CrabError::LTXCorrupted)? - .min(1_024); - let mut merge = LocatorMerge::new( - inputs - .iter() - .map(|input| input.descriptor.info.database_pages), - ); - let mut cursors = Vec::with_capacity(inputs.len()); - for input in inputs { - let cursor = SpoolCursor::new(input, source.as_mut(), buffer_entries)?; - if let Some(entry) = &cursor.current { - merge.push(cursors.len(), entry.page); - } - cursors.push(cursor); - } - Ok(Self { - source, - cursors, - merge, - }) - }) - .await? - } - - fn next_batch(&mut self) -> Result<(Vec, bool)> { - let mut batch = Vec::new(); - let mut visited = 0; - // Stop between page groups, including discarded pages. One group - // visits at most the admitted descriptor count. - while visited < 4_096 { - let next = self.merge.next_group(|index| { - visited += 1; - let cursor = self.cursors.get_mut(index).ok_or(CrabError::LTXCorrupted)?; - let entry = cursor.current.take().ok_or(CrabError::LTXCorrupted)?; - cursor.advance(self.source.as_mut())?; - Ok((entry, cursor.current.as_ref().map(|entry| entry.page))) - }); - match next { - Some(Ok(Some(entry))) => batch.push(entry), - Some(Ok(None)) => {} - Some(Err(error)) => return Err(error), - None => return Ok((batch, true)), - } - } - Ok((batch, false)) - } - - fn stream(self, host: crate::Host) -> impl futures_util::Stream> { - stream::try_unfold(Some(self), move |entries| { - let host = host.clone(); - async move { - let Some(mut entries) = entries else { - return Ok::<_, CrabError>(None); - }; - let (entries, batch, finished) = host - .run(move || { - let (batch, finished) = entries.next_batch()?; - Ok::<_, CrabError>((entries, batch, finished)) - }) - .await??; - if finished && batch.is_empty() { - return Ok(None); - } - Ok(Some(( - stream::iter(batch.into_iter().map(Ok)), - (!finished).then_some(entries), - ))) - } - }) - .try_flatten() - } -} diff --git a/crates/crab-ltx/src/replica/compaction/output.rs b/crates/crab-ltx/src/replica/compaction/output.rs deleted file mode 100644 index a4ea81010..000000000 --- a/crates/crab-ltx/src/replica/compaction/output.rs +++ /dev/null @@ -1,275 +0,0 @@ -//! Compacted LTX output, sidecar index, and digests for one compaction pass. - -use super::*; - -pub(super) async fn write_compacted( - replica: &CellReplica, - inputs: Vec, - body_inputs: &[BodySpoolInput], - files: &CompactionFiles, -) -> Result { - let first = inputs - .first() - .ok_or(CrabError::TxNotAvailable)? - .descriptor - .info - .clone(); - let last = inputs - .last() - .ok_or(CrabError::TxNotAvailable)? - .descriptor - .info - .clone(); - let page_size = first.page_size; - let post_checksum = last.post_checksum; - let source = files.scratch.open(&files.original_indexes).await?; - let mut body_source = files.scratch.open(&files.original_bodies).await?; - let entries = MergedEntries::open(&replica.host, source, inputs) - .await? - .stream(replica.host.clone()); - futures_util::pin_mut!(entries); - let output_file = files.scratch.open(&files.compacted_ltx).await?; - let codec_index_file = files.scratch.open(&files.codec_index).await?; - let sidecar_file = files.scratch.open(&files.compacted_index).await?; - let limits = replica.limits; - let mut state = replica - .host - .run(move || { - OutputState::new( - output_file, - codec_index_file, - sidecar_file, - limits, - &first, - &last, - ) - }) - .await??; - let mut next = entries.try_next().await?; - while let Some(first_entry) = next.take() { - let mut encoded = u64::from(first_entry.length); - let mut batch = vec![first_entry]; - while let Some(entry) = entries.try_next().await? { - let previous = batch.last().ok_or(CrabError::LTXCorrupted)?; - let combined = encoded - .checked_add(u64::from(entry.length)) - .ok_or(CrabError::LTXCorrupted)?; - if entry.object != previous.object - || entry.offset != previous.offset + u64::from(previous.length) - || combined > FRAME_READ_BYTES - || batch.len() as u64 * u64::from(page_size) >= FRAME_READ_BYTES - { - next = Some(entry); - break; - } - encoded = combined; - batch.push(entry); - } - let range = body_range(&batch, body_inputs)?; - let returned = replica - .host - .run(move || { - let frames = body_source.read_exact_at(range.start, range.length)?; - let pages = decode_pages(&batch, &frames, page_size)?; - Ok::<_, CrabError>((body_source, state.encode(pages)?)) - }) - .await??; - body_source = returned.0; - state = returned.1; - } - replica - .host - .run(move || state.finish(post_checksum)) - .await? -} - -fn body_range(entries: &[DirectoryEntry], inputs: &[BodySpoolInput]) -> Result { - let first = entries.first().ok_or(CrabError::LTXCorrupted)?; - let last = entries.last().ok_or(CrabError::LTXCorrupted)?; - let end = last - .offset - .checked_add(u64::from(last.length)) - .ok_or(CrabError::LTXCorrupted)?; - if entries.iter().any(|entry| entry.object != first.object) { - return Err(CrabError::LTXCorrupted); - } - let input = inputs - .iter() - .find(|input| { - input.descriptor.object_digest() == first.object - && input.descriptor.offset() <= first.offset - && input - .descriptor - .offset() - .checked_add(input.descriptor.info.size_bytes) - .is_some_and(|input_end| end <= input_end) - }) - .ok_or(CrabError::LTXCorrupted)?; - let start = input - .start - .checked_add(first.offset - input.descriptor.offset()) - .ok_or(CrabError::LTXCorrupted)?; - let length = usize::try_from(end - first.offset).map_err(|_| CrabError::LTXCorrupted)?; - Ok(LocalBodyRange { start, length }) -} - -fn decode_pages( - entries: &[DirectoryEntry], - frames: &[u8], - page_size: u32, -) -> Result)>> { - let first = entries.first().ok_or(CrabError::LTXCorrupted)?; - let mut pages = Vec::with_capacity(entries.len()); - for entry in entries { - let start = - usize::try_from(entry.offset - first.offset).map_err(|_| CrabError::LTXCorrupted)?; - let frame = frames - .get(start..start + entry.length as usize) - .ok_or(CrabError::LTXCorrupted)?; - if *blake3::hash(frame).as_bytes() != entry.frame_hash { - return Err(CrabError::ChecksumMismatch); - } - let bytes = crate::paged::decode_frame(frame, page_size, entry.page)?; - if crate::ltx::checksum_page(entry.page, &bytes) != entry.checksum { - return Err(CrabError::ChecksumMismatch); - } - pages.push((entry.page, bytes)); - } - Ok(pages) -} - -struct OutputState { - encoder: crate::codec::Encoder>, - sidecar: io::BufWriter, -} - -impl OutputState { - fn new( - output: Box, - codec_index: Box, - sidecar: Box, - limits: crate::Limits, - first: &SegmentInfo, - last: &SegmentInfo, - ) -> Result { - if first.page_size != last.page_size { - return Err(CrabError::LTXCorrupted); - } - let output = DigestWriter::new(output, limits.max_file_bytes); - let mut encoder = crate::codec::Encoder::new_block_with_index( - io::BufWriter::with_capacity(64 << 10, output), - Some(codec_index), - ); - encoder.encode_header(crate::ltx::Header { - version: crate::ltx::VERSION, - flags: 0, - page_size: first.page_size, - commit: last.database_pages, - min_txid: Txid(first.min_txid), - max_txid: Txid(last.max_txid), - timestamp: 0, - pre_apply_checksum: first.pre_checksum, - ..crate::ltx::Header::default() - })?; - Ok(Self { - encoder, - sidecar: io::BufWriter::with_capacity( - 64 << 10, - DigestWriter::new(sidecar, limits.max_plan_bytes), - ), - }) - } - - fn encode(mut self, pages: Vec<(u32, Vec)>) -> Result { - for (page, bytes) in pages { - let encoded = self.encoder.encode_page( - crate::ltx::PageHeader { - pgno: page, - flags: 0, - }, - &bytes, - )?; - write_sidecar_entry(&mut self.sidecar, &encoded)?; - } - Ok(self) - } - - fn finish(mut self, post_checksum: u64) -> Result { - self.encoder.close(post_checksum)?; - // Flush both streams before DigestWriter checks the exact stored length - // and syncs; a partial buffered output must never become publishable. - let ltx = self - .encoder - .into_writer() - .into_inner() - .map_err(|error| error.into_error())? - .finish()?; - let index = self - .sidecar - .into_inner() - .map_err(|error| error.into_error())? - .finish()?; - Ok(CompactedArtifacts { ltx, index }) - } -} - -fn write_sidecar_entry( - writer: &mut impl io::Write, - page: &crate::codec::EncodedPage, -) -> Result<()> { - writer.write_all(&page.page.to_be_bytes())?; - writer.write_all(&page.offset.to_be_bytes())?; - writer.write_all(&page.size.to_be_bytes())?; - writer.write_all(&page.frame_hash)?; - writer.write_all(&page.checksum.to_be_bytes())?; - Ok(()) -} - -struct DigestWriter { - file: Box, - hasher: blake3::Hasher, - length: u64, - limit: u64, -} - -impl DigestWriter { - fn new(file: Box, limit: u64) -> Self { - Self { - file, - hasher: blake3::Hasher::new(), - length: 0, - limit, - } - } - - fn finish(mut self) -> Result { - if self.file.file_len()? != self.length { - return Err(CrabError::LTXCorrupted); - } - self.file.sync_all()?; - Ok(Artifact { - digest: *self.hasher.finalize().as_bytes(), - length: self.length, - }) - } -} - -impl io::Write for DigestWriter { - fn write(&mut self, bytes: &[u8]) -> io::Result { - let next = self - .length - .checked_add(bytes.len() as u64) - .ok_or_else(|| io::Error::other("compaction output length overflow"))?; - if next > self.limit { - return Err(io::Error::other("compaction output limit exceeded")); - } - self.file.write_all(bytes)?; - self.hasher.update(bytes); - self.length = next; - Ok(bytes.len()) - } - - fn flush(&mut self) -> io::Result<()> { - Ok(()) - } -} diff --git a/crates/crab-ltx/src/replica/compaction/scratch.rs b/crates/crab-ltx/src/replica/compaction/scratch.rs deleted file mode 100644 index 481b25770..000000000 --- a/crates/crab-ltx/src/replica/compaction/scratch.rs +++ /dev/null @@ -1,198 +0,0 @@ -use std::{io, path::Path, path::PathBuf, sync::Arc}; - -use bytes::Bytes; -use crab_storage::{MultipartUploadSource, StorageError}; - -use super::CellReplica; -use crate::{CellObjectKind, CrabError, Host, Result, environment::FileIo}; - -pub(super) async fn upload( - replica: &CellReplica, - scratch: &Arc, - source: &Path, - size: u64, - digest: &[u8; 32], - kind: CellObjectKind, -) -> Result<()> { - let upload: Arc = Arc::new(HostUploadSource { - scratch: Arc::clone(scratch), - path: source.to_owned(), - }); - upload_source(replica, upload, size, digest, kind).await -} - -pub(crate) async fn upload_source( - replica: &CellReplica, - upload: Arc, - size: u64, - digest: &[u8; 32], - kind: CellObjectKind, -) -> Result<()> { - let path = - replica - .layout - .incarnation_object_path(&replica.cell, &replica.incarnation, digest, kind); - let _permit = replica.host.io_permit().await?; - super::super::upload::put_source(replica, &path, upload, size, *digest).await?; - replica.cost.record(size); - Ok(()) -} - -struct HostUploadSource { - scratch: Arc, - path: PathBuf, -} - -#[async_trait::async_trait] -impl MultipartUploadSource for HostUploadSource { - async fn byte_len(&self) -> crab_storage::Result { - let scratch = Arc::clone(&self.scratch); - let path = self.path.clone(); - self.scratch - .host - .run(move || scratch.host.filesystem.file_len(&path)) - .await - .map_err(storage_read_error)? - .map_err(|error| StorageError::ReadRejected { - source: Box::new(error), - }) - } - - async fn read_exact(&self, offset: u64, length: usize) -> crab_storage::Result { - let scratch = Arc::clone(&self.scratch); - let path = self.path.clone(); - self.scratch - .host - .run(move || { - let mut file = scratch.host.filesystem.open(&path)?; - file.read_exact_at(offset, length).map(Bytes::from) - }) - .await - .map_err(storage_read_error)? - .map_err(|error| StorageError::ReadRejected { - source: Box::new(error), - }) - } -} - -fn storage_read_error(error: CrabError) -> StorageError { - StorageError::ReadRejected { - source: Box::new(error), - } -} - -pub(super) struct ScratchFiles { - pub(super) host: Host, - runtime: tokio::runtime::Handle, - cleaned: Option>, - directory: PathBuf, - paths: Vec, -} - -impl ScratchFiles { - pub(super) fn new( - host: Host, - directory: &Path, - runtime: tokio::runtime::Handle, - cleaned: tokio::sync::oneshot::Sender<()>, - ) -> Self { - Self { - host, - runtime, - cleaned: Some(cleaned), - directory: directory.to_owned(), - paths: Vec::new(), - } - } - - pub(super) fn create(&mut self, label: &str) -> Result { - static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); - for _ in 0..16 { - let path = self.directory.join(format!( - ".crab-compaction-{}-{}-{label}", - std::process::id(), - NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed) - )); - match self.host.filesystem.create(&path) { - Ok(file) => { - drop(file); - self.paths.push(path.clone()); - return Ok(path); - } - Err(error) if error.kind() == io::ErrorKind::AlreadyExists => continue, - Err(error) => return Err(error.into()), - } - } - Err(io::Error::new( - io::ErrorKind::AlreadyExists, - "compaction scratch namespace exhausted", - ) - .into()) - } - - pub(super) async fn open(self: &Arc, path: &Path) -> Result> { - let scratch = Arc::clone(self); - let path = path.to_owned(); - self.host - .run(move || { - let file = scratch.host.filesystem.open_rw(&path)?; - Ok(Box::new(ScratchFile { - file, - _scratch: scratch, - }) as Box) - }) - .await? - } -} - -// The file drops before its scratch owner. Dispatched jobs can outlive a -// canceled caller without closing admission or unlinking files they still use. -struct ScratchFile { - file: Box, - _scratch: Arc, -} - -impl FileIo for ScratchFile { - fn write_all(&mut self, bytes: &[u8]) -> io::Result<()> { - self.file.write_all(bytes) - } - fn write_all_at(&mut self, offset: u64, bytes: &[u8]) -> io::Result<()> { - self.file.write_all_at(offset, bytes) - } - fn read_exact_at(&mut self, offset: u64, len: usize) -> io::Result> { - self.file.read_exact_at(offset, len) - } - fn sync_all(&mut self) -> io::Result<()> { - self.file.sync_all() - } - fn file_len(&self) -> io::Result { - self.file.file_len() - } - fn set_len(&mut self, len: u64) -> io::Result<()> { - self.file.set_len(len) - } -} - -impl Drop for ScratchFiles { - fn drop(&mut self) { - let paths = std::mem::take(&mut self.paths); - let host = self.host.clone(); - let cleaned = self.cleaned.take(); - // Keep dirty/recovery/scratch admission until the last file and queued - // job have finished, then remove files through the same job ceiling. - self.runtime.spawn(async move { - let filesystem = Arc::clone(&host.filesystem); - let _ = host - .run(move || { - for path in paths { - let _ = filesystem.remove_file(&path); - } - }) - .await; - drop(host); - if let Some(cleaned) = cleaned { - let _ = cleaned.send(()); - } - }); - } -} diff --git a/crates/crab-ltx/src/replica/compaction/source.rs b/crates/crab-ltx/src/replica/compaction/source.rs deleted file mode 100644 index 57bba07a5..000000000 --- a/crates/crab-ltx/src/replica/compaction/source.rs +++ /dev/null @@ -1,251 +0,0 @@ -//! Source spooling for one compaction pass. - -use super::*; - -pub(super) async fn spool_selected_bodies( - replica: &CellReplica, - descriptors: &[SegmentDescriptor], - scratch: &Arc, - destination: &Path, -) -> Result> { - let mut planned = Vec::with_capacity(descriptors.len()); - let mut total_bytes = 0_u64; - for descriptor in descriptors { - let start = total_bytes; - total_bytes = total_bytes - .checked_add(descriptor.info.size_bytes) - .ok_or(CrabError::Limit(crate::LimitKind::CompactionBodySpool))?; - planned.push((descriptor.clone(), start)); - } - - let results = stream::iter( - planned - .into_iter() - .map(|(descriptor, output_start)| async move { - let mut file = scratch.open(destination).await?; - let source_start = descriptor.offset(); - let source_end = source_start - .checked_add(descriptor.info.size_bytes) - .ok_or(CrabError::LTXCorrupted)?; - let path = replica.layout.incarnation_object_path( - &replica.cell, - &replica.incarnation, - &descriptor.object_digest(), - descriptor.object_kind(), - ); - let mut source_offset = source_start; - let mut hasher = blake3::Hasher::new(); - while source_offset < source_end { - let next = source_offset - .checked_add((source_end - source_offset).min(FRAME_READ_BYTES)) - .ok_or(CrabError::LTXCorrupted)?; - let _permit = replica.host.io_permit().await?; - let bytes = replica - .layout - .store() - .range_get(&path, source_offset..next) - .await?; - drop(_permit); - if bytes.len() as u64 != next - source_offset { - return Err(CrabError::ChecksumMismatch); - } - hasher.update(&bytes); - let output_offset = output_start - .checked_add(source_offset - source_start) - .ok_or(CrabError::Limit(crate::LimitKind::CompactionBodySpool))?; - file = replica - .host - .run(move || { - file.write_all_at(output_offset, &bytes)?; - Ok::<_, CrabError>(file) - }) - .await??; - source_offset = next; - } - if *hasher.finalize().as_bytes() != descriptor.info.blake3 { - return Err(CrabError::ChecksumMismatch); - } - Ok(BodySpoolInput { - descriptor, - start: output_start, - }) - }), - ) - .buffered(SEGMENT_TRANSFER_CONCURRENCY) - .collect::>() - .await; - let spooled = results.into_iter().collect::>>()?; - sync_spool(scratch, destination, total_bytes).await?; - Ok(spooled) -} - -pub(super) async fn spool_indexes( - replica: &CellReplica, - descriptors: &[SegmentDescriptor], - scratch: &Arc, - destination: &Path, -) -> Result> { - let mut planned = Vec::with_capacity(descriptors.len()); - let mut total_bytes = 0_u64; - for descriptor in descriptors { - if descriptor.index_length == 0 - || descriptor.index_length % crate::paged::ENTRY_BYTES as u64 != 0 - { - return Err(CrabError::LTXCorrupted); - } - let start = total_bytes; - total_bytes = total_bytes - .checked_add(descriptor.index_length) - .ok_or(CrabError::Limit(crate::LimitKind::CompactionIndexSpool))?; - planned.push((descriptor.clone(), start)); - } - - let results = stream::iter( - planned - .into_iter() - .map(|(descriptor, output_start)| async move { - let mut file = scratch.open(destination).await?; - let path = replica.layout.incarnation_object_path( - &replica.cell, - &replica.incarnation, - &descriptor.index_digest, - CellObjectKind::Index, - ); - let mut source_offset = 0_u64; - let mut hasher = blake3::Hasher::new(); - let mut validator = crate::paged::IndexValidator::new(&descriptor.info); - while source_offset < descriptor.index_length { - let length = (descriptor.index_length - source_offset).min(INDEX_READ_BYTES); - let _permit = replica.host.io_permit().await?; - let bytes = replica - .layout - .store() - .range_get(&path, source_offset..source_offset + length) - .await?; - drop(_permit); - if bytes.len() as u64 != length { - return Err(CrabError::ChecksumMismatch); - } - hasher.update(&bytes); - let output_offset = output_start - .checked_add(source_offset) - .ok_or(CrabError::Limit(crate::LimitKind::CompactionIndexSpool))?; - let returned = replica - .host - .run(move || { - for entry in bytes.as_chunks::<{ crate::paged::ENTRY_BYTES }>().0 { - validator.validate(crate::paged::decode_index_entry(entry)?)?; - } - file.write_all_at(output_offset, &bytes)?; - Ok::<_, CrabError>((file, validator)) - }) - .await??; - file = returned.0; - validator = returned.1; - source_offset += length; - } - if *hasher.finalize().as_bytes() != descriptor.index_digest { - return Err(CrabError::ChecksumMismatch); - } - let length = descriptor.index_length; - Ok(SpoolInput { - descriptor, - start: output_start, - length, - }) - }), - ) - .buffered(SEGMENT_TRANSFER_CONCURRENCY) - .collect::>() - .await; - let inputs = results.into_iter().collect::>>()?; - sync_spool(scratch, destination, total_bytes).await?; - Ok(inputs) -} - -async fn sync_spool(scratch: &Arc, path: &Path, expected_bytes: u64) -> Result<()> { - let mut file = scratch.open(path).await?; - scratch - .host - .run(move || { - if file.file_len()? != expected_bytes { - return Err(CrabError::LTXCorrupted); - } - file.sync_all()?; - Ok::<_, CrabError>(()) - }) - .await??; - Ok(()) -} - -pub(super) struct SpoolCursor { - input: SpoolInput, - offset: u64, - buffer: Vec, - buffered_offset: usize, - buffer_bytes: usize, - pub(super) current: Option, -} - -impl SpoolCursor { - pub(super) fn new( - input: SpoolInput, - source: &mut dyn FileIo, - buffer_entries: usize, - ) -> Result { - let mut cursor = Self { - offset: input.start, - buffer: Vec::new(), - buffered_offset: 0, - buffer_bytes: buffer_entries * crate::paged::ENTRY_BYTES, - input, - current: None, - }; - cursor.advance(source)?; - Ok(cursor) - } - - pub(super) fn advance(&mut self, source: &mut dyn FileIo) -> Result<()> { - let end = self - .input - .start - .checked_add(self.input.length) - .ok_or(CrabError::LTXCorrupted)?; - if self.offset == end { - self.current = None; - return Ok(()); - } - if self.offset > end || end - self.offset < crate::paged::ENTRY_BYTES as u64 { - return Err(CrabError::LTXCorrupted); - } - if self.buffered_offset == self.buffer.len() { - let length = (end - self.offset).min(self.buffer_bytes as u64) as usize; - self.buffer = source.read_exact_at(self.offset, length)?; - if self.buffer.len() != length { - return Err(CrabError::LTXCorrupted); - } - self.buffered_offset = 0; - } - let next = self.buffered_offset + crate::paged::ENTRY_BYTES; - let bytes = self - .buffer - .get(self.buffered_offset..next) - .ok_or(CrabError::LTXCorrupted)?; - let entry = crate::paged::decode_index_entry(bytes)?; - self.buffered_offset = next; - let descriptor = &self.input.descriptor; - self.current = Some(DirectoryEntry { - page: entry.page, - object: descriptor.object_digest(), - offset: descriptor - .offset() - .checked_add(entry.offset) - .ok_or(CrabError::LTXCorrupted)?, - length: u32::try_from(entry.size).map_err(|_| CrabError::LTXCorrupted)?, - frame_hash: entry.hash, - checksum: entry.checksum, - }); - self.offset += crate::paged::ENTRY_BYTES as u64; - Ok(()) - } -} diff --git a/crates/crab-ltx/src/replica/directory.rs b/crates/crab-ltx/src/replica/directory.rs deleted file mode 100644 index 2e54b4088..000000000 --- a/crates/crab-ltx/src/replica/directory.rs +++ /dev/null @@ -1,905 +0,0 @@ -use std::{collections::BTreeMap, sync::Arc}; - -use crate::{CellObjectKind, CellStorageLayout, CrabError, Host, Result}; - -mod checksums; -mod initial; -mod relocate; -mod update; - -pub(super) use checksums::load_checksums; -pub(super) use initial::{ - build_and_upload as build_initial_and_upload, entries as initial_entries, -}; -pub(super) use relocate::run as relocate_and_upload; - -const MAGIC: &[u8; 8] = b"CRBDIR01"; -const HEADER_BYTES: usize = 32; -const LEAF_RECORD_BYTES: usize = 88; -const BRANCH_RECORD_BYTES: usize = 56; -const FANOUT: usize = 256; -const MAX_NODE_BYTES: u64 = (HEADER_BYTES + FANOUT * LEAF_RECORD_BYTES) as u64; - -#[derive(Clone)] -pub(super) struct DirectoryEntry { - pub page: u32, - pub object: [u8; 32], - pub offset: u64, - pub length: u32, - pub frame_hash: [u8; 32], - pub checksum: u64, -} - -pub(super) struct DirectorySpan { - pub object: [u8; 32], - pub start: u64, - pub end: u64, - pub entries: Vec, -} - -pub(super) struct DirectoryObject { - pub digest: [u8; 32], - pub bytes: Vec, -} - -pub(super) struct DirectoryTree { - objects: Vec, - root: Node, - height: u32, -} - -impl DirectoryTree { - #[cfg(test)] - pub(super) fn build( - entries: BTreeMap, - page_size: u32, - database_pages: u32, - ) -> Result { - if entries.is_empty() || database_pages == 0 { - return Err(CrabError::LTXCorrupted); - } - let lock = crate::ltx::lock_pgno(page_size); - let expected = u64::from(database_pages) - u64::from(lock <= database_pages); - if entries.len() as u64 != expected - || entries - .keys() - .any(|page| *page == 0 || *page > database_pages || *page == lock) - { - return Err(CrabError::LTXCorrupted); - } - - let mut objects = Vec::new(); - let mut groups: BTreeMap> = BTreeMap::new(); - for entry in entries.into_values() { - groups - .entry((entry.page - 1) / FANOUT as u32) - .or_default() - .push(entry); - } - let mut nodes = groups - .into_iter() - .map(|(index, entries)| { - let bytes = encode_leaf(&entries)?; - let node = Node::from_bytes(index, &bytes)?; - objects.push(DirectoryObject { - digest: node.digest, - bytes, - }); - Ok(node) - }) - .collect::>>()?; - let mut height = 0; - while nodes.len() > 1 { - let mut parents = Vec::new(); - for group in nodes - .chunk_by(|left, right| left.index / FANOUT as u32 == right.index / FANOUT as u32) - { - let index = group[0].index / FANOUT as u32; - let bytes = encode_branch(group)?; - let node = Node::from_bytes(index, &bytes)?; - objects.push(DirectoryObject { - digest: node.digest, - bytes, - }); - parents.push(node); - } - nodes = parents; - height += 1; - } - let root = nodes.pop().ok_or(CrabError::LTXCorrupted)?; - Ok(Self { - objects, - root, - height, - }) - } - - pub(super) fn objects(&self) -> &[DirectoryObject] { - &self.objects - } - - pub(super) fn root_digest(&self) -> [u8; 32] { - self.root.digest - } - - pub(super) fn height(&self) -> u32 { - self.height - } - - pub(super) fn checksum(&self) -> u64 { - self.root.aggregate.checksum - } - - pub(super) async fn update( - base: Verification<'_>, - root: [u8; 32], - height: u32, - aggregate: Aggregate, - changes: BTreeMap, - retain_through: u32, - final_state: Verification<'_>, - expected_checksum: u64, - ) -> Result { - update::run( - base, - root, - height, - aggregate, - changes, - retain_through, - final_state, - expected_checksum, - ) - .await - } -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub(super) struct Aggregate { - pub live_pages: u64, - pub checksum: u64, - first: u32, - last: u32, -} - -#[derive(Clone)] -struct Node { - index: u32, - digest: [u8; 32], - aggregate: Aggregate, -} - -impl Node { - fn from_bytes(index: u32, bytes: &[u8]) -> Result { - let header = Header::parse(bytes)?; - let (first, last) = if header.kind == 0 { - let first = read_u32(bytes, HEADER_BYTES)?; - let last = read_u32( - bytes, - HEADER_BYTES + (header.entries as usize - 1) * LEAF_RECORD_BYTES, - )?; - (first, last) - } else { - let first = read_u32(bytes, HEADER_BYTES)?; - let last = read_u32( - bytes, - HEADER_BYTES - + (header.entries as usize - 1) * BRANCH_RECORD_BYTES - + size_of::(), - )?; - (first, last) - }; - Ok(Self { - index, - digest: *blake3::hash(bytes).as_bytes(), - aggregate: Aggregate { - live_pages: header.live_pages, - checksum: header.checksum, - first, - last, - }, - }) - } -} - -struct Header { - kind: u8, - entries: u32, - live_pages: u64, - checksum: u64, -} - -impl Header { - fn parse(bytes: &[u8]) -> Result { - let prefix = bytes.get(..HEADER_BYTES).ok_or(CrabError::LTXCorrupted)?; - if &prefix[..8] != MAGIC - || u16::from_be_bytes(array(&prefix[8..10])?) != 1 - || prefix[10] > 1 - || prefix[11] != 0 - { - return Err(CrabError::LTXCorrupted); - } - let entries = u32::from_be_bytes(array(&prefix[12..16])?); - if entries == 0 || entries > FANOUT as u32 { - return Err(CrabError::LTXCorrupted); - } - let record_bytes = if prefix[10] == 0 { - LEAF_RECORD_BYTES - } else { - BRANCH_RECORD_BYTES - }; - if bytes.len() != HEADER_BYTES + entries as usize * record_bytes { - return Err(CrabError::LTXCorrupted); - } - Ok(Self { - kind: prefix[10], - entries, - live_pages: u64::from_be_bytes(array(&prefix[16..24])?), - checksum: u64::from_be_bytes(array(&prefix[24..32])?), - }) - } -} - -#[derive(Clone, Copy)] -pub(super) struct Verification<'a> { - pub layout: &'a CellStorageLayout, - pub cell: &'a [u8; 32], - pub incarnation: &'a [u8; 16], - pub page_size: u32, - pub database_pages: u32, - pub extents: &'a BTreeMap<[u8; 32], ObjectExtent>, - pub host: &'a Host, - pub origin: crate::LtxReadOrigin, -} - -#[derive(Clone)] -pub(super) struct ObjectExtent { - pub kind: CellObjectKind, - pub ranges: Vec>, -} - -pub(super) async fn verify_root( - verification: Verification<'_>, - root: [u8; 32], - height: u32, -) -> Result { - if height > 3 || verification.database_pages == 0 { - return Err(CrabError::LTXCorrupted); - } - let bytes = read_node(&verification, root).await?; - let header = Header::parse(&bytes)?; - if (height == 0) != (header.kind == 0) { - return Err(CrabError::LTXCorrupted); - } - let root_aggregate = if header.kind == 0 { - verify_leaf( - &bytes, - &header, - verification.page_size, - verification.database_pages, - verification.extents, - )? - .0 - } else { - verify_branch(&bytes, &header)?.0 - }; - let lock = crate::ltx::lock_pgno(verification.page_size); - let expected_pages = - u64::from(verification.database_pages) - u64::from(lock <= verification.database_pages); - if root_aggregate.live_pages != expected_pages - || root_aggregate.first == 0 - || root_aggregate.last > verification.database_pages - { - return Err(CrabError::ChecksumMismatch); - } - Ok(root_aggregate) -} - -pub(super) async fn reachable_digests( - verification: Verification<'_>, - root: [u8; 32], - height: u32, - expected_root: Aggregate, -) -> Result> { - if height > 3 || verification.database_pages == 0 { - return Err(CrabError::LTXCorrupted); - } - let mut pending = vec![(root, height, Some(expected_root))]; - let mut digests = Vec::new(); - let mut previous_page = 0u32; - let mut seen = 0u64; - let mut checksum = crate::CHECKSUM_FLAG; - while let Some((digest, remaining, expected)) = pending.pop() { - // Backup/collection inventory must prove the origin still contains - // every dependency; a process cache cannot establish remote presence. - let bytes = read_node_uncached(&verification, digest).await?; - let header = Header::parse(&bytes)?; - if (remaining == 0) != (header.kind == 0) { - return Err(CrabError::LTXCorrupted); - } - digests.push(digest); - if header.kind == 0 { - let (aggregate, entries) = verify_leaf( - &bytes, - &header, - verification.page_size, - verification.database_pages, - verification.extents, - )?; - if expected.is_some_and(|value| value != aggregate) { - return Err(CrabError::ChecksumMismatch); - } - for entry in entries { - let expected_page = previous_page - .checked_add(1) - .ok_or(CrabError::LTXCorrupted)?; - let lock = crate::ltx::lock_pgno(verification.page_size); - if expected_page == lock { - previous_page = lock; - } - if entry.page - != previous_page - .checked_add(1) - .ok_or(CrabError::LTXCorrupted)? - { - return Err(CrabError::LTXCorrupted); - } - checksum = crate::CHECKSUM_FLAG | (checksum ^ entry.checksum); - previous_page = entry.page; - seen += 1; - } - continue; - } - let (aggregate, children) = verify_branch(&bytes, &header)?; - if expected.is_some_and(|value| value != aggregate) { - return Err(CrabError::ChecksumMismatch); - } - let next = remaining.checked_sub(1).ok_or(CrabError::LTXCorrupted)?; - pending.extend( - children - .into_iter() - .rev() - .map(|child| (child.digest, next, Some(child.aggregate))), - ); - } - let lock = crate::ltx::lock_pgno(verification.page_size); - if previous_page < verification.database_pages { - if previous_page - .checked_add(1) - .ok_or(CrabError::LTXCorrupted)? - != lock - || lock != verification.database_pages - { - return Err(CrabError::LTXCorrupted); - } - previous_page = lock; - } - let expected_pages = - u64::from(verification.database_pages) - u64::from(lock <= verification.database_pages); - if previous_page != verification.database_pages - || seen != expected_pages - || seen != expected_root.live_pages - || checksum != expected_root.checksum - { - return Err(CrabError::ChecksumMismatch); - } - Ok(digests) -} - -pub(super) async fn lookup( - verification: Verification<'_>, - root: [u8; 32], - mut height: u32, - page: u32, -) -> Result { - if page == 0 || page > verification.database_pages { - return Err(CrabError::TxNotAvailable); - } - let mut digest = root; - let mut expected = None; - loop { - let bytes = read_node(&verification, digest).await?; - let header = Header::parse(&bytes)?; - if (height == 0) != (header.kind == 0) { - return Err(CrabError::LTXCorrupted); - } - if header.kind == 0 { - let (aggregate, entries) = verify_leaf( - &bytes, - &header, - verification.page_size, - verification.database_pages, - verification.extents, - )?; - if expected.is_some_and(|value| value != aggregate) { - return Err(CrabError::ChecksumMismatch); - } - return entries - .into_iter() - .find(|entry| entry.page == page) - .ok_or(CrabError::LTXCorrupted); - } - let (aggregate, children) = verify_branch(&bytes, &header)?; - if expected.is_some_and(|value| value != aggregate) { - return Err(CrabError::ChecksumMismatch); - } - let child = children - .into_iter() - .find(|child| child.aggregate.first <= page && page <= child.aggregate.last) - .ok_or(CrabError::LTXCorrupted)?; - digest = child.digest; - expected = Some(child.aggregate); - height = height.checked_sub(1).ok_or(CrabError::LTXCorrupted)?; - } -} - -/// Returns one contiguous, authenticated directory run without re-walking the -/// radix path for every page in the window. -pub(super) async fn lookup_run( - verification: Verification<'_>, - root: [u8; 32], - height: u32, - first: u32, - max_pages: u32, -) -> Result> { - if max_pages == 0 || first == 0 || first > verification.database_pages { - return Ok(Vec::new()); - } - let lock = crate::ltx::lock_pgno(verification.page_size); - let requested_last = first - .checked_add(max_pages - 1) - .ok_or(CrabError::LTXCorrupted)? - .min(verification.database_pages); - let last = if (first..=requested_last).contains(&lock) { - lock.checked_sub(1).ok_or(CrabError::LTXCorrupted)? - } else { - requested_last - }; - if last < first { - return Ok(Vec::new()); - } - - let mut pending = vec![(root, height, None)]; - let mut entries = Vec::new(); - while let Some((digest, remaining, expected)) = pending.pop() { - let bytes = read_node(&verification, digest).await?; - let header = Header::parse(&bytes)?; - if (remaining == 0) != (header.kind == 0) { - return Err(CrabError::LTXCorrupted); - } - if header.kind == 0 { - let (aggregate, leaf_entries) = verify_leaf( - &bytes, - &header, - verification.page_size, - verification.database_pages, - verification.extents, - )?; - if expected.is_some_and(|value| value != aggregate) { - return Err(CrabError::ChecksumMismatch); - } - entries.extend( - leaf_entries - .into_iter() - .filter(|entry| (first..=last).contains(&entry.page)), - ); - continue; - } - - let (aggregate, children) = verify_branch(&bytes, &header)?; - if expected.is_some_and(|value| value != aggregate) { - return Err(CrabError::ChecksumMismatch); - } - let next = remaining.checked_sub(1).ok_or(CrabError::LTXCorrupted)?; - pending.extend( - children - .into_iter() - .rev() - .filter(|child| child.aggregate.last >= first && child.aggregate.first <= last) - .map(|child| (child.digest, next, Some(child.aggregate))), - ); - } - - let expected_entries = - usize::try_from(last - first + 1).map_err(|_| CrabError::LTXCorrupted)?; - if entries.len() != expected_entries { - return Err(CrabError::LTXCorrupted); - } - for (index, entry) in entries.iter().enumerate() { - let expected_page = first - .checked_add(u32::try_from(index).map_err(|_| CrabError::LTXCorrupted)?) - .ok_or(CrabError::LTXCorrupted)?; - if entry.page != expected_page { - return Err(CrabError::LTXCorrupted); - } - } - Ok(entries) -} - -/// Returns every authenticated same-object span in one fixed page window. -pub(super) async fn lookup_spans( - verification: Verification<'_>, - root: [u8; 32], - height: u32, - first: u32, - max_pages: u32, -) -> Result> { - let entries = lookup_run(verification, root, height, first, max_pages).await?; - let mut spans: Vec = Vec::new(); - for entry in entries { - let end = entry - .offset - .checked_add(u64::from(entry.length)) - .ok_or(CrabError::LTXCorrupted)?; - if let Some(span) = spans.last_mut() - && span.object == entry.object - && span.end == entry.offset - { - span.end = end; - span.entries.push(entry); - continue; - } - spans.push(DirectorySpan { - object: entry.object, - start: entry.offset, - end, - entries: vec![entry], - }); - } - Ok(spans) -} - -async fn read_node(verification: &Verification<'_>, digest: [u8; 32]) -> Result> { - let path = verification.layout.incarnation_object_path( - verification.cell, - verification.incarnation, - &digest, - CellObjectKind::Directory, - ); - if let Some(bytes) = super::cache::get( - verification.layout, - verification.cell, - verification.incarnation, - digest, - CellObjectKind::Directory, - )? { - return Ok(bytes); - } - let persistent_key = format!( - "v1:{}:{}", - path, - digest - .iter() - .map(|byte| format!("{byte:02x}")) - .collect::() - ); - if let Some(bytes) = verification - .host - .directory_cache_get(persistent_key.clone(), MAX_NODE_BYTES) - .await? - { - if bytes.len() <= MAX_NODE_BYTES as usize && *blake3::hash(&bytes).as_bytes() == digest { - let bytes: Arc<[u8]> = bytes.into(); - return super::cache::insert( - verification.layout, - verification.cell, - verification.incarnation, - digest, - CellObjectKind::Directory, - bytes, - ); - } - let _ = verification - .host - .directory_cache_invalidate(persistent_key.clone()) - .await; - } - let permit = verification.host.io_permit().await?; - let result = verification - .layout - .store() - .get_with_etag_bounded(&path, MAX_NODE_BYTES) - .await; - verification.host.observe_ltx_origin_request( - verification.origin, - result.is_ok(), - result.as_ref().map_or(0, |(bytes, _)| bytes.len()), - ); - let (bytes, _) = result?; - if *blake3::hash(&bytes).as_bytes() != digest { - return Err(CrabError::ChecksumMismatch); - } - // Verified bytes no longer need origin admission. Cache fills admit their - // own bounded job without queuing, so slow disk cannot occupy network slots. - drop(permit); - let _ = verification - .host - .directory_cache_put(persistent_key, bytes.to_vec(), MAX_NODE_BYTES); - let bytes: Arc<[u8]> = bytes.to_vec().into(); - super::cache::insert( - verification.layout, - verification.cell, - verification.incarnation, - digest, - CellObjectKind::Directory, - bytes, - ) -} - -async fn read_node_uncached( - verification: &Verification<'_>, - digest: [u8; 32], -) -> Result> { - let path = verification.layout.incarnation_object_path( - verification.cell, - verification.incarnation, - &digest, - CellObjectKind::Directory, - ); - let _permit = verification.host.io_permit().await?; - let result = verification - .layout - .store() - .get_with_etag_bounded(&path, MAX_NODE_BYTES) - .await; - verification.host.observe_ltx_origin_request( - verification.origin, - result.is_ok(), - result.as_ref().map_or(0, |(bytes, _)| bytes.len()), - ); - let (bytes, _) = result?; - if *blake3::hash(&bytes).as_bytes() != digest { - return Err(CrabError::ChecksumMismatch); - } - Ok(bytes.to_vec().into()) -} - -fn encode_leaf(entries: &[DirectoryEntry]) -> Result> { - if entries.is_empty() || entries.len() > FANOUT { - return Err(CrabError::LTXCorrupted); - } - let live_pages = entries.len() as u64; - let checksum = entries.iter().fold(0, |sum, entry| sum ^ entry.checksum) | crate::CHECKSUM_FLAG; - let mut bytes = header(0, entries.len(), live_pages, checksum); - let mut previous = 0; - for entry in entries { - if entry.page <= previous || entry.length == 0 { - return Err(CrabError::LTXCorrupted); - } - bytes.extend_from_slice(&entry.page.to_be_bytes()); - bytes.extend_from_slice(&entry.object); - bytes.extend_from_slice(&entry.offset.to_be_bytes()); - bytes.extend_from_slice(&entry.length.to_be_bytes()); - bytes.extend_from_slice(&entry.frame_hash); - bytes.extend_from_slice(&entry.checksum.to_be_bytes()); - previous = entry.page; - } - Ok(bytes) -} - -fn encode_branch(children: &[Node]) -> Result> { - if children.is_empty() || children.len() > FANOUT { - return Err(CrabError::LTXCorrupted); - } - let live_pages = children.iter().try_fold(0u64, |total, child| { - total - .checked_add(child.aggregate.live_pages) - .ok_or(CrabError::LTXCorrupted) - })?; - let checksum = children - .iter() - .fold(0, |sum, child| sum ^ child.aggregate.checksum) - | crate::CHECKSUM_FLAG; - let mut bytes = header(1, children.len(), live_pages, checksum); - let mut previous = 0; - for child in children { - if child.aggregate.first == 0 - || child.aggregate.first <= previous - || child.aggregate.last < child.aggregate.first - { - return Err(CrabError::LTXCorrupted); - } - bytes.extend_from_slice(&child.aggregate.first.to_be_bytes()); - bytes.extend_from_slice(&child.aggregate.last.to_be_bytes()); - bytes.extend_from_slice(&child.digest); - bytes.extend_from_slice(&child.aggregate.live_pages.to_be_bytes()); - bytes.extend_from_slice(&child.aggregate.checksum.to_be_bytes()); - previous = child.aggregate.last; - } - Ok(bytes) -} - -fn verify_leaf( - bytes: &[u8], - header: &Header, - page_size: u32, - database_pages: u32, - extents: &BTreeMap<[u8; 32], ObjectExtent>, -) -> Result<(Aggregate, Vec)> { - let lock = crate::ltx::lock_pgno(page_size); - let mut checksum = 0; - let mut first = 0; - let mut last = 0; - let mut previous = 0; - let mut entries = Vec::with_capacity(header.entries as usize); - for index in 0..header.entries as usize { - let start = HEADER_BYTES + index * LEAF_RECORD_BYTES; - let page = read_u32(bytes, start)?; - let object = array(&bytes[start + 4..start + 36])?; - let offset = read_u64(bytes, start + 36)?; - let length = read_u32(bytes, start + 44)?; - let frame_hash = array(&bytes[start + 48..start + 80])?; - let page_checksum = read_u64(bytes, start + 80)?; - let extent = extents.get(&object).ok_or(CrabError::LTXCorrupted)?; - let frame_end = offset - .checked_add(u64::from(length)) - .ok_or(CrabError::LTXCorrupted)?; - if page == 0 - || page > database_pages - || page == lock - || page <= previous - || page_checksum & crate::CHECKSUM_FLAG == 0 - || length == 0 - || !extent - .ranges - .iter() - .any(|range| range.start <= offset && frame_end <= range.end) - { - return Err(CrabError::LTXCorrupted); - } - if first == 0 { - first = page; - } - last = page; - previous = page; - checksum ^= page_checksum; - entries.push(DirectoryEntry { - page, - object, - offset, - length, - frame_hash, - checksum: page_checksum, - }); - } - if (first - 1) / FANOUT as u32 != (last - 1) / FANOUT as u32 { - return Err(CrabError::LTXCorrupted); - } - let aggregate = Aggregate { - live_pages: u64::from(header.entries), - checksum: checksum | crate::CHECKSUM_FLAG, - first, - last, - }; - if aggregate.live_pages != header.live_pages || aggregate.checksum != header.checksum { - return Err(CrabError::ChecksumMismatch); - } - Ok((aggregate, entries)) -} - -fn verify_branch(bytes: &[u8], header: &Header) -> Result<(Aggregate, Vec)> { - let mut children = Vec::with_capacity(header.entries as usize); - let mut live_pages = 0u64; - let mut checksum = 0u64; - let mut first = 0; - let mut last = 0; - for index in 0..header.entries as usize { - let start = HEADER_BYTES + index * BRANCH_RECORD_BYTES; - let child_first = read_u32(bytes, start)?; - let child_last = read_u32(bytes, start + 4)?; - let digest = array(&bytes[start + 8..start + 40])?; - let child_pages = read_u64(bytes, start + 40)?; - let child_checksum = read_u64(bytes, start + 48)?; - if child_first == 0 - || child_last < child_first - || child_first <= last - || child_pages == 0 - || child_checksum & crate::CHECKSUM_FLAG == 0 - { - return Err(CrabError::LTXCorrupted); - } - if first == 0 { - first = child_first; - } - last = child_last; - live_pages = live_pages - .checked_add(child_pages) - .ok_or(CrabError::LTXCorrupted)?; - checksum ^= child_checksum; - children.push(Node { - index: 0, - digest, - aggregate: Aggregate { - live_pages: child_pages, - checksum: child_checksum, - first: child_first, - last: child_last, - }, - }); - } - let aggregate = Aggregate { - live_pages, - checksum: checksum | crate::CHECKSUM_FLAG, - first, - last, - }; - if aggregate.live_pages != header.live_pages || aggregate.checksum != header.checksum { - return Err(CrabError::ChecksumMismatch); - } - Ok((aggregate, children)) -} - -fn header(kind: u8, entries: usize, live_pages: u64, checksum: u64) -> Vec { - let mut bytes = Vec::with_capacity(HEADER_BYTES); - bytes.extend_from_slice(MAGIC); - bytes.extend_from_slice(&1u16.to_be_bytes()); - bytes.push(kind); - bytes.push(0); - bytes.extend_from_slice(&(entries as u32).to_be_bytes()); - bytes.extend_from_slice(&live_pages.to_be_bytes()); - bytes.extend_from_slice(&checksum.to_be_bytes()); - bytes -} - -fn read_u32(bytes: &[u8], offset: usize) -> Result { - Ok(u32::from_be_bytes(array( - bytes - .get(offset..offset + size_of::()) - .ok_or(CrabError::LTXCorrupted)?, - )?)) -} - -fn read_u64(bytes: &[u8], offset: usize) -> Result { - Ok(u64::from_be_bytes(array( - bytes - .get(offset..offset + size_of::()) - .ok_or(CrabError::LTXCorrupted)?, - )?)) -} - -fn array(bytes: &[u8]) -> Result<[u8; N]> { - bytes.try_into().map_err(|_| CrabError::LTXCorrupted) -} - -/// Shape-checks one encoded directory node without a root graph. -/// -/// Authentication needs the exact root's object extents, which this entry point -/// does not have: a leaf is checked against extents derived from its own -/// records, so every structural rule still runs while extent membership and the -/// lock-page rule stay unverified. Audit tooling that must authenticate a node -/// walks the root graph instead. -pub(crate) fn inspect_node(bytes: &[u8]) -> Result<()> { - let header = Header::parse(bytes)?; - if header.kind == 1 { - verify_branch(bytes, &header)?; - return Ok(()); - } - let mut extents: BTreeMap<[u8; 32], ObjectExtent> = BTreeMap::new(); - let mut database_pages = 0_u32; - for index in 0..header.entries as usize { - let start = HEADER_BYTES + index * LEAF_RECORD_BYTES; - let page = read_u32(bytes, start)?; - let object = array(&bytes[start + 4..start + 36])?; - let offset = read_u64(bytes, start + 36)?; - let length = read_u32(bytes, start + 44)?; - let end = offset - .checked_add(u64::from(length)) - .ok_or(CrabError::LTXCorrupted)?; - database_pages = database_pages.max(page); - extents - .entry(object) - .or_insert_with(|| ObjectExtent { - kind: crate::CellObjectKind::Ltx, - ranges: Vec::new(), - }) - .ranges - .push(offset..end); - } - // Page size zero disables the lock-page membership rule: with no root graph - // there is no page size to work from, and every other rule still runs. - verify_leaf(bytes, &header, 0, database_pages, &extents)?; - Ok(()) -} - -#[cfg(test)] -mod tests; diff --git a/crates/crab-ltx/src/replica/directory/checksums.rs b/crates/crab-ltx/src/replica/directory/checksums.rs deleted file mode 100644 index 659ef6018..000000000 --- a/crates/crab-ltx/src/replica/directory/checksums.rs +++ /dev/null @@ -1,275 +0,0 @@ -use std::path::{Path, PathBuf}; - -use futures_util::{StreamExt as _, stream}; - -use super::{Header, Verification, read_node, verify_branch, verify_leaf}; -use crate::{CrabError, Host, Limits, Result, environment::FileIo, pages::PageChecksums}; - -const LEAF_READS_IN_FLIGHT: usize = 8; - -pub(in crate::replica) async fn load_checksums( - verification: Verification<'_>, - root: [u8; 32], - height: u32, - destination: &Path, - limits: Limits, -) -> Result { - if height > 3 || verification.database_pages == 0 { - return Err(CrabError::LTXCorrupted); - } - if u64::from(verification.database_pages) * 8 > limits.max_database_bytes { - return Err(CrabError::Limit(crate::LimitKind::ChecksumFileBytes)); - } - let mut writer = ChecksumWriter::create(verification.host, destination).await?; - let result = async { - let mut pending = vec![vec![(root, height, None)]]; - let mut previous_page = 0u32; - let mut seen = 0u64; - let mut checksum = crate::CHECKSUM_FLAG; - while let Some(batch) = pending.pop() { - // Batch only sibling leaves. Internal nodes remain depth first so - // prefetched branches cannot reorder page coverage or checksums. - // At most eight reads use the shared host I/O admission. - let mut reads = stream::iter(batch.into_iter().map(|(digest, remaining, expected)| { - let verification = &verification; - async move { - read_node(verification, digest) - .await - .map(|bytes| (bytes, remaining, expected)) - } - })) - .buffered(LEAF_READS_IN_FLIGHT); - while let Some(read) = reads.next().await { - let (bytes, remaining, expected) = read?; - let header = Header::parse(&bytes)?; - if (remaining == 0) != (header.kind == 0) { - return Err(CrabError::LTXCorrupted); - } - if header.kind == 0 { - let (aggregate, entries) = verify_leaf( - &bytes, - &header, - verification.page_size, - verification.database_pages, - verification.extents, - )?; - if expected.is_some_and(|value| value != aggregate) { - return Err(CrabError::ChecksumMismatch); - } - for entry in entries { - let expected_page = previous_page - .checked_add(1) - .ok_or(CrabError::LTXCorrupted)?; - let lock = crate::ltx::lock_pgno(verification.page_size); - if expected_page == lock { - writer.append(0).await?; - previous_page = lock; - } - if entry.page - != previous_page - .checked_add(1) - .ok_or(CrabError::LTXCorrupted)? - { - return Err(CrabError::LTXCorrupted); - } - writer.append(entry.checksum).await?; - checksum = crate::CHECKSUM_FLAG | (checksum ^ entry.checksum); - previous_page = entry.page; - seen += 1; - } - continue; - } - let (aggregate, children) = verify_branch(&bytes, &header)?; - if expected.is_some_and(|value| value != aggregate) { - return Err(CrabError::ChecksumMismatch); - } - let next = remaining.checked_sub(1).ok_or(CrabError::LTXCorrupted)?; - let batch_size = if next == 0 { super::FANOUT } else { 1 }; - pending.extend(children.chunks(batch_size).rev().map(|children| { - children - .iter() - .map(|child| (child.digest, next, Some(child.aggregate))) - .collect() - })); - } - } - let lock = crate::ltx::lock_pgno(verification.page_size); - if previous_page < verification.database_pages { - if previous_page - .checked_add(1) - .ok_or(CrabError::LTXCorrupted)? - != lock - || lock != verification.database_pages - { - return Err(CrabError::LTXCorrupted); - } - writer.append(0).await?; - } - let expected = - u64::from(verification.database_pages) - u64::from(lock <= verification.database_pages); - if seen != expected { - return Err(CrabError::LTXCorrupted); - } - writer.flush().await?; - let page_size = verification.page_size; - let database_pages = verification.database_pages; - writer - .run(move |output| { - let file = output.file.take().ok_or(CrabError::LTXCorrupted)?; - drop(file); - // This checksum base is derived scratch. Durable warm handoff - // writes a separate synced sidecar; crashes restore the pinned root. - PageChecksums::from_file( - crate::LtxHost { - // Only activation work retains admission. The immutable - // checksum base lives with the writer after delivery. - facilities: output - .host - .clone() - .without_recovery() - .without_dirty() - .without_scratch(), - max_database_bytes: limits.max_database_bytes, - max_file_bytes: limits.max_database_bytes, - }, - &output.path, - page_size, - database_pages, - checksum, - ) - }) - .await - } - .await; - if result.is_ok() { - // Disarm after delivery: cancellation of a dispatched final metadata check must - // still remove its undelivered activation file. - if let Some(output) = &mut writer.output { - output.keep = true; - } - } else { - let _ = writer - .run(|output| { - output.remove(); - Ok(()) - }) - .await; - } - result -} - -struct ChecksumWriter { - output: Option, - buffer: Vec, -} - -impl ChecksumWriter { - async fn create(host: &Host, destination: &Path) -> Result { - let destination = destination.to_owned(); - let path = crate::resume::checksum_path(&destination); - let runtime = tokio::runtime::Handle::try_current().map_err(std::io::Error::other)?; - let job_host = host.clone(); - let output = host - .run(move || { - if job_host.filesystem.exists(&destination)? || job_host.filesystem.exists(&path)? { - return Err(CrabError::InvalidState( - "writable activation destination already exists", - )); - } - let file = job_host.filesystem.create(&path)?; - Ok(ChecksumFile { - host: job_host, - path, - file: Some(file), - runtime, - keep: false, - }) - }) - .await??; - Ok(Self { - output: Some(output), - buffer: Vec::with_capacity(64 << 10), - }) - } - - async fn append(&mut self, checksum: u64) -> Result<()> { - self.buffer.extend_from_slice(&checksum.to_be_bytes()); - if self.buffer.len() >= 64 << 10 { - self.flush().await?; - } - Ok(()) - } - - async fn flush(&mut self) -> Result<()> { - if self.buffer.is_empty() { - return Ok(()); - } - let mut bytes = std::mem::take(&mut self.buffer); - self.buffer = self - .run(move |output| { - output - .file - .as_mut() - .ok_or(CrabError::LTXCorrupted)? - .write_all(&bytes)?; - bytes.clear(); - Ok(bytes) - }) - .await?; - Ok(()) - } - - async fn run( - &mut self, - operation: impl FnOnce(&mut ChecksumFile) -> Result + Send + 'static, - ) -> Result { - let mut output = self.output.take().ok_or(CrabError::LTXCorrupted)?; - let host = output.host.clone(); - let (output, result) = host - .run(move || { - let result = operation(&mut output); - (output, result) - }) - .await?; - self.output = Some(output); - result - } -} - -struct ChecksumFile { - host: Host, - path: PathBuf, - file: Option>, - runtime: tokio::runtime::Handle, - keep: bool, -} - -impl ChecksumFile { - fn remove(&mut self) { - drop(self.file.take()); - let _ = self.host.filesystem.remove_file(&self.path); - self.keep = true; - } -} - -impl Drop for ChecksumFile { - fn drop(&mut self) { - if self.keep { - return; - } - let mut cleanup = Self { - host: self.host.clone(), - path: std::mem::take(&mut self.path), - file: self.file.take(), - runtime: self.runtime.clone(), - keep: true, - }; - // Cancellation can occur between network reads or while a file job - // is queued. Keep the file and dirty admission through admitted cleanup; - // never block the async worker or release capacity before cleanup ends. - self.runtime.spawn(async move { - let host = cleanup.host.clone(); - let _ = host.run(move || cleanup.remove()).await; - }); - } -} diff --git a/crates/crab-ltx/src/replica/directory/initial.rs b/crates/crab-ltx/src/replica/directory/initial.rs deleted file mode 100644 index 243f086ca..000000000 --- a/crates/crab-ltx/src/replica/directory/initial.rs +++ /dev/null @@ -1,204 +0,0 @@ -use crate::{CellObjectKind, CrabError, Result}; -use futures_util::{Stream, StreamExt as _}; - -use super::super::merge::LocatorMerge; -use super::super::{CellReplica, DirectoryInput, OBJECT_UPLOAD_CONCURRENCY}; -use super::{DirectoryEntry, DirectoryTree, FANOUT, Node, encode_branch, encode_leaf}; - -struct IndexCursor<'a> { - input: &'a DirectoryInput, - entries: crate::paged::ValidatedIndexEntries<'a>, - current: Option, -} - -impl<'a> IndexCursor<'a> { - fn new(input: &'a DirectoryInput) -> Result { - let mut cursor = Self { - input, - entries: crate::paged::validated_index_entries(&input.index, &input.descriptor.info)?, - current: None, - }; - cursor.advance()?; - Ok(cursor) - } - - fn advance(&mut self) -> Result<()> { - let Some(entry) = self.entries.next().transpose()? else { - self.current = None; - return Ok(()); - }; - let descriptor = &self.input.descriptor; - self.current = Some(DirectoryEntry { - page: entry.page, - object: descriptor.object_digest(), - offset: descriptor - .offset() - .checked_add(entry.offset) - .ok_or(CrabError::LTXCorrupted)?, - length: u32::try_from(entry.size).map_err(|_| CrabError::LTXCorrupted)?, - frame_hash: entry.hash, - checksum: entry.checksum, - }); - Ok(()) - } -} - -pub(in crate::replica) struct Entries<'a> { - cursors: Vec>, - merge: LocatorMerge, -} - -pub(in crate::replica) fn entries(inputs: &[DirectoryInput]) -> Result> { - if inputs.is_empty() { - return Err(CrabError::LTXCorrupted); - } - let mut merge = LocatorMerge::new( - inputs - .iter() - .map(|input| input.descriptor.info.database_pages), - ); - let mut cursors = Vec::with_capacity(inputs.len()); - for input in inputs { - let cursor = IndexCursor::new(input)?; - if let Some(entry) = &cursor.current { - merge.push(cursors.len(), entry.page); - } - cursors.push(cursor); - } - Ok(Entries { cursors, merge }) -} - -impl Iterator for Entries<'_> { - type Item = Result; - - fn next(&mut self) -> Option { - let Self { cursors, merge } = self; - loop { - let next = merge.next_group(|index| { - let cursor = cursors.get_mut(index).ok_or(CrabError::LTXCorrupted)?; - let entry = cursor.current.take().ok_or(CrabError::LTXCorrupted)?; - cursor.advance()?; - Ok((entry, cursor.current.as_ref().map(|entry| entry.page))) - })?; - match next { - Ok(None) => continue, - Ok(Some(entry)) => return Some(Ok(entry)), - Err(error) => return Some(Err(error)), - } - } - } -} - -pub(in crate::replica) async fn build_and_upload( - entries: impl Stream>, - page_size: u32, - database_pages: u32, - replica: &CellReplica, -) -> Result { - if database_pages == 0 { - return Err(CrabError::LTXCorrupted); - } - let lock = crate::ltx::lock_pgno(page_size); - let mut expected_page = 1_u64; - let mut live_pages = 0_u64; - let mut leaf_index = None; - let mut leaf_entries = Vec::with_capacity(FANOUT); - let mut nodes = Vec::new(); - let mut pending = Vec::with_capacity(OBJECT_UPLOAD_CONCURRENCY); - futures_util::pin_mut!(entries); - while let Some(entry) = entries.next().await { - let entry = entry?; - if expected_page == u64::from(lock) { - expected_page += 1; - } - if u64::from(entry.page) != expected_page - || entry.page > database_pages - || entry.page == lock - { - return Err(CrabError::LTXCorrupted); - } - let index = (entry.page - 1) / FANOUT as u32; - if leaf_index.is_some_and(|current| current != index) { - // Directory nodes are immutable and the private PreparedRoot is - // unreachable until every dependency is uploaded and verified. - pending.push(encode_leaf_node(leaf_index, &leaf_entries)?); - if pending.len() == OBJECT_UPLOAD_CONCURRENCY { - flush_node_uploads(replica, &mut pending, &mut nodes).await?; - } - leaf_entries.clear(); - } - leaf_index = Some(index); - leaf_entries.push(entry); - live_pages += 1; - expected_page += 1; - } - if !leaf_entries.is_empty() { - pending.push(encode_leaf_node(leaf_index, &leaf_entries)?); - } - flush_node_uploads(replica, &mut pending, &mut nodes).await?; - if expected_page == u64::from(lock) { - expected_page += 1; - } - let expected_live = u64::from(database_pages) - u64::from(lock <= database_pages); - if expected_page != u64::from(database_pages) + 1 || live_pages != expected_live { - return Err(CrabError::LTXCorrupted); - } - - let mut height = 0; - while nodes.len() > 1 { - let mut parents = Vec::new(); - let mut pending = Vec::with_capacity(OBJECT_UPLOAD_CONCURRENCY); - for group in - nodes.chunk_by(|left, right| left.index / FANOUT as u32 == right.index / FANOUT as u32) - { - let index = group[0].index / FANOUT as u32; - let bytes = encode_branch(group)?; - let node = Node::from_bytes(index, &bytes)?; - pending.push((node, bytes)); - if pending.len() == OBJECT_UPLOAD_CONCURRENCY { - flush_node_uploads(replica, &mut pending, &mut parents).await?; - } - } - flush_node_uploads(replica, &mut pending, &mut parents).await?; - nodes = parents; - height += 1; - } - let root = nodes.pop().ok_or(CrabError::LTXCorrupted)?; - Ok(DirectoryTree { - objects: Vec::new(), - root, - height, - }) -} - -fn encode_leaf_node(index: Option, entries: &[DirectoryEntry]) -> Result<(Node, Vec)> { - let index = index.ok_or(CrabError::LTXCorrupted)?; - let bytes = encode_leaf(entries)?; - let node = Node::from_bytes(index, &bytes)?; - Ok((node, bytes)) -} - -async fn flush_node_uploads( - replica: &CellReplica, - pending: &mut Vec<(Node, Vec)>, - output: &mut Vec, -) -> Result<()> { - if pending.is_empty() { - return Ok(()); - } - let pending = std::mem::replace(pending, Vec::with_capacity(OBJECT_UPLOAD_CONCURRENCY)); - let mut nodes = Vec::with_capacity(pending.len()); - let objects = pending - .into_iter() - .map(|(node, bytes)| { - let digest = node.digest; - nodes.push(node); - (digest, bytes) - }) - .collect(); - replica - .put_objects(CellObjectKind::Directory, objects) - .await?; - output.extend(nodes); - Ok(()) -} diff --git a/crates/crab-ltx/src/replica/directory/relocate.rs b/crates/crab-ltx/src/replica/directory/relocate.rs deleted file mode 100644 index 1a15a6e14..000000000 --- a/crates/crab-ltx/src/replica/directory/relocate.rs +++ /dev/null @@ -1,207 +0,0 @@ -//! Stream representation-only locator changes through an authenticated directory. - -use std::{collections::BTreeMap, future::Future, pin::Pin}; - -use futures_util::{Stream, TryStreamExt as _}; - -use super::super::{ - CellReplica, LoadedGraph, OBJECT_UPLOAD_CONCURRENCY, SegmentDescriptor, object_extents, -}; -use super::{ - DirectoryEntry, DirectoryTree, Header, Node, ObjectExtent, Verification, encode_branch, - encode_leaf, read_node, verify_branch, verify_leaf, -}; -use crate::{CellObjectKind, CrabError, Result}; - -pub(in crate::replica) async fn run( - replica: &CellReplica, - graph: &LoadedGraph, - descriptors: &[SegmentDescriptor], - selected: &[SegmentDescriptor], - entries: impl Stream> + Send, -) -> Result { - let base_extents = object_extents(&graph.descriptors)?; - let final_extents = object_extents(descriptors)?; - let selected_extents = object_extents(selected)?; - let base = Verification { - layout: &replica.layout, - cell: &replica.cell, - incarnation: &replica.incarnation, - page_size: graph.document.page_size, - database_pages: graph.document.database_pages, - extents: &base_extents, - host: &replica.host, - origin: crate::LtxReadOrigin::Cold, - }; - let mut state = Relocator { - replica, - base, - final_extents: &final_extents, - selected_extents: &selected_extents, - entries: Box::pin(entries), - next: None, - previous: 0, - pending: Vec::with_capacity(OBJECT_UPLOAD_CONCURRENCY), - }; - state.advance().await?; - let root = Node { - index: 0, - digest: graph.document.directory_digest, - aggregate: graph.aggregate, - }; - let root = state.rewrite(root, graph.document.directory_height).await?; - if state.next.is_some() || root.aggregate != graph.aggregate { - return Err(CrabError::ChecksumMismatch); - } - state.flush().await?; - Ok(DirectoryTree { - objects: Vec::new(), - root, - height: graph.document.directory_height, - }) -} - -struct Relocator<'a> { - replica: &'a CellReplica, - base: Verification<'a>, - final_extents: &'a BTreeMap<[u8; 32], ObjectExtent>, - selected_extents: &'a BTreeMap<[u8; 32], ObjectExtent>, - entries: Pin> + Send + 'a>>, - next: Option, - previous: u32, - pending: Vec<([u8; 32], Vec)>, -} - -impl Relocator<'_> { - async fn advance(&mut self) -> Result<()> { - self.next = None; - while let Some(entry) = self.entries.try_next().await? { - if entry.page <= self.previous - || entry.page == crate::ltx::lock_pgno(self.base.page_size) - { - return Err(CrabError::LTXCorrupted); - } - self.previous = entry.page; - // Later cuts may truncate pages retained at the selected range's - // endpoint. Their compacted bytes must not revive final-root pages. - if entry.page <= self.base.database_pages { - self.next = Some(entry); - break; - } - } - Ok(()) - } - - fn rewrite( - &mut self, - old: Node, - level: u32, - ) -> Pin> + Send + '_>> { - Box::pin(async move { - let Some(next) = &self.next else { - return Ok(old); - }; - if next.page > old.aggregate.last { - return Ok(old); - } - if next.page < old.aggregate.first || level > 3 { - return Err(CrabError::LTXCorrupted); - } - let bytes = read_node(&self.base, old.digest).await?; - let header = Header::parse(&bytes)?; - if (level == 0) != (header.kind == 0) { - return Err(CrabError::LTXCorrupted); - } - let bytes = if level == 0 { - self.leaf(&old, &bytes, &header).await? - } else { - let (aggregate, mut children) = verify_branch(&bytes, &header)?; - if aggregate != old.aggregate { - return Err(CrabError::ChecksumMismatch); - } - for child in &mut children { - *child = self.rewrite(child.clone(), level - 1).await?; - } - encode_branch(&children)? - }; - let node = Node::from_bytes(old.index, &bytes)?; - if node.aggregate != old.aggregate { - return Err(CrabError::ChecksumMismatch); - } - if node.digest != old.digest { - self.pending.push((node.digest, bytes)); - if self.pending.len() == OBJECT_UPLOAD_CONCURRENCY { - self.flush().await?; - } - } - Ok(node) - }) - } - - async fn leaf(&mut self, old: &Node, bytes: &[u8], header: &Header) -> Result> { - let (aggregate, mut entries) = verify_leaf( - bytes, - header, - self.base.page_size, - self.base.database_pages, - self.base.extents, - )?; - if aggregate != old.aggregate { - return Err(CrabError::ChecksumMismatch); - } - for entry in &mut entries { - let Some(next) = &self.next else { - break; - }; - if next.page < entry.page { - return Err(CrabError::LTXCorrupted); - } - if next.page != entry.page { - continue; - } - // Object identity alone is insufficient: selected and newer cuts - // may occupy disjoint ranges of the same follower bundle. - let selected = self - .selected_extents - .get(&entry.object) - .is_some_and(|extent| { - extent.ranges.iter().any(|range| { - range.start <= entry.offset - && entry.offset + u64::from(entry.length) <= range.end - }) - }); - if selected { - if entry.checksum != next.checksum { - return Err(CrabError::ChecksumMismatch); - } - *entry = next.clone(); - } - self.advance().await?; - } - let bytes = encode_leaf(&entries)?; - let header = Header::parse(&bytes)?; - verify_leaf( - &bytes, - &header, - self.base.page_size, - self.base.database_pages, - self.final_extents, - )?; - Ok(bytes) - } - - async fn flush(&mut self) -> Result<()> { - if self.pending.is_empty() { - return Ok(()); - } - let objects = std::mem::replace( - &mut self.pending, - Vec::with_capacity(OBJECT_UPLOAD_CONCURRENCY), - ); - // Only immutable dependencies escape here. The owner cannot publish - // a proposal until every changed branch and its children have uploaded. - self.replica - .put_objects(CellObjectKind::Directory, objects) - .await - } -} diff --git a/crates/crab-ltx/src/replica/directory/tests.rs b/crates/crab-ltx/src/replica/directory/tests.rs deleted file mode 100644 index 985509e90..000000000 --- a/crates/crab-ltx/src/replica/directory/tests.rs +++ /dev/null @@ -1,238 +0,0 @@ -use std::collections::BTreeMap; -use std::sync::Arc; - -use bytes::Bytes; -use crab_storage::Store; -use object_store::{ - memory::InMemory, - path::Path, - throttle::{ThrottleConfig, ThrottledStore}, -}; - -use super::*; - -#[test] -fn radix_tree_round_trips_multi_level_aggregates() { - let entries = (1..=70_000) - .map(|page| { - ( - page, - DirectoryEntry { - page, - object: [1; 32], - offset: u64::from(page) * 100, - length: 100, - frame_hash: [2; 32], - checksum: u64::from(page) | crate::CHECKSUM_FLAG, - }, - ) - }) - .collect(); - let tree = DirectoryTree::build(entries, 4096, 70_000).unwrap(); - assert_eq!(tree.height(), 2); - assert_eq!(tree.root.aggregate.live_pages, 70_000); - assert!(tree.objects.len() > 256); -} - -#[tokio::test(start_paused = true)] -async fn streamed_tree_matches_canonical_root_without_retaining_objects() { - let entries = (1..=70_000) - .map(|page| DirectoryEntry { - page, - object: [1; 32], - offset: u64::from(page) * 100, - length: 100, - frame_hash: [2; 32], - checksum: u64::from(page) | crate::CHECKSUM_FLAG, - }) - .collect::>(); - let canonical = DirectoryTree::build( - entries - .iter() - .cloned() - .map(|entry| (entry.page, entry)) - .collect(), - 4096, - 70_000, - ) - .unwrap(); - let delay = std::time::Duration::from_millis(10); - let store = Store::new(Arc::new(ThrottledStore::new( - InMemory::new(), - ThrottleConfig { - wait_put_per_call: delay, - ..ThrottleConfig::default() - }, - ))); - let layout = CellStorageLayout::new(store.clone(), Path::from("streaming"), [3; 16]); - let replica = - super::super::CellReplica::new(layout, [1; 32], [2; 16], crate::Limits::default()).unwrap(); - let started = tokio::time::Instant::now(); - let streamed = build_initial_and_upload( - futures_util::stream::iter(entries.into_iter().map(Ok)), - 4096, - 70_000, - &replica, - ) - .await - .unwrap(); - let mut level_nodes = 70_000_usize.div_ceil(FANOUT); - let mut upload_intervals = 0_usize; - loop { - upload_intervals += level_nodes.div_ceil(super::super::OBJECT_UPLOAD_CONCURRENCY); - if level_nodes == 1 { - break; - } - level_nodes = level_nodes.div_ceil(FANOUT); - } - assert_eq!( - started.elapsed(), - delay * u32::try_from(upload_intervals).unwrap() - ); - - assert_eq!(streamed.root_digest(), canonical.root_digest()); - assert_eq!(streamed.height(), canonical.height()); - assert_eq!(streamed.checksum(), canonical.checksum()); - assert!(streamed.objects().is_empty()); - assert_eq!( - store - .list_prefix(&Path::from("streaming/cells/v1")) - .await - .unwrap() - .len(), - canonical.objects().len() - ); -} - -#[tokio::test] -async fn incremental_update_rebuilds_the_last_height_two_branch() { - let page_size = 4096; - let base_pages = 401_938; - let final_pages = 403_220; - let lock = crate::ltx::lock_pgno(page_size); - let cell = [8; 32]; - let incarnation = [9; 16]; - let store = Store::new(Arc::new(InMemory::new())); - let layout = CellStorageLayout::new(store.clone(), Path::from("update"), incarnation); - let replica = - super::super::CellReplica::new(layout.clone(), cell, incarnation, crate::Limits::default()) - .unwrap(); - let entry = |page: u32, object: [u8; 32]| DirectoryEntry { - page, - object, - offset: u64::from(page) * 100, - length: 100, - frame_hash: [page as u8; 32], - checksum: crate::CHECKSUM_FLAG | u64::from(page), - }; - let pages = |end: u32, object| { - (1..=end) - .filter(|page| *page != lock) - .map(|page| (page, entry(page, object))) - .collect::>() - }; - let base_entries = pages(base_pages, [1; 32]); - let final_entries = pages(final_pages, [1; 32]) - .into_iter() - .map(|(page, mut entry)| { - if page <= 2 || page > base_pages { - entry.object = [2; 32]; - } - (page, entry) - }) - .collect::>(); - let base_tree = DirectoryTree::build(base_entries, page_size, base_pages).unwrap(); - for object in &base_tree.objects { - let path = layout.incarnation_object_path( - &cell, - &incarnation, - &object.digest, - CellObjectKind::Directory, - ); - store - .put(&path, Bytes::from(object.bytes.clone())) - .await - .unwrap(); - } - let final_tree = DirectoryTree::build(final_entries.clone(), page_size, final_pages).unwrap(); - let changes = final_entries - .into_iter() - .filter(|(page, _)| *page <= 2 || *page > base_pages) - .collect::>(); - let base_extents = BTreeMap::from([( - [1; 32], - ObjectExtent { - kind: CellObjectKind::Ltx, - ranges: std::iter::once(0..50_000_000).collect(), - }, - )]); - let final_extents = BTreeMap::from([ - ( - [1; 32], - ObjectExtent { - kind: CellObjectKind::Ltx, - ranges: std::iter::once(0..50_000_000).collect(), - }, - ), - ( - [2; 32], - ObjectExtent { - kind: CellObjectKind::Ltx, - ranges: std::iter::once(0..50_000_000).collect(), - }, - ), - ]); - let updated = DirectoryTree::update( - Verification { - layout: &layout, - cell: &cell, - incarnation: &incarnation, - page_size, - database_pages: base_pages, - extents: &base_extents, - host: &replica.host, - origin: crate::LtxReadOrigin::Cold, - }, - base_tree.root_digest(), - base_tree.height(), - base_tree.root.aggregate, - changes, - base_pages, - Verification { - layout: &layout, - cell: &cell, - incarnation: &incarnation, - page_size, - database_pages: final_pages, - extents: &final_extents, - host: &replica.host, - origin: crate::LtxReadOrigin::Cold, - }, - final_tree.checksum(), - ) - .await - .unwrap(); - assert_eq!(updated.root_digest(), final_tree.root_digest()); - assert_eq!(updated.height(), final_tree.height()); -} - -#[test] -fn radix_tree_rejects_missing_allocated_page() { - let entries = [1, 3] - .into_iter() - .map(|page| { - ( - page, - DirectoryEntry { - page, - object: [1; 32], - offset: u64::from(page) * 100, - length: 100, - frame_hash: [2; 32], - checksum: u64::from(page) | crate::CHECKSUM_FLAG, - }, - ) - }) - .collect(); - assert!(DirectoryTree::build(entries, 4096, 3).is_err()); -} diff --git a/crates/crab-ltx/src/replica/directory/update.rs b/crates/crab-ltx/src/replica/directory/update.rs deleted file mode 100644 index c795547ce..000000000 --- a/crates/crab-ltx/src/replica/directory/update.rs +++ /dev/null @@ -1,368 +0,0 @@ -use std::{ - collections::{BTreeMap, BTreeSet}, - future::Future, - pin::Pin, -}; - -use crate::{CrabError, Result}; - -use super::{ - Aggregate, DirectoryEntry, DirectoryObject, DirectoryTree, FANOUT, Header, Node, Verification, - encode_branch, encode_leaf, read_node, verify_branch, verify_leaf, -}; - -type ChangedLeaves = BTreeMap>; - -pub(super) async fn run( - base: Verification<'_>, - root: [u8; 32], - height: u32, - aggregate: Aggregate, - changes: BTreeMap, - retain_through: u32, - final_state: Verification<'_>, - expected_checksum: u64, -) -> Result { - if height > 3 - || base.page_size != final_state.page_size - || retain_through > base.database_pages - || final_state.database_pages == 0 - { - return Err(CrabError::LTXCorrupted); - } - let changed = group_changes(changes, &final_state)?; - let mut objects = Vec::new(); - let old = Node { - index: 0, - digest: root, - aggregate, - }; - let root_span = node_span(height)?; - let root_end = u32::try_from(root_span).map_err(|_| CrabError::LTXCorrupted)?; - let root_changes = changed - .range(..root_end) - .map(|(leaf, entries)| (*leaf, entries.clone())) - .collect::(); - let mut nodes = Vec::new(); - if let Some(node) = mutate_node( - base, - final_state, - Some(old), - height, - 0, - &root_changes, - retain_through, - &mut objects, - ) - .await? - { - nodes.push(node); - } - - let mut outside = changed.range(root_end..).peekable(); - while let Some((leaf, _)) = outside.peek() { - let node_index = u64::from(**leaf) / root_span; - let node_index = u32::try_from(node_index).map_err(|_| CrabError::LTXCorrupted)?; - let end = u64::from(node_index) - .checked_add(1) - .and_then(|value| value.checked_mul(root_span)) - .ok_or(CrabError::LTXCorrupted)?; - let mut subtree = ChangedLeaves::new(); - while let Some((leaf, entries)) = outside.peek() { - if u64::from(**leaf) >= end { - break; - } - subtree.insert(**leaf, (*entries).clone()); - outside.next(); - } - let node = mutate_node( - base, - final_state, - None, - height, - node_index, - &subtree, - retain_through, - &mut objects, - ) - .await? - .ok_or(CrabError::LTXCorrupted)?; - nodes.push(node); - } - - let mut final_height = height; - while nodes.len() > 1 { - nodes = build_parent_level(nodes, &mut objects)?; - final_height = final_height - .checked_add(1) - .filter(|height| *height <= 3) - .ok_or(CrabError::LTXCorrupted)?; - } - let root = nodes.pop().ok_or(CrabError::LTXCorrupted)?; - let lock = crate::ltx::lock_pgno(final_state.page_size); - let expected_pages = - u64::from(final_state.database_pages) - u64::from(lock <= final_state.database_pages); - if root.index != 0 - || root.aggregate.live_pages != expected_pages - || root.aggregate.checksum != expected_checksum - || root.aggregate.first == 0 - || root.aggregate.last > final_state.database_pages - { - return Err(CrabError::ChecksumMismatch); - } - Ok(DirectoryTree { - objects, - root, - height: final_height, - }) -} - -fn mutate_node<'a>( - base: Verification<'a>, - final_state: Verification<'a>, - old: Option, - level: u32, - index: u32, - changes: &'a ChangedLeaves, - retain_through: u32, - objects: &'a mut Vec, -) -> Pin>> + Send + 'a>> { - Box::pin(async move { - if old - .as_ref() - .is_some_and(|node| node.aggregate.last <= retain_through && changes.is_empty()) - { - return Ok(old); - } - if old - .as_ref() - .is_some_and(|node| node.aggregate.first > retain_through && changes.is_empty()) - { - // Authenticated child ranges let truncation discard whole subtrees - // without loading their leaves or reviving any retired locator. - return Ok(None); - } - if old.is_none() && changes.is_empty() { - return Ok(None); - } - if level == 0 { - return mutate_leaf( - base, - final_state, - old, - index, - changes, - retain_through, - objects, - ) - .await; - } - - let mut children = BTreeMap::new(); - if let Some(node) = &old { - let bytes = read_node(&base, node.digest).await?; - let header = Header::parse(&bytes)?; - if header.kind != 1 { - return Err(CrabError::LTXCorrupted); - } - let (actual, decoded) = verify_branch(&bytes, &header)?; - if actual != node.aggregate { - return Err(CrabError::ChecksumMismatch); - } - for child in decoded { - let child_index = index_for_page(child.aggregate.first, level - 1)?; - let last_index = index_for_page(child.aggregate.last, level - 1)?; - let invalid_range = - last_index != child_index || child_index / FANOUT as u32 != index; - let duplicate = children.contains_key(&child_index); - if invalid_range || duplicate { - return Err(CrabError::LTXCorrupted); - } - // Branch records authenticate ranges but omit indexes; restore the - // derived placement before parent rebuilding can sort children. - let child = Node { - index: child_index, - ..child - }; - children.insert(child_index, child); - } - } - - let mut selected = children.keys().copied().collect::>(); - for leaf in changes.keys() { - selected.insert(node_index_for_leaf(*leaf, level - 1)?); - } - let mut output = Vec::with_capacity(selected.len()); - for child_index in selected { - let old_child = children.remove(&child_index); - let subtree = changes_for_node(changes, child_index, level - 1)?; - let affected = !subtree.is_empty() - || old_child - .as_ref() - .is_some_and(|child| child.aggregate.last > retain_through); - if !affected { - if let Some(child) = old_child { - output.push(child); - } - continue; - } - if let Some(child) = mutate_node( - base, - final_state, - old_child, - level - 1, - child_index, - &subtree, - retain_through, - objects, - ) - .await? - { - output.push(child); - } - } - if output.is_empty() { - return Ok(None); - } - output.sort_by_key(|child| child.index); - let bytes = encode_branch(&output)?; - let node = Node::from_bytes(index, &bytes)?; - if old.as_ref().is_some_and(|old| old.digest == node.digest) { - return Ok(old); - } - objects.push(DirectoryObject { - digest: node.digest, - bytes, - }); - Ok(Some(node)) - }) -} - -async fn mutate_leaf( - base: Verification<'_>, - final_state: Verification<'_>, - old: Option, - index: u32, - changes: &ChangedLeaves, - retain_through: u32, - objects: &mut Vec, -) -> Result> { - let mut entries = BTreeMap::new(); - if let Some(node) = &old { - let bytes = read_node(&base, node.digest).await?; - let header = Header::parse(&bytes)?; - if header.kind != 0 { - return Err(CrabError::LTXCorrupted); - } - let (actual, decoded) = verify_leaf( - &bytes, - &header, - base.page_size, - base.database_pages, - base.extents, - )?; - if actual != node.aggregate { - return Err(CrabError::ChecksumMismatch); - } - entries.extend( - decoded - .into_iter() - .filter(|entry| entry.page <= retain_through) - .map(|entry| (entry.page, entry)), - ); - } - if let Some(changed) = changes.get(&index) { - entries.extend(changed.iter().cloned().map(|entry| (entry.page, entry))); - } - entries.retain(|page, _| *page <= final_state.database_pages); - if entries.is_empty() { - return Ok(None); - } - let entries = entries.into_values().collect::>(); - let bytes = encode_leaf(&entries)?; - let header = Header::parse(&bytes)?; - verify_leaf( - &bytes, - &header, - final_state.page_size, - final_state.database_pages, - final_state.extents, - )?; - let node = Node::from_bytes(index, &bytes)?; - if old.as_ref().is_some_and(|old| old.digest == node.digest) { - return Ok(old); - } - objects.push(DirectoryObject { - digest: node.digest, - bytes, - }); - Ok(Some(node)) -} - -fn group_changes( - changes: BTreeMap, - final_state: &Verification<'_>, -) -> Result { - let lock = crate::ltx::lock_pgno(final_state.page_size); - let mut leaves = ChangedLeaves::new(); - for (page, entry) in changes { - if page != entry.page || page == 0 || page > final_state.database_pages || page == lock { - return Err(CrabError::LTXCorrupted); - } - leaves - .entry((page - 1) / FANOUT as u32) - .or_default() - .push(entry); - } - Ok(leaves) -} - -fn changes_for_node(changes: &ChangedLeaves, index: u32, level: u32) -> Result { - let span = node_span(level)?; - let first = u64::from(index) - .checked_mul(span) - .ok_or(CrabError::LTXCorrupted)?; - let end = first.checked_add(span).ok_or(CrabError::LTXCorrupted)?; - let first = u32::try_from(first).map_err(|_| CrabError::LTXCorrupted)?; - let end = u32::try_from(end).map_err(|_| CrabError::LTXCorrupted)?; - Ok(changes - .range(first..end) - .map(|(leaf, entries)| (*leaf, entries.clone())) - .collect()) -} - -fn build_parent_level(nodes: Vec, objects: &mut Vec) -> Result> { - let mut parents = Vec::new(); - for group in - nodes.chunk_by(|left, right| left.index / FANOUT as u32 == right.index / FANOUT as u32) - { - let index = group[0].index / FANOUT as u32; - let bytes = encode_branch(group)?; - let node = Node::from_bytes(index, &bytes)?; - objects.push(DirectoryObject { - digest: node.digest, - bytes, - }); - parents.push(node); - } - Ok(parents) -} - -fn index_for_page(page: u32, level: u32) -> Result { - if page == 0 { - return Err(CrabError::LTXCorrupted); - } - node_index_for_leaf((page - 1) / FANOUT as u32, level) -} - -fn node_index_for_leaf(leaf: u32, level: u32) -> Result { - let span = node_span(level)?; - u32::try_from(u64::from(leaf) / span).map_err(|_| CrabError::LTXCorrupted) -} - -fn node_span(level: u32) -> Result { - (0..level).try_fold(1_u64, |span, _| { - span.checked_mul(FANOUT as u64) - .ok_or(CrabError::LTXCorrupted) - }) -} diff --git a/crates/crab-ltx/src/replica/merge.rs b/crates/crab-ltx/src/replica/merge.rs deleted file mode 100644 index 073c8c89f..000000000 --- a/crates/crab-ltx/src/replica/merge.rs +++ /dev/null @@ -1,92 +0,0 @@ -//! Newest-wins merge of one replica view's page locators. - -use std::{cmp::Reverse, collections::BinaryHeap}; - -use crate::Result; - -use super::DirectoryEntry; - -/// Merges the page locators of one view's sources, oldest source first. -/// -/// A later truncation invalidates every older locator above its boundary, so -/// suffix minima of each source's `database_pages` drop those locators without a -/// page map, and at one page the newest surviving cut is authoritative. Both the -/// compaction rewrite and the initial directory build must resolve a page the -/// same way, so the decision lives here once. -pub(in crate::replica) struct LocatorMerge { - heap: BinaryHeap>, - valid_through: Vec, - failed: bool, -} - -impl LocatorMerge { - /// Prepares the merge for one `database_pages` entry per source. - pub(in crate::replica) fn new(database_pages: impl DoubleEndedIterator) -> Self { - let mut valid_through = Vec::new(); - let mut suffix_min = u32::MAX; - for pages in database_pages.rev() { - suffix_min = suffix_min.min(pages); - valid_through.push(suffix_min); - } - valid_through.reverse(); - Self { - heap: BinaryHeap::new(), - valid_through, - failed: false, - } - } - - /// Records that `index` is positioned on `page`. - pub(in crate::replica) fn push(&mut self, index: usize, page: u32) { - self.heap.push(Reverse((page, index))); - } - - /// Resolves one page group, returning no locator when truncation discards it. - /// - /// `take` removes one source's current locator and reports the page that - /// source moved to, if any. After an error the merge yields nothing further, - /// so a caller cannot build a view from a partially read set of sources. - pub(in crate::replica) fn next_group( - &mut self, - mut take: impl FnMut(usize) -> Result<(DirectoryEntry, Option)>, - ) -> Option>> { - if self.failed { - return None; - } - let Reverse((page, first_index)) = self.heap.pop()?; - let mut selected = None; - let mut index = first_index; - loop { - let (entry, next_page) = match take(index) { - Ok(taken) => taken, - Err(error) => { - self.failed = true; - self.heap.clear(); - return Some(Err(error)); - } - }; - if self - .valid_through - .get(index) - .is_some_and(|pages| page <= *pages) - && selected - .as_ref() - .is_none_or(|(selected_index, _)| index > *selected_index) - { - selected = Some((index, entry)); - } - if let Some(next_page) = next_page { - self.heap.push(Reverse((next_page, index))); - } - let Some(Reverse((next_page, next_index))) = self.heap.peek().copied() else { - break; - }; - if next_page != page { - break; - } - self.heap.pop(); - index = next_index; - } - Some(Ok(selected.map(|(_, entry)| entry))) - } -} diff --git a/crates/crab-ltx/src/replica/prepare.rs b/crates/crab-ltx/src/replica/prepare.rs deleted file mode 100644 index 9139102e2..000000000 --- a/crates/crab-ltx/src/replica/prepare.rs +++ /dev/null @@ -1,643 +0,0 @@ -//! Preparing an immutable root and its follower-visible inputs. -//! -//! Every entry point here verifies the base it continues, admits the -//! segments, uploads the objects a root needs, and finishes only once the -//! root document matches the admitted frames. - -use super::*; - -impl CellReplica { - /// Verifies and uploads a new immutable root without changing authority. - pub async fn prepare( - &self, - base: Option<&RootRef>, - cuts: &CaptureBatch, - commit_sequence: u64, - schema: u32, - ) -> Result { - let started = self.host.now_monotonic(); - let result = async { - let mut replica = self.clone(); - replica.host = self.host.for_dirty().await?; - replica - .prepare_captured(base, cuts, commit_sequence, schema) - .await - } - .await; - self.host - .observe_ltx_phase(crate::LtxPhase::RootPreparation, started, result.is_ok()); - result - } - - async fn prepare_captured( - &self, - base: Option<&RootRef>, - cuts: &CaptureBatch, - commit_sequence: u64, - schema: u32, - ) -> Result { - self.validate_metadata(commit_sequence, schema)?; - if cuts.segments.is_empty() { - return Err(CrabError::InvalidState("empty Cell append")); - } - let captured_bytes = cuts.segments.iter().try_fold(0_u64, |total, segment| { - // A full database image may legitimately exceed the incremental - // bound; the per-representation check below rejects an oversized - // delta after its index proves the coverage. - if segment.info().size_bytes > self.limits.max_file_bytes { - return Err(CrabError::Limit(crate::LimitKind::CapturedCellLtxBytes)); - } - total - .checked_add(segment.info().size_bytes) - .ok_or(CrabError::Limit(crate::LimitKind::CapturedCellLtxBytes)) - })?; - if captured_bytes > self.limits.max_plan_bytes { - return Err(CrabError::Limit(crate::LimitKind::CapturedCellLtxBytes)); - } - let load_base = async { - match base { - Some(root) => self.load_graph(root).await.map(Some), - None => Ok(None), - } - }; - // Local captures and the immutable predecessor cannot affect each - // other; chain validation still waits for both exact inputs. - let (base_graph, inputs) = - futures_util::future::try_join(load_base, self.prepare_captured_inputs(&cuts.segments)) - .await?; - self.validate_append_sequence(&base_graph, commit_sequence)?; - - // Admit the complete prospective chain from trusted capture metadata - // before reading local bodies or starting immutable uploads. - let mut descriptors = base_graph - .as_ref() - .map(|graph| graph.descriptors.clone()) - .unwrap_or_default(); - descriptors.extend( - cuts.segments - .iter() - .map(|segment| SegmentDescriptor::native(segment.info().clone(), [0; 32], 0)), - ); - self.validate_chain(&descriptors, cuts.position)?; - - // Keep each exact capture handle open through verification and upload. - // A path replacement cannot redirect retries, while the inspected LTX - // digest still rejects in-place mutation before authority may publish. - self.prepare_append( - base, - base_graph, - inputs, - cuts.position, - commit_sequence, - schema, - None, - ) - .await - } - - async fn prepare_captured_inputs( - &self, - segments: &[crate::LocalSegment], - ) -> Result> { - stream::iter(segments.iter().cloned().map(|segment| async move { - let source = segment.path().to_owned(); - let info = segment.info().clone(); - let source = upload::PinnedCapture::open(&self.host, source, info.size_bytes).await?; - let index = match segment.captured_index() { - Some(index) => index, - None => Bytes::from( - upload::inspect_segment_source(self, Arc::clone(&source), &info).await?, - ), - }; - self.admit_segment_representation(&info, index.len())?; - Ok(AppendInput { - info, - location: BodyLocation::Native, - index, - body: AppendBody::Native(source), - }) - })) - // Preserve descriptor order while overlapping independent file jobs. - // Host job permits remain the shared process-wide admission boundary. - .buffered(SEGMENT_TRANSFER_CONCURRENCY) - .try_collect() - .await - } - - /// Admits one captured segment's representation against the publication bounds. - /// - /// A segment larger than the incremental bound must be a full database - /// image: its encoded index has to cover every page the commit published. - /// The capture writer already escalates an oversized delta to that - /// representation, so a large partial index is a corrupt or foreign cut. - fn admit_segment_representation( - &self, - info: &crate::SegmentInfo, - index_bytes: usize, - ) -> Result<()> { - if info.size_bytes <= self.limits.max_capture_bytes { - return Ok(()); - } - let lock = crate::ltx::lock_pgno(info.page_size); - let expected_pages = - u64::from(info.database_pages) - u64::from(lock <= info.database_pages); - if (index_bytes / crate::paged::ENTRY_BYTES) as u64 != expected_pages { - return Err(CrabError::Limit(crate::LimitKind::CapturedCellLtxBytes)); - } - Ok(()) - } - - /// Verifies selected Cell rows from a shared bundle and prepares one root append. - /// - /// Bundle row identity is routing metadata, not authorization. Only rows using - /// the canonical Cell/incarnation identity are selected, and their complete LTX - /// chain is independently verified before the immutable bundle is retained. - pub async fn prepare_bundle( - &self, - base: Option<&RootRef>, - bundle: &crate::bundle::Bundle, - commit_sequence: u64, - schema: u32, - ) -> Result { - let mut replica = self.clone(); - replica.host = self.host.for_dirty().await?; - replica - .prepare_bundle_admitted(base, bundle, commit_sequence, schema) - .await - } - - /// Prepares the exact successor pinned by a recovered node-log overlay. - /// - /// Recovery policy and ownership remain caller-owned. This method accepts - /// only this replica's Cell/incarnation rows, requires the declared final - /// position to match the bundle, and reuses normal root preparation. - pub async fn prepare_recovered_overlay( - &self, - overlay: &RecoveryOverlay, - schema: u32, - ) -> Result { - if overlay.predecessor.cell != self.cell - || overlay.predecessor.incarnation != self.incarnation - || overlay.final_commit_sequence <= overlay.predecessor.commit_sequence - { - return Err(CrabError::InvalidState("recovery overlay scope")); - } - let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); - let final_position = overlay - .bundle - .rows() - .iter() - .rfind(|row| row.repository == repository && row.epoch == epoch) - .map(|row| row.info.position()) - .ok_or(CrabError::TxNotAvailable)?; - if final_position != overlay.final_position { - return Err(CrabError::ChecksumMismatch); - } - let prepared = self - .prepare_bundle( - Some(&overlay.predecessor), - &overlay.bundle, - overlay.final_commit_sequence, - schema, - ) - .await?; - if prepared.root().position != overlay.final_position { - return Err(CrabError::ChecksumMismatch); - } - Ok(prepared) - } - - async fn prepare_bundle_admitted( - &self, - base: Option<&RootRef>, - bundle: &crate::bundle::Bundle, - commit_sequence: u64, - schema: u32, - ) -> Result { - self.validate_metadata(commit_sequence, schema)?; - if bundle.len() > self.limits.max_plan_bytes { - return Err(CrabError::Limit(crate::LimitKind::CellBundleBytes)); - } - let base_graph = match base { - Some(root) => Some(self.load_graph(root).await?), - None => None, - }; - self.validate_append_sequence(&base_graph, commit_sequence)?; - - let (repository, epoch) = crate::bundle::cell_identity(&self.cell, &self.incarnation); - let bundle_digest = bundle.digest(); - let mut inputs = Vec::new(); - let mut selected_bytes = 0_u64; - let mut prospective = base_graph - .as_ref() - .map(|graph| graph.descriptors.clone()) - .unwrap_or_default(); - for (index, row) in bundle.rows().iter().enumerate() { - if row.repository != repository || row.epoch != epoch { - continue; - } - selected_bytes = selected_bytes - .checked_add(row.info.size_bytes) - .ok_or(CrabError::Limit(crate::LimitKind::CapturedCellBundleBytes))?; - if row.info.size_bytes > self.limits.max_file_bytes - || selected_bytes > self.limits.max_plan_bytes - { - return Err(CrabError::Limit(crate::LimitKind::CapturedCellBundleBytes)); - } - prospective.push(SegmentDescriptor::bundled( - row.info.clone(), - [0; 32], - 0, - bundle_digest, - row.offset, - )); - let bytes = bundle.read_segment(index)?; - let (file, size, digest, pages) = crate::ltx::inspect_bytes_with_index(&bytes)?; - if size != row.info.size_bytes - || digest != row.info.blake3 - || crate::SegmentInfo::from_inspected(&file, size, digest) != row.info - { - return Err(CrabError::ChecksumMismatch); - } - self.admit_segment_representation(&row.info, pages.len() * crate::paged::ENTRY_BYTES)?; - let index_bytes = Bytes::from(crate::paged::encode_index_from_pages(&pages)?); - inputs.push(AppendInput { - info: row.info.clone(), - location: BodyLocation::Bundle { - digest: bundle_digest, - offset: row.offset, - }, - index: index_bytes, - body: AppendBody::Bundle, - }); - } - let target = inputs - .last() - .map(|input| input.info.position()) - .ok_or(CrabError::TxNotAvailable)?; - self.validate_chain(&prospective, target)?; - self.prepare_append( - base, - base_graph, - inputs, - target, - commit_sequence, - schema, - Some(bundle), - ) - .await - } - - /// Prepares an exact representation-only compaction of a pinned root. - /// - /// The output retains the base TXID, checksum, commit sequence and schema. - /// Only the authority owner may later publish the proposal as a normal root CAS. - /// `scratch_directory` must already exist, be private to the caller and have - /// space for selected bodies and indexes plus compacted LTX/index outputs. - /// Owned scratch files are removed after success or failure. - pub async fn prepare_compaction( - &self, - base: &RootRef, - range: std::ops::Range, - level: u8, - scratch_directory: &Path, - ) -> Result { - let started = self.host.now_monotonic(); - let result = async { - let mut replica = self.clone(); - replica.host = self.host.for_recovery().await?; - let graph = replica.load_graph(base).await?; - let scratch_bytes = compaction_scratch_bytes(&graph, range.clone())?; - replica.host = replica.host.for_scratch(scratch_bytes).await?; - compaction::prepare(&replica, base, graph, range, level, scratch_directory).await - } - .await; - self.host - .observe_ltx_phase(crate::LtxPhase::Compaction, started, result.is_ok()); - result - } - - /// Prepares one bounded level promotion, or an emergency full compaction. - /// - /// Normal promotions require eight contiguous inputs from the preceding - /// level. A root near its segment or byte ceiling is compacted completely so - /// the next append cannot strand an otherwise healthy writer at admission. - pub async fn prepare_scheduled_compaction( - &self, - base: &RootRef, - scratch_directory: &Path, - ) -> Result> { - let started = self.host.now_monotonic(); - let result = self - .prepare_scheduled_compaction_inner(base, scratch_directory) - .await; - self.host - .observe_ltx_phase(crate::LtxPhase::Compaction, started, result.is_ok()); - result - } - - async fn prepare_scheduled_compaction_inner( - &self, - base: &RootRef, - scratch_directory: &Path, - ) -> Result> { - let mut replica = self.clone(); - replica.host = self.host.for_recovery().await?; - let graph = replica.load_graph(base).await?; - let segment_limit = MAX_SEGMENTS.min(replica.limits.max_segments); - let stored_bytes = graph - .descriptors - .iter() - .try_fold(0_u64, |total, descriptor| { - total - .checked_add(descriptor.info.size_bytes) - .and_then(|value| value.checked_add(descriptor.index_length)) - .ok_or(CrabError::Limit(crate::LimitKind::CellRootBytes)) - })?; - let byte_pressure = stored_bytes >= replica.limits.max_plan_bytes.saturating_mul(3) / 4; - let selected = if graph.descriptors.len() > 1 - && (graph.descriptors.len() >= segment_limit.saturating_sub(1).max(1) || byte_pressure) - { - let end = graph.descriptors.len(); - Some((0..end, 9)) - } else { - let mut selected = None; - for level in 1..=8 { - if let Some(range) = scheduled_compaction_range( - &graph.descriptors, - level, - replica.limits.max_file_bytes, - ) { - selected = Some((range, level)); - break; - } - } - selected - }; - let Some((range, level)) = selected else { - return Ok(None); - }; - let scratch_bytes = compaction_scratch_bytes(&graph, range.clone())?; - replica.host = replica.host.for_scratch(scratch_bytes).await?; - compaction::prepare(&replica, base, graph, range, level, scratch_directory) - .await - .map(Some) - } - - async fn prepare_append( - &self, - base: Option<&RootRef>, - base_graph: Option, - inputs: Vec, - target: Position, - commit_sequence: u64, - schema: u32, - bundle: Option<&crate::bundle::Bundle>, - ) -> Result { - let prepared = inputs - .into_iter() - .map(|input| { - let digest = *blake3::hash(&input.index).as_bytes(); - let descriptor = match input.location { - BodyLocation::Native => { - SegmentDescriptor::native(input.info, digest, input.index.len() as u64) - } - BodyLocation::Bundle { digest, offset } => SegmentDescriptor::bundled( - input.info, - *blake3::hash(&input.index).as_bytes(), - input.index.len() as u64, - digest, - offset, - ), - }; - Ok(PreparedSegment { - descriptor, - index: input.index, - body: input.body, - }) - }) - .collect::>>()?; - - let mut descriptors = base_graph - .as_ref() - .map(|graph| graph.descriptors.clone()) - .unwrap_or_default(); - descriptors.extend(prepared.iter().map(|segment| segment.descriptor.clone())); - self.validate_chain(&descriptors, target)?; - let directory_inputs = prepared - .iter() - .map(|segment| DirectoryInput { - descriptor: segment.descriptor.clone(), - index: segment.index.clone(), - }) - .collect::>(); - let dependency_uploads = async { - if let Some(bundle) = bundle { - self.put_bundle(bundle).await?; - } - stream::iter( - prepared - .into_iter() - .map(|segment| self.upload_prepared_segment(segment)), - ) - .buffered(SEGMENT_TRANSFER_CONCURRENCY) - .try_collect::>() - .await?; - Ok::<(), CrabError>(()) - }; - let root_preparation = self.finish_preparation( - base, - base_graph, - descriptors, - &directory_inputs, - target, - commit_sequence, - schema, - ); - // Content-addressed dependencies and root metadata can upload in - // parallel. The private proposal is returned only after both branches - // finish, so a failed branch can leave only unreachable objects. - let (_, prepared) = - futures_util::future::try_join(dependency_uploads, root_preparation).await?; - Ok(prepared) - } - - async fn finish_preparation( - &self, - base: Option<&RootRef>, - base_graph: Option, - descriptors: Vec, - directory_inputs: &[DirectoryInput], - target: Position, - commit_sequence: u64, - schema: u32, - ) -> Result { - for descriptor in &descriptors { - descriptor.validate_published(self.limits)?; - } - let endpoint = descriptors.last().ok_or(CrabError::LTXCorrupted)?; - let page_size = endpoint.info.page_size; - let database_pages = endpoint.info.database_pages; - let extents = object_extents(&descriptors)?; - let directory = if let Some(graph) = &base_graph { - let (changes, retain_through) = - directory_changes(directory_inputs, graph.document.database_pages)?; - let base_extents = object_extents(&graph.descriptors)?; - DirectoryTree::update( - directory::Verification { - layout: &self.layout, - cell: &self.cell, - incarnation: &self.incarnation, - page_size: graph.document.page_size, - database_pages: graph.document.database_pages, - extents: &base_extents, - host: &self.host, - origin: crate::LtxReadOrigin::Cold, - }, - graph.document.directory_digest, - graph.document.directory_height, - graph.aggregate, - changes, - retain_through, - directory::Verification { - layout: &self.layout, - cell: &self.cell, - incarnation: &self.incarnation, - page_size, - database_pages, - extents: &extents, - host: &self.host, - origin: crate::LtxReadOrigin::Cold, - }, - target.checksum, - ) - .await? - } else { - let entries = directory::initial_entries(directory_inputs)?; - let directory = directory::build_initial_and_upload( - stream::iter(entries), - page_size, - database_pages, - self, - ) - .await?; - if directory.checksum() != target.checksum { - return Err(CrabError::ChecksumMismatch); - } - directory - }; - let directory_uploads = self.put_objects( - CellObjectKind::Directory, - directory - .objects() - .iter() - .map(|node| (node.digest, node.bytes.clone())) - .collect(), - ); - let root_uploads = self.finish_root( - base, - descriptors, - target, - commit_sequence, - schema, - page_size, - database_pages, - directory, - ); - // Both object sets are immutable; no proposal escapes unless every - // upload succeeds, and a failed sibling leaves only unreachable data. - let (_, prepared) = futures_util::future::try_join(directory_uploads, root_uploads).await?; - Ok(prepared) - } - - #[expect(clippy::too_many_arguments)] - pub(super) async fn finish_root( - &self, - base: Option<&RootRef>, - descriptors: Vec, - target: Position, - commit_sequence: u64, - schema: u32, - page_size: u32, - database_pages: u32, - directory: DirectoryTree, - ) -> Result { - if directory.checksum() != target.checksum { - return Err(CrabError::ChecksumMismatch); - } - - let mut segment_pages = Vec::new(); - let mut root_objects = Vec::new(); - for page in descriptors.chunks(SEGMENTS_PER_PAGE) { - let bytes = encode_segment_page(page)?; - let digest = *blake3::hash(&bytes).as_bytes(); - root_objects.push((digest, bytes)); - segment_pages.push(digest); - } - if segment_pages.len() > MAX_SEGMENT_PAGES { - return Err(CrabError::Limit(crate::LimitKind::CellRootSegmentPages)); - } - let document = RootDocument { - cell: self.cell, - checksum: target.checksum, - commit_sequence, - database_pages, - directory_digest: directory.root_digest(), - directory_height: directory.height(), - incarnation: self.incarnation, - page_size, - schema, - segment_pages, - txid: target.txid, - }; - let bytes = encode_root(&document)?; - let digest = *blake3::hash(&bytes).as_bytes(); - root_objects.push((digest, bytes)); - // The document and its immutable segment pages can be uploaded in - // parallel. The root digest remains private until all uploads finish. - self.put_objects(CellObjectKind::Root, root_objects).await?; - let root = RootRef { - cell: self.cell, - incarnation: self.incarnation, - digest, - position: target, - commit_sequence, - }; - let host = self - .host - .clone() - .without_recovery() - .without_dirty() - .without_scratch(); - Ok(PreparedRoot { - predecessor: base.copied(), - verified: VerifiedRoot::from_graph( - self.clone().with_host(host), - root, - &document, - descriptors, - )?, - }) - } - - fn validate_metadata(&self, commit_sequence: u64, schema: u32) -> Result<()> { - if schema == 0 || commit_sequence > i64::MAX as u64 { - return Err(CrabError::InvalidState("invalid Cell root metadata")); - } - Ok(()) - } - - fn validate_append_sequence( - &self, - base: &Option, - commit_sequence: u64, - ) -> Result<()> { - if base - .as_ref() - .is_some_and(|graph| commit_sequence <= graph.document.commit_sequence) - { - return Err(CrabError::InvalidState("commit sequence did not advance")); - } - Ok(()) - } -} diff --git a/crates/crab-ltx/src/replica/read_only.rs b/crates/crab-ltx/src/replica/read_only.rs deleted file mode 100644 index 5e4423035..000000000 --- a/crates/crab-ltx/src/replica/read_only.rs +++ /dev/null @@ -1,69 +0,0 @@ -//! Owned read-only SQLite view of one verified immutable Cell root. - -use std::{ - path::Path, - sync::{Mutex, MutexGuard}, -}; - -use rusqlite::{Connection, OpenFlags}; - -use super::{RootRef, VerifiedRoot}; -use crate::{CrabError, Result, writable_vfs::Registration}; - -/// A read-only SQLite view backed by authenticated pages of one immutable root. -/// -/// Queries hold the connection lock through execution. Page bodies use the -/// host's bounded cache; Cell ownership and freshness remain the caller's duty. -pub struct ReadOnlyRoot { - root: RootRef, - connection: Option>, - registration: Registration, -} - -impl ReadOnlyRoot { - pub(super) fn open(root: &VerifiedRoot, destination: &Path) -> Result { - let database = crate::paged_io::Database::Snapshot(root.pages.clone()); - let registration = Registration::new(database, destination)?; - let flags = OpenFlags::SQLITE_OPEN_READ_ONLY | OpenFlags::SQLITE_OPEN_NO_MUTEX; - let connection = (|| -> Result { - let connection = - Connection::open_with_flags_and_vfs(destination, flags, registration.vfs())?; - crate::db::configure_managed_connection(&connection)?; - connection.pragma_update(None, "query_only", true)?; - Ok(connection) - })() - .map_err(|error| registration.take_error().unwrap_or(error))?; - Ok(Self { - root: root.root, - connection: Some(Mutex::new(connection)), - registration, - }) - } - - /// Takes the provider/checksum source behind a sparse SQLite I/O error. - pub fn take_io_error(&self) -> Option { - self.registration.take_error() - } - - /// Returns the immutable root this view reads. - #[must_use] - pub fn root(&self) -> RootRef { - self.root - } - - /// Locks the read-only SQLite connection for one bounded query. - pub fn connection(&self) -> Result> { - self.connection - .as_ref() - .ok_or(CrabError::InvalidState("read-only root is closed"))? - .lock() - .map_err(|_| CrabError::InvalidState("read-only root connection poisoned")) - } -} - -impl Drop for ReadOnlyRoot { - fn drop(&mut self) { - // SQLite must release its handle before registration removes the placeholder. - self.connection.take(); - } -} diff --git a/crates/crab-ltx/src/replica/restore.rs b/crates/crab-ltx/src/replica/restore.rs deleted file mode 100644 index 139b9fb07..000000000 --- a/crates/crab-ltx/src/replica/restore.rs +++ /dev/null @@ -1,260 +0,0 @@ -use std::{ - io, - path::{Path, PathBuf}, - sync::Arc, -}; - -use futures_util::{StreamExt as _, stream}; - -use super::{CellPagedDatabase, FetchedSpan, RESTORE_IN_FLIGHT_WINDOWS, RESTORE_WINDOW_BYTES}; -use crate::{CrabError, Host, Position, Result}; - -enum DownloadedWindow { - Lock, - Remote(Vec), -} - -pub(super) async fn run(database: &CellPagedDatabase, destination: &Path) -> Result { - let scratch_bytes = - crate::recovery::full_job_scratch_bytes(database.page_size, database.database_pages)?; - let host = database - .replica - .host - .for_recovery() - .await? - .for_scratch(scratch_bytes) - .await?; - crate::recovery::reject_sidecars(destination, &host)?; - if host.filesystem.exists(destination)? { - return Err(io::Error::new( - io::ErrorKind::AlreadyExists, - "restore destination already exists", - ) - .into()); - } - let (mut scratch, mut file) = RestoreScratch::create(&host, destination)?; - let mut database = database.clone(); - database.replica = database.replica.with_host(host.clone()); - let lock = crate::ltx::lock_pgno(database.page_size); - // Worst-case frame sizing keeps every prefetched window within 1 MiB; - // ordered installation decodes only one additional page at a time. - let pages_per_window = RESTORE_WINDOW_BYTES - .checked_div(crate::paged::maximum_frame_bytes(database.page_size)?) - .filter(|pages| *pages > 0) - .ok_or(CrabError::LTXCorrupted)?; - let database_pages = database.database_pages; - let page_size = database.page_size; - let mut next_page = Some(1); - let windows = std::iter::from_fn(move || { - let first = next_page?; - let last = if first == lock { - first - } else { - let mut last = first - .saturating_add(pages_per_window - 1) - .min(database_pages); - if first < lock { - last = last.min(lock - 1); - } - last - }; - next_page = if last == database_pages { - None - } else { - last.checked_add(1) - }; - Some((first, last - first + 1)) - }); - let download_database = database.clone(); - let mut downloads = stream::iter(windows.map(move |(first, count)| { - let database = download_database.clone(); - async move { - if first == lock { - Ok::<_, CrabError>((first, count, DownloadedWindow::Lock)) - } else { - Ok(( - first, - count, - DownloadedWindow::Remote(database.read_restore_window(first, count).await?), - )) - } - } - })) - .buffered(RESTORE_IN_FLIGHT_WINDOWS); - let mut checksum = crate::CHECKSUM_FLAG; - let mut expected_page = 1u64; - while let Some(window) = downloads.next().await { - let (first, count, downloaded) = window?; - if u64::from(first) != expected_page { - return Err(CrabError::LTXCorrupted); - } - let write_started = host.now_monotonic(); - let write = host - .run(move || { - let mut next = u64::from(first); - let end = next + u64::from(count); - let window_bytes = (count as usize) - .checked_mul(page_size as usize) - .filter(|bytes| *bytes <= RESTORE_WINDOW_BYTES as usize) - .ok_or(CrabError::LTXCorrupted)?; - let mut window = Vec::with_capacity(window_bytes); - match downloaded { - DownloadedWindow::Lock => { - window.resize(page_size as usize, 0); - next += 1; - } - DownloadedWindow::Remote(spans) => { - for span in spans { - span.try_for_each_page(page_size, |number, bytes| { - if u64::from(number) != next || next >= end { - return Err(CrabError::LTXCorrupted); - } - checksum = (checksum ^ crate::ltx::checksum_page(number, &bytes)) - | crate::CHECKSUM_FLAG; - window.extend_from_slice(&bytes); - next += 1; - Ok(()) - })?; - } - } - } - if next != end || window.len() != window_bytes { - return Err(CrabError::LTXCorrupted); - } - // Commit only fully verified windows to the private scratch file; - // the 1 MiB window bound also caps this coalescing buffer. - file.write_all(&window)?; - Ok::<_, CrabError>((file, checksum)) - }) - .await; - host.observe_ltx_phase( - crate::LtxPhase::RestoreWrite, - write_started, - matches!(&write, Ok(Ok(_))), - ); - let (next_file, next_checksum) = write??; - file = next_file; - checksum = next_checksum; - expected_page += u64::from(count); - } - if expected_page != u64::from(database.database_pages) + 1 { - return Err(CrabError::LTXCorrupted); - } - if checksum != database.position.checksum { - return Err(CrabError::ChecksumMismatch); - } - let expected_bytes = u64::from(database.page_size) * u64::from(database.database_pages); - let sync_started = host.now_monotonic(); - let sync = host - .run(move || { - if file.file_len()? != expected_bytes { - return Err(CrabError::LTXCorrupted); - } - file.sync_all()?; - Ok::<_, CrabError>(()) - }) - .await; - host.observe_ltx_phase( - crate::LtxPhase::RestoreWrite, - sync_started, - matches!(&sync, Ok(Ok(()))), - ); - sync??; - let filesystem = Arc::clone(&host.filesystem); - let source = scratch.path.clone(); - let destination = destination.to_owned(); - let install_started = host.now_monotonic(); - let install = host - .run(move || { - filesystem.persist_file_new(&source, &destination)?; - // A dispatched install can finish after its async waiter is - // cancelled. The returned guard removes that unclaimed file. - Ok::<_, CrabError>(InstalledRestore { - filesystem, - destination, - retained: false, - }) - }) - .await; - host.observe_ltx_phase( - crate::LtxPhase::RestoreWrite, - install_started, - matches!(&install, Ok(Ok(_))), - ); - let mut installed = install??; - installed.retained = true; - scratch.installed = true; - Ok(database.position) -} - -struct InstalledRestore { - filesystem: Arc, - destination: PathBuf, - retained: bool, -} - -impl Drop for InstalledRestore { - fn drop(&mut self) { - if !self.retained { - let _ = self.filesystem.remove_file(&self.destination); - } - } -} - -struct RestoreScratch { - filesystem: Arc, - path: PathBuf, - installed: bool, -} - -impl RestoreScratch { - fn create( - host: &Host, - destination: &Path, - ) -> Result<(Self, Box)> { - let parent = destination - .parent() - .filter(|path| !path.as_os_str().is_empty()) - .unwrap_or(Path::new(".")); - let filename = destination - .file_name() - .ok_or(CrabError::InvalidState("missing restore filename"))?; - static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); - for _ in 0..16 { - let mut scratch_name = filename.to_owned(); - scratch_name.push(format!( - ".crab-restore-{}-{}", - std::process::id(), - NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed) - )); - let path = parent.join(scratch_name); - match host.filesystem.create(&path) { - Ok(file) => { - return Ok(( - Self { - filesystem: Arc::clone(&host.filesystem), - path, - installed: false, - }, - file, - )); - } - Err(error) if error.kind() == io::ErrorKind::AlreadyExists => continue, - Err(error) => return Err(error.into()), - } - } - Err(io::Error::new( - io::ErrorKind::AlreadyExists, - "restore scratch namespace exhausted", - ) - .into()) - } -} - -impl Drop for RestoreScratch { - fn drop(&mut self) { - if !self.installed { - let _ = self.filesystem.remove_file(&self.path); - } - } -} diff --git a/crates/crab-ltx/src/replica/root.rs b/crates/crab-ltx/src/replica/root.rs deleted file mode 100644 index 9cd6c7c0b..000000000 --- a/crates/crab-ltx/src/replica/root.rs +++ /dev/null @@ -1,475 +0,0 @@ -use serde::{Deserialize, Serialize}; - -use crate::{CrabError, Limits, Result, SegmentInfo}; - -use super::{MAX_SEGMENT_PAGES, ROOT_BYTES, SEGMENT_PAGE_BYTES}; - -use crate::hex::encode_hex; - -#[derive(Clone)] -pub(super) struct RootDocument { - pub cell: [u8; 32], - pub checksum: u64, - pub commit_sequence: u64, - pub database_pages: u32, - pub directory_digest: [u8; 32], - pub directory_height: u32, - pub incarnation: [u8; 16], - pub page_size: u32, - pub schema: u32, - pub segment_pages: Vec<[u8; 32]>, - pub txid: u64, -} - -#[derive(Clone)] -pub(super) struct SegmentDescriptor { - pub info: SegmentInfo, - pub index_digest: [u8; 32], - pub index_length: u64, - object_digest: [u8; 32], - offset: u64, - length: u64, - level: u8, -} - -impl SegmentDescriptor { - pub(super) fn native(info: SegmentInfo, index_digest: [u8; 32], index_length: u64) -> Self { - let object_digest = info.blake3; - let length = info.size_bytes; - Self { - info, - index_digest, - index_length, - object_digest, - offset: 0, - length, - level: 0, - } - } - - pub(super) fn bundled( - info: SegmentInfo, - index_digest: [u8; 32], - index_length: u64, - object_digest: [u8; 32], - offset: u64, - ) -> Self { - let length = info.size_bytes; - Self { - info, - index_digest, - index_length, - object_digest, - offset, - length, - level: 0, - } - } - - pub(super) fn with_level(mut self, level: u8) -> Self { - self.level = level; - self - } - - pub(super) const fn level(&self) -> u8 { - self.level - } - - pub(super) const fn object_digest(&self) -> [u8; 32] { - self.object_digest - } - - pub(super) const fn offset(&self) -> u64 { - self.offset - } - - #[cfg(test)] - pub(super) const fn length(&self) -> u64 { - self.length - } - - pub(super) fn object_kind(&self) -> crate::CellObjectKind { - if self.object_digest == self.info.blake3 { - crate::CellObjectKind::Ltx - } else { - crate::CellObjectKind::Bundle - } - } - - pub(super) fn object_extent(&self) -> ([u8; 32], u64, u64, crate::CellObjectKind) { - ( - self.object_digest, - self.offset, - self.length, - self.object_kind(), - ) - } - - pub(super) fn validate(&self, limits: Limits) -> Result<()> { - let info = &self.info; - if self.level > 9 - || info.max_txid < info.min_txid - || info.database_pages == 0 - || !(512..=65536).contains(&info.page_size) - || !info.page_size.is_power_of_two() - || info.post_checksum & crate::CHECKSUM_FLAG == 0 - || info.size_bytes < 128 - || info.size_bytes > limits.max_file_bytes - || self.index_length > u64::from(info.database_pages) * 60 - || self.length != info.size_bytes - || self.offset.checked_add(self.length).is_none() - || (self.object_kind() == crate::CellObjectKind::Ltx && self.offset != 0) - || self - .offset - .checked_add(self.length) - .is_none_or(|end| end > limits.max_plan_bytes) - || u64::from(info.database_pages) * u64::from(info.page_size) - > limits.max_database_bytes - { - return Err(CrabError::LTXCorrupted); - } - Ok(()) - } - - pub(super) fn validate_published(&self, limits: Limits) -> Result<()> { - self.validate(limits)?; - if self.index_digest == [0; 32] || self.index_length == 0 || self.object_digest == [0; 32] { - return Err(CrabError::LTXCorrupted); - } - Ok(()) - } -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct RootWire { - cell: String, - checksum: String, - commit_sequence: String, - database_pages: u32, - directory_digest: String, - directory_height: u32, - incarnation: String, - page_size: u32, - schema: u32, - segment_pages: Vec, - txid: String, - version: u32, -} - -#[derive(Serialize, Deserialize)] -#[serde(deny_unknown_fields)] -struct SegmentWire { - blake3: String, - database_pages: u32, - index_digest: String, - index_length: String, - length: String, - level: u8, - max_txid: String, - min_txid: String, - object_digest: String, - offset: String, - page_size: u32, - post_checksum: String, - pre_checksum: String, - size_bytes: String, -} - -pub(super) fn encode_root(root: &RootDocument) -> Result> { - if root.schema == 0 - || root.txid == 0 - || root.checksum & crate::CHECKSUM_FLAG == 0 - || root.commit_sequence > i64::MAX as u64 - || root.database_pages == 0 - || root.segment_pages.is_empty() - || root.segment_pages.len() > MAX_SEGMENT_PAGES - { - return Err(CrabError::LTXCorrupted); - } - let bytes = serde_json::to_vec(&RootWire { - cell: encode_hex(&root.cell), - checksum: checksum(root.checksum), - commit_sequence: root.commit_sequence.to_string(), - database_pages: root.database_pages, - directory_digest: encode_hex(&root.directory_digest), - directory_height: root.directory_height, - incarnation: encode_hex(&root.incarnation), - page_size: root.page_size, - schema: root.schema, - segment_pages: root - .segment_pages - .iter() - .map(|value| encode_hex(value)) - .collect(), - txid: root.txid.to_string(), - version: 1, - })?; - if bytes.len() as u64 > ROOT_BYTES { - return Err(CrabError::Limit(crate::LimitKind::CellRootBytes)); - } - Ok(bytes) -} - -/// Decodes one root document and reports its descriptor-page count. -pub(crate) fn inspect_root(bytes: &[u8]) -> Result { - Ok(decode_root(bytes)?.segment_pages.len()) -} - -/// Decodes one root descriptor page and reports its descriptor count. -pub(crate) fn inspect_segment_page(bytes: &[u8]) -> Result { - Ok(decode_segment_page(bytes)?.len()) -} - -pub(super) fn decode_root(bytes: &[u8]) -> Result { - if bytes.len() as u64 > ROOT_BYTES { - return Err(CrabError::Limit(crate::LimitKind::CellRootBytes)); - } - let wire: RootWire = serde_json::from_slice(bytes)?; - if wire.version != 1 { - return Err(CrabError::LTXCorrupted); - } - let root = RootDocument { - cell: parse_hex(&wire.cell)?, - checksum: parse_checksum(&wire.checksum)?, - commit_sequence: decimal(&wire.commit_sequence)?, - database_pages: wire.database_pages, - directory_digest: parse_hex(&wire.directory_digest)?, - directory_height: wire.directory_height, - incarnation: parse_hex(&wire.incarnation)?, - page_size: wire.page_size, - schema: wire.schema, - segment_pages: wire - .segment_pages - .iter() - .map(|value| parse_hex(value)) - .collect::>>()?, - txid: decimal(&wire.txid)?, - }; - if encode_root(&root)? != bytes { - return Err(CrabError::LTXCorrupted); - } - Ok(root) -} - -pub(super) fn encode_segment_page(segments: &[SegmentDescriptor]) -> Result> { - let wire = segments - .iter() - .map(|segment| SegmentWire { - blake3: encode_hex(&segment.info.blake3), - database_pages: segment.info.database_pages, - index_digest: encode_hex(&segment.index_digest), - index_length: segment.index_length.to_string(), - length: segment.length.to_string(), - level: segment.level, - max_txid: segment.info.max_txid.to_string(), - min_txid: segment.info.min_txid.to_string(), - object_digest: encode_hex(&segment.object_digest), - offset: segment.offset.to_string(), - page_size: segment.info.page_size, - post_checksum: checksum(segment.info.post_checksum), - pre_checksum: checksum(segment.info.pre_checksum), - size_bytes: segment.info.size_bytes.to_string(), - }) - .collect::>(); - let bytes = serde_json::to_vec(&wire)?; - if bytes.len() as u64 > SEGMENT_PAGE_BYTES { - return Err(CrabError::Limit(crate::LimitKind::CellSegmentPageBytes)); - } - Ok(bytes) -} - -pub(super) fn decode_segment_page(bytes: &[u8]) -> Result> { - if bytes.len() as u64 > SEGMENT_PAGE_BYTES { - return Err(CrabError::Limit(crate::LimitKind::CellSegmentPageBytes)); - } - let wire: Vec = serde_json::from_slice(bytes)?; - let segments = wire - .into_iter() - .map(|segment| { - let object_digest = parse_hex(&segment.object_digest)?; - Ok(SegmentDescriptor { - info: SegmentInfo { - min_txid: decimal(&segment.min_txid)?, - max_txid: decimal(&segment.max_txid)?, - page_size: segment.page_size, - database_pages: segment.database_pages, - pre_checksum: parse_checksum(&segment.pre_checksum)?, - post_checksum: parse_checksum(&segment.post_checksum)?, - size_bytes: decimal(&segment.size_bytes)?, - blake3: parse_hex(&segment.blake3)?, - }, - index_digest: parse_hex(&segment.index_digest)?, - index_length: decimal(&segment.index_length)?, - object_digest, - offset: decimal(&segment.offset)?, - length: decimal(&segment.length)?, - level: segment.level, - }) - }) - .collect::>>()?; - if encode_segment_page(&segments)? != bytes { - return Err(CrabError::LTXCorrupted); - } - Ok(segments) -} - -fn decimal(value: &str) -> Result { - let parsed = value.parse::().map_err(|_| CrabError::LTXCorrupted)?; - if parsed.to_string() != value { - return Err(CrabError::LTXCorrupted); - } - Ok(parsed) -} - -fn checksum(value: u64) -> String { - format!("{value:016x}") -} - -fn parse_checksum(value: &str) -> Result { - if value.len() != 16 - || value - .bytes() - .any(|byte| !byte.is_ascii_digit() && !(b'a'..=b'f').contains(&byte)) - { - return Err(CrabError::LTXCorrupted); - } - u64::from_str_radix(value, 16).map_err(|_| CrabError::LTXCorrupted) -} - -fn parse_hex(value: &str) -> Result<[u8; N]> { - if value.len() != N * 2 - || value - .bytes() - .any(|byte| !byte.is_ascii_digit() && !(b'a'..=b'f').contains(&byte)) - { - return Err(CrabError::LTXCorrupted); - } - let mut output = [0; N]; - for (index, pair) in value.as_bytes().as_chunks::<2>().0.iter().enumerate() { - output[index] = (nibble(pair[0])? << 4) | nibble(pair[1])?; - } - Ok(output) -} - -fn nibble(value: u8) -> Result { - match value { - b'0'..=b'9' => Ok(value - b'0'), - b'a'..=b'f' => Ok(value - b'a' + 10), - _ => Err(CrabError::LTXCorrupted), - } -} - -#[cfg(test)] -mod tests { - use super::*; - - fn root() -> RootDocument { - RootDocument { - cell: [1; 32], - checksum: crate::CHECKSUM_FLAG | 7, - commit_sequence: 3, - database_pages: 9, - directory_digest: [2; 32], - directory_height: 0, - incarnation: [3; 16], - page_size: 4096, - schema: 1, - segment_pages: vec![[4; 32]], - txid: 2, - } - } - - #[test] - fn encode_root_accepts_the_segment_page_ceiling() { - let mut document = root(); - document.segment_pages = vec![[4; 32]; super::super::MAX_SEGMENT_PAGES]; - assert!(encode_root(&document).is_ok()); - } - - #[test] - fn encode_root_rejects_more_than_the_segment_page_ceiling() { - let mut document = root(); - document.segment_pages = vec![[4; 32]; super::super::MAX_SEGMENT_PAGES + 1]; - assert!(matches!( - encode_root(&document), - Err(CrabError::LTXCorrupted) - )); - } - - #[test] - fn decode_root_rejects_a_body_past_the_root_bound() { - let oversized = vec![b' '; super::super::ROOT_BYTES as usize + 1]; - assert!(matches!( - decode_root(&oversized), - Err(CrabError::Limit(crate::LimitKind::CellRootBytes)) - )); - } - - #[test] - fn decode_segment_page_rejects_a_body_past_the_page_bound() { - let oversized = vec![b' '; super::super::SEGMENT_PAGE_BYTES as usize + 1]; - assert!(matches!( - decode_segment_page(&oversized), - Err(CrabError::Limit(crate::LimitKind::CellSegmentPageBytes)) - )); - } - - #[test] - fn root_codec_rejects_noncanonical_and_duplicate_fields() { - let bytes = encode_root(&root()).unwrap(); - assert_eq!(decode_root(&bytes).unwrap().txid, 2); - let text = String::from_utf8(bytes).unwrap(); - assert!(decode_root(text.replace("\"txid\":\"2\"", "\"txid\":\"02\"").as_bytes()).is_err()); - assert!( - decode_root( - text.replace("\"schema\":1", "\"schema\":1,\"schema\":1") - .as_bytes() - ) - .is_err() - ); - } - - #[test] - fn maximum_descriptor_page_fits_the_wire_bound() { - let info = SegmentInfo { - min_txid: u64::MAX - 1, - max_txid: u64::MAX, - page_size: 65536, - database_pages: u32::MAX, - pre_checksum: u64::MAX, - post_checksum: u64::MAX, - size_bytes: u64::MAX, - blake3: [1; 32], - }; - let descriptor = SegmentDescriptor::native(info, [2; 32], u64::MAX); - let one = encode_segment_page(std::slice::from_ref(&descriptor)).unwrap(); - let page = vec![descriptor; super::super::SEGMENTS_PER_PAGE]; - assert!( - encode_segment_page(&page).is_ok(), - "one maximum descriptor occupies {} bytes", - one.len() - ); - } - - #[test] - fn bundled_descriptor_roundtrips_its_exact_extent() { - let info = SegmentInfo { - min_txid: 1, - max_txid: 2, - page_size: 4096, - database_pages: 9, - pre_checksum: 0, - post_checksum: crate::CHECKSUM_FLAG | 7, - size_bytes: 1024, - blake3: [1; 32], - }; - let descriptor = SegmentDescriptor::bundled(info, [2; 32], 60, [3; 32], 4096); - let decoded = decode_segment_page(&encode_segment_page(&[descriptor]).unwrap()).unwrap(); - let decoded = decoded.first().unwrap(); - assert_eq!(decoded.object_kind(), crate::CellObjectKind::Bundle); - assert_eq!(decoded.object_digest(), [3; 32]); - assert_eq!(decoded.offset(), 4096); - assert_eq!(decoded.length(), 1024); - } -} diff --git a/crates/crab-ltx/src/replica/upload.rs b/crates/crab-ltx/src/replica/upload.rs deleted file mode 100644 index 6bd9c4e6e..000000000 --- a/crates/crab-ltx/src/replica/upload.rs +++ /dev/null @@ -1,406 +0,0 @@ -//! Uploading prepared LTX artifacts, indexes, bodies, and bundles. -//! -//! Transfers use the replica's retry policy and shared I/O admission. Small -//! bodies are verified before a single PUT; larger bodies stream from replayable -//! sources, including pinned capture files. - -use super::*; - -impl CellReplica { - async fn put_object( - &self, - digest: &[u8; 32], - kind: CellObjectKind, - bytes: Vec, - ) -> Result<()> { - self.put_object_bytes(digest, kind, Bytes::from(bytes)) - .await - } - - pub(super) async fn put_object_bytes( - &self, - digest: &[u8; 32], - kind: CellObjectKind, - bytes: Bytes, - ) -> Result<()> { - if *blake3::hash(&bytes).as_bytes() != *digest { - return Err(CrabError::ChecksumMismatch); - } - let _permit = self.host.io_permit().await?; - let path = self - .layout - .incarnation_object_path(&self.cell, &self.incarnation, digest, kind); - self.layout.store().put(&path, bytes.clone()).await?; - self.cost.record(bytes.len() as u64); - if matches!(kind, CellObjectKind::Root | CellObjectKind::Directory) { - cache::insert( - &self.layout, - &self.cell, - &self.incarnation, - *digest, - kind, - bytes.to_vec().into(), - )?; - } - Ok(()) - } - - pub(super) async fn put_objects( - &self, - kind: CellObjectKind, - objects: Vec<([u8; 32], Vec)>, - ) -> Result<()> { - stream::iter( - objects - .into_iter() - .map(|(digest, bytes)| async move { self.put_object(&digest, kind, bytes).await }), - ) - .buffer_unordered(OBJECT_UPLOAD_CONCURRENCY) - .try_collect::>() - .await?; - Ok(()) - } - - pub(super) async fn upload_prepared_segment(&self, segment: PreparedSegment) -> Result<()> { - let PreparedSegment { - descriptor, - index, - body, - } = segment; - let body_upload = async { - if descriptor.object_kind() != CellObjectKind::Ltx { - return Ok(()); - } - let AppendBody::Native(source) = body else { - return Err(CrabError::InvalidState("native Cell body source missing")); - }; - compaction::upload_source( - self, - source, - descriptor.info.size_bytes, - &descriptor.info.blake3, - CellObjectKind::Ltx, - ) - .await - }; - let index_upload = - self.put_object_bytes(&descriptor.index_digest, CellObjectKind::Index, index); - futures_util::future::try_join(body_upload, index_upload).await?; - Ok(()) - } - - pub(super) async fn put_bundle(&self, bundle: &crate::bundle::Bundle) -> Result<()> { - let digest = bundle.digest(); - if bundle.len() > self.limits.max_plan_bytes { - return Err(CrabError::Limit(crate::LimitKind::CellBundleBytes)); - } - let path = self.layout.incarnation_object_path( - &self.cell, - &self.incarnation, - &digest, - CellObjectKind::Bundle, - ); - let staged = self.layout.incarnation_staging_path( - &self.cell, - &self.incarnation, - &digest, - CellObjectKind::Bundle, - ); - let _permit = self.host.io_permit().await?; - let upload = put_source(self, &staged, bundle.upload_source(), bundle.len(), digest).await; - if let Err(error) = upload { - return match cleanup_staged(self.layout.store(), &staged).await { - Ok(()) => Err(error), - Err(cleanup_error) => Err(cleanup_error), - }; - } - let promotion = self - .layout - .store() - .promote_staged_content_addressed_object(&staged, &path, digest, bundle.len()) - .await; - match cleanup_staged(self.layout.store(), &staged).await { - Err(error) => Err(error), - Ok(()) => promotion - .map(|_| self.cost.record(bundle.len())) - .map_err(Into::into), - } - } -} - -const MULTIPART_BYTES: usize = 8 << 20; -// Each caller holds a shared host I/O permit. Keep small-object buffering well -// below one multipart chunk so many publishing Cells stay within node memory. -const SINGLE_PUT_BYTES: u64 = 256 << 10; - -pub(super) async fn put_source( - replica: &CellReplica, - path: &object_store::path::Path, - source: Arc, - size: u64, - digest: [u8; 32], -) -> Result<()> { - if size <= SINGLE_PUT_BYTES { - if source.byte_len().await? != size { - return Err(CrabError::ChecksumMismatch); - } - let bytes = source.read_exact(0, size as usize).await?; - if bytes.len() as u64 != size - || *blake3::hash(&bytes).as_bytes() != digest - || source.byte_len().await? != size - { - return Err(CrabError::ChecksumMismatch); - } - // Freeze verified bytes before provider I/O. Exact writes retain staged - // routing and reconcile an uncertain create without overwriting a conflict. - replica.layout.store().put_exact(path, bytes).await?; - return Ok(()); - } - replica - .layout - .store() - .put_multipart_source_retry( - path, - source, - size, - digest, - MULTIPART_BYTES, - &tokio_util::sync::CancellationToken::new(), - None, - ) - .await?; - Ok(()) -} - -async fn cleanup_staged( - store: &crab_storage::Store, - path: &object_store::path::Path, -) -> Result<()> { - match store.delete(path).await { - Ok(()) | Err(crab_storage::StorageError::NotFound { .. }) => Ok(()), - Err(error) => Err(error.into()), - } -} - -pub(super) struct PinnedCapture { - host: Host, - file: Arc>>, - size: u64, -} - -impl PinnedCapture { - pub(super) async fn open(host: &Host, path: PathBuf, expected_size: u64) -> Result> { - let source_host = host.clone(); - let filesystem = Arc::clone(&host.filesystem); - host.run(move || { - let file = filesystem.open(&path)?; - if file.file_len()? != expected_size { - return Err(CrabError::ChecksumMismatch); - } - Ok(Arc::new(Self { - host: source_host, - file: Arc::new(Mutex::new(file)), - size: expected_size, - })) - }) - .await? - } - - fn read_exact(&self, offset: u64, length: usize) -> io::Result> { - let mut file = self - .file - .lock() - .map_err(|_| io::Error::other("capture file lock poisoned"))?; - file.read_exact_at(offset, length) - } -} - -pub(super) struct PinnedCaptureReader { - pub(super) source: Arc, - pub(super) offset: u64, -} - -impl io::Read for PinnedCaptureReader { - fn read(&mut self, bytes: &mut [u8]) -> io::Result { - let remaining = self.source.size.saturating_sub(self.offset); - let length = - usize::try_from(remaining.min(bytes.len() as u64)).map_err(io::Error::other)?; - if length == 0 { - return Ok(0); - } - let read = self.source.read_exact(self.offset, length)?; - bytes[..length].copy_from_slice(&read); - self.offset = self - .offset - .checked_add(length as u64) - .ok_or_else(|| io::Error::other("capture offset overflow"))?; - Ok(length) - } -} - -#[async_trait::async_trait] -impl crab_storage::MultipartUploadSource for PinnedCapture { - async fn byte_len(&self) -> crab_storage::Result { - let file = Arc::clone(&self.file); - self.host - .run(move || { - let file = file - .lock() - .map_err(|_| io::Error::other("capture file lock poisoned"))?; - file.file_len() - }) - .await - .map_err(pinned_storage_error)? - .map_err(|error| crab_storage::StorageError::ReadRejected { - source: Box::new(error), - }) - } - - async fn read_exact(&self, offset: u64, length: usize) -> crab_storage::Result { - let file = Arc::clone(&self.file); - self.host - .run(move || { - let mut file = file - .lock() - .map_err(|_| io::Error::other("capture file lock poisoned"))?; - file.read_exact_at(offset, length).map(Bytes::from) - }) - .await - .map_err(pinned_storage_error)? - .map_err(|error| crab_storage::StorageError::ReadRejected { - source: Box::new(error), - }) - } -} - -pub(super) async fn inspect_segment_source( - replica: &CellReplica, - source: Arc, - expected: &crate::SegmentInfo, -) -> Result> { - let expected = expected.clone(); - replica - .host - .run(move || { - let reader = PinnedCaptureReader { source, offset: 0 }; - let (file, size, digest, pages) = crate::ltx::inspect_reader_with_index(reader)?; - if size != expected.size_bytes || digest != expected.blake3 { - return Err(CrabError::ChecksumMismatch); - } - if crate::SegmentInfo::from_inspected(&file, size, digest) != expected { - return Err(CrabError::ChecksumMismatch); - } - crate::paged::encode_index_from_pages(&pages) - }) - .await? -} - -fn pinned_storage_error(error: CrabError) -> crab_storage::StorageError { - crab_storage::StorageError::ReadRejected { - source: Box::new(error), - } -} - -#[cfg(test)] -mod tests { - #[cfg(unix)] - use std::io::Read as _; - - use super::*; - - #[cfg(unix)] - #[tokio::test] - async fn pinned_capture_ignores_later_path_replacement() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("capture.ltx"); - let displaced = directory.path().join("original.ltx"); - let original = b"verified capture bytes"; - std::fs::write(&path, original).unwrap(); - let source = PinnedCapture::open(&Host::default(), path.clone(), original.len() as u64) - .await - .unwrap(); - - std::fs::rename(&path, &displaced).unwrap(); - std::fs::write(&path, b"replacement contents!").unwrap(); - - let mut reader = PinnedCaptureReader { source, offset: 0 }; - let mut observed = Vec::new(); - reader.read_to_end(&mut observed).unwrap(); - assert_eq!(observed, original); - } - - #[tokio::test] - async fn small_source_checks_bytes_before_provider_writes_and_never_clobbers() { - use crab_storage::{RetryPolicy, Store}; - use object_store::{ObjectStoreExt as _, memory::InMemory, path::Path}; - - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("capture"); - let bytes = b"verified"; - std::fs::write(&path, bytes).unwrap(); - let backend = Arc::new(InMemory::new()); - let store = Store::with_retry( - backend.clone(), - RetryPolicy { - max_attempts: 1, - base: std::time::Duration::ZERO, - cap: std::time::Duration::ZERO, - }, - ); - let replica = CellReplica::new( - CellStorageLayout::new(store, Path::from("test"), [1; 16]), - [2; 32], - [3; 16], - Limits::default(), - ) - .unwrap(); - let destination = Path::from("immutable.ltx"); - let source = PinnedCapture::open(&Host::default(), path.clone(), bytes.len() as u64) - .await - .unwrap(); - let digest = *blake3::hash(bytes).as_bytes(); - for (size, expected) in [ - (bytes.len() as u64 + 1, digest), - (bytes.len() as u64, [0; 32]), - ] { - assert!(matches!( - put_source(&replica, &destination, source.clone(), size, expected).await, - Err(CrabError::ChecksumMismatch) - )); - assert!(backend.head(&destination).await.is_err()); - } - std::fs::write(&path, b"short").unwrap(); - assert!(matches!( - put_source( - &replica, - &destination, - source.clone(), - bytes.len() as u64, - digest - ) - .await, - Err(CrabError::ChecksumMismatch) - )); - std::fs::write(&path, bytes).unwrap(); - backend - .put(&destination, Bytes::from_static(b"different").into()) - .await - .unwrap(); - assert!(matches!( - put_source(&replica, &destination, source, bytes.len() as u64, digest).await, - Err(CrabError::Storage( - crab_storage::StorageError::StateConflict { .. } - )) - )); - assert_eq!( - backend - .get(&destination) - .await - .unwrap() - .bytes() - .await - .unwrap(), - Bytes::from_static(b"different") - ); - } -} diff --git a/crates/crab-ltx/src/replica/verify.rs b/crates/crab-ltx/src/replica/verify.rs deleted file mode 100644 index f4bbf72d4..000000000 --- a/crates/crab-ltx/src/replica/verify.rs +++ /dev/null @@ -1,411 +0,0 @@ -//! Reopening and verifying an exact immutable root. -//! -//! Verification walks the root graph, authenticates each object against -//! its descriptor, and only then hands back a `VerifiedRoot` the replica -//! can restore from. - -use super::*; - -impl CellReplica { - /// Reopens and verifies an exact immutable root and its metadata graph. - /// - /// A same-store authenticated metadata cache may avoid body reads, while - /// origin HEADs check cached metadata presence. Use - /// [`Self::reachable_objects`] to authenticate current remote bytes and - /// inventory every dependency. - pub async fn open_root(&self, root: &RootRef) -> Result { - let graph = self.load_graph(root).await?; - VerifiedRoot::from_graph(self.clone(), *root, &graph.document, graph.descriptors) - } - - /// Verifies and returns the complete immutable dependency set for an exact root. - /// - /// Callers may use this bounded inventory for backup pinning and reachability - /// collection. A missing or corrupt dependency fails the traversal closed. - pub async fn reachable_objects(&self, root: &RootRef) -> Result> { - // Inventory must prove origin presence even for metadata uploaded here. - let graph = self.load_graph_with_cache(root, false).await?; - let extents = object_extents(&graph.descriptors)?; - let verification = directory::Verification { - layout: &self.layout, - cell: &self.cell, - incarnation: &self.incarnation, - page_size: graph.document.page_size, - database_pages: graph.document.database_pages, - extents: &extents, - host: &self.host, - origin: crate::LtxReadOrigin::Cold, - }; - let directory = directory::reachable_digests( - verification, - graph.document.directory_digest, - graph.document.directory_height, - graph.aggregate, - ) - .await?; - - let mut objects = std::collections::BTreeSet::new(); - let mut streamed = std::collections::BTreeMap::new(); - objects.insert(RootObjectRef { - digest: root.digest, - kind: CellObjectKind::Root, - }); - objects.extend( - graph - .document - .segment_pages - .iter() - .map(|digest| RootObjectRef { - digest: *digest, - kind: CellObjectKind::Root, - }), - ); - for descriptor in &graph.descriptors { - let body = RootObjectRef { - digest: descriptor.object_digest(), - kind: descriptor.object_kind(), - }; - let body_limit = match body.kind { - CellObjectKind::Bundle => self.limits.max_plan_bytes, - CellObjectKind::Ltx => self.limits.max_file_bytes, - _ => return Err(CrabError::LTXCorrupted), - }; - let body_length = - (body.kind == CellObjectKind::Ltx).then_some(descriptor.info.size_bytes); - if streamed - .insert(body, (body_limit, body_length)) - .is_some_and(|previous| previous != (body_limit, body_length)) - { - return Err(CrabError::LTXCorrupted); - } - let index = RootObjectRef { - digest: descriptor.index_digest, - kind: CellObjectKind::Index, - }; - if streamed - .insert( - index, - (self.limits.max_plan_bytes, Some(descriptor.index_length)), - ) - .is_some_and(|(_, length)| length != Some(descriptor.index_length)) - { - return Err(CrabError::LTXCorrupted); - } - } - for (object, (limit, length)) in &streamed { - self.verify_remote_object(*object, *limit, *length).await?; - } - objects.extend(streamed.into_keys()); - objects.extend(directory.into_iter().map(|digest| RootObjectRef { - digest, - kind: CellObjectKind::Directory, - })); - Ok(objects.into_iter().collect()) - } - - async fn verify_remote_object( - &self, - object: RootObjectRef, - max_bytes: u64, - expected_bytes: Option, - ) -> Result<()> { - let path = self.layout.incarnation_object_path( - &self.cell, - &self.incarnation, - &object.digest, - object.kind, - ); - let _permit = self.host.io_permit().await?; - let request = self.layout.store().get_stream(&path, None).await; - if request.is_err() { - self.host - .observe_ltx_origin_request(crate::LtxReadOrigin::Cold, false, 0); - } - let (metadata, _, mut stream) = request?; - if metadata.size > max_bytes || expected_bytes.is_some_and(|size| size != metadata.size) { - self.host - .observe_ltx_origin_request(crate::LtxReadOrigin::Cold, true, 0); - return Err(CrabError::LTXCorrupted); - } - let mut digest = blake3::Hasher::new(); - let mut read_bytes = 0usize; - loop { - match stream.try_next().await { - Ok(Some(chunk)) => { - digest.update(&chunk); - read_bytes = read_bytes.saturating_add(chunk.len()); - } - Ok(None) => break, - Err(error) => { - self.host.observe_ltx_origin_request( - crate::LtxReadOrigin::Cold, - false, - read_bytes, - ); - return Err(error.into()); - } - } - } - self.host - .observe_ltx_origin_request(crate::LtxReadOrigin::Cold, true, read_bytes); - if digest.finalize().as_bytes() != &object.digest { - return Err(CrabError::ChecksumMismatch); - } - Ok(()) - } - - pub(super) async fn load_graph(&self, root: &RootRef) -> Result { - self.load_graph_with_cache(root, true).await - } - - async fn load_graph_with_cache(&self, root: &RootRef, use_cache: bool) -> Result { - let started = self.host.now_monotonic(); - let result = self.load_graph_inner(root, use_cache).await; - self.host - .observe_ltx_phase(crate::LtxPhase::RootOpen, started, result.is_ok()); - result - } - - async fn load_graph_inner(&self, root: &RootRef, use_cache: bool) -> Result { - self.check_scope(root)?; - let (bytes, cached_root) = self - .read_object(&root.digest, CellObjectKind::Root, ROOT_BYTES, use_cache) - .await?; - if *blake3::hash(&bytes).as_bytes() != root.digest { - return Err(CrabError::ChecksumMismatch); - } - let document = decode_root(&bytes)?; - if document.cell != self.cell - || document.incarnation != self.incarnation - || document.txid != root.position.txid - || document.checksum != root.position.checksum - || document.commit_sequence != root.commit_sequence - { - return Err(CrabError::InvalidState("Cell root reference mismatch")); - } - if document.segment_pages.is_empty() || document.segment_pages.len() > MAX_SEGMENT_PAGES { - return Err(CrabError::LTXCorrupted); - } - let pages = stream::iter( - document - .segment_pages - .iter() - .copied() - .map(|digest| async move { - let (bytes, cached) = self - .read_object(&digest, CellObjectKind::Root, SEGMENT_PAGE_BYTES, use_cache) - .await?; - if *blake3::hash(&bytes).as_bytes() != digest { - return Err(CrabError::ChecksumMismatch); - } - let page = decode_segment_page(&bytes)?; - if page.is_empty() || page.len() > SEGMENTS_PER_PAGE { - return Err(CrabError::LTXCorrupted); - } - Ok((page, cached.then_some((digest, bytes.len())))) - }), - ) - .buffered(OBJECT_FETCH_CONCURRENCY) - .try_collect::>() - .await?; - let mut cached_metadata = Vec::new(); - if cached_root { - cached_metadata.push((root.digest, bytes.len())); - } - let mut descriptors = Vec::new(); - for (page, cached) in pages { - descriptors.extend(page); - cached_metadata.extend(cached); - } - // A cached predecessor cannot justify a new root if its metadata has - // disappeared from origin. Check all cached objects in one bounded wave. - self.verify_cached_metadata(&cached_metadata).await?; - self.validate_chain(&descriptors, root.position)?; - for descriptor in &descriptors { - descriptor.validate_published(self.limits)?; - } - let endpoint = descriptors.last().ok_or(CrabError::LTXCorrupted)?; - if document.page_size != endpoint.info.page_size - || document.database_pages != endpoint.info.database_pages - || document.schema == 0 - { - return Err(CrabError::LTXCorrupted); - } - let extents = object_extents(&descriptors)?; - let directory_started = self.host.now_monotonic(); - let aggregate = directory::verify_root( - directory::Verification { - layout: &self.layout, - cell: &self.cell, - incarnation: &self.incarnation, - page_size: document.page_size, - database_pages: document.database_pages, - extents: &extents, - host: &self.host, - origin: crate::LtxReadOrigin::Cold, - }, - document.directory_digest, - document.directory_height, - ) - .await; - self.host.observe_ltx_phase( - crate::LtxPhase::Directory, - directory_started, - aggregate.is_ok(), - ); - let aggregate = aggregate?; - if aggregate.checksum != document.checksum { - return Err(CrabError::ChecksumMismatch); - } - Ok(LoadedGraph { - aggregate, - document, - descriptors, - }) - } - - pub(super) fn validate_chain( - &self, - descriptors: &[SegmentDescriptor], - target: Position, - ) -> Result<()> { - if descriptors.is_empty() || descriptors.len() > MAX_SEGMENTS.min(self.limits.max_segments) - { - return Err(CrabError::Limit(crate::LimitKind::CellRootSegments)); - } - let mut previous = Position::default(); - let mut page_size = None; - let mut total = 0u64; - for descriptor in descriptors { - descriptor.validate(self.limits)?; - let info = &descriptor.info; - total = total - .checked_add(info.size_bytes) - .and_then(|value| value.checked_add(descriptor.index_length)) - .ok_or(CrabError::Limit(crate::LimitKind::CellRootBytes))?; - if total > self.limits.max_plan_bytes - || previous.txid.checked_add(1) != Some(info.min_txid) - || info.pre_checksum != previous.checksum - || page_size.is_some_and(|value| value != info.page_size) - { - return Err(CrabError::LTXCorrupted); - } - previous = info.position(); - page_size = Some(info.page_size); - } - if previous != target { - return Err(CrabError::ChecksumMismatch); - } - Ok(()) - } - - fn check_scope(&self, root: &RootRef) -> Result<()> { - if root.cell != self.cell || root.incarnation != self.incarnation { - return Err(CrabError::InvalidState( - "root belongs to another Cell incarnation", - )); - } - Ok(()) - } - - async fn read_object( - &self, - digest: &[u8; 32], - kind: CellObjectKind, - max_bytes: u64, - use_cache: bool, - ) -> Result<(Vec, bool)> { - // Hot roots use the same immutable-cache contract as directory nodes; - // the inventory path bypasses it to detect missing remote objects. - if use_cache - && kind == CellObjectKind::Root - && let Some(bytes) = - cache::get(&self.layout, &self.cell, &self.incarnation, *digest, kind)? - { - if bytes.len() as u64 > max_bytes { - return Err(CrabError::LTXCorrupted); - } - return Ok((bytes.to_vec(), true)); - } - let _permit = self.host.io_permit().await?; - let path = self - .layout - .incarnation_object_path(&self.cell, &self.incarnation, digest, kind); - let result = self - .layout - .store() - .get_with_etag_bounded(&path, max_bytes) - .await; - self.host.observe_ltx_origin_request( - crate::LtxReadOrigin::Cold, - result.is_ok(), - result.as_ref().map_or(0, |(bytes, _)| bytes.len()), - ); - let (bytes, _) = result?; - if kind == CellObjectKind::Root { - if *blake3::hash(&bytes).as_bytes() != *digest { - return Err(CrabError::ChecksumMismatch); - } - cache::insert( - &self.layout, - &self.cell, - &self.incarnation, - *digest, - kind, - bytes.to_vec().into(), - )?; - } - Ok((bytes.to_vec(), false)) - } - - async fn verify_cached_metadata(&self, objects: &[([u8; 32], usize)]) -> Result<()> { - stream::iter(objects.iter().copied().map(|(digest, size)| async move { - let _permit = self.host.io_permit().await?; - let path = self.layout.incarnation_object_path( - &self.cell, - &self.incarnation, - &digest, - CellObjectKind::Root, - ); - let result = self.layout.store().head(&path).await; - self.host - .observe_ltx_origin_request(crate::LtxReadOrigin::Cold, result.is_ok(), 0); - if result?.size != size as u64 { - return Err(CrabError::LTXCorrupted); - } - Ok(()) - })) - .buffer_unordered(OBJECT_FETCH_CONCURRENCY) - .try_collect::>() - .await?; - Ok(()) - } -} - -impl VerifiedRoot { - pub(super) fn from_graph( - replica: CellReplica, - root: RootRef, - document: &RootDocument, - descriptors: Vec, - ) -> Result { - let extents = object_extents(&descriptors)?; - Ok(Self { - root, - page_size: document.page_size, - database_pages: document.database_pages, - schema: document.schema, - segment_count: descriptors.len(), - directory_height: document.directory_height, - pages: CellPagedDatabase { - replica, - directory_digest: document.directory_digest, - directory_height: document.directory_height, - extents: Arc::new(extents), - page_size: document.page_size, - database_pages: document.database_pages, - position: root.position, - }, - }) - } -} diff --git a/crates/crab-ltx/src/resume.rs b/crates/crab-ltx/src/resume.rs deleted file mode 100644 index 8c7418668..000000000 --- a/crates/crab-ltx/src/resume.rs +++ /dev/null @@ -1,234 +0,0 @@ -//! Local resume records: the sidecars that make a closed database reusable. -//! -//! A resumed session continues the capture lineage of the file it opens, so the -//! database must hold exactly the state one published position names. The record -//! proves structure (page count, aggregate checksum, dense page checksums); it -//! never proves ownership, which stays with the runtime's own resume record and -//! the authoritative control. - -use std::path::{Path, PathBuf}; - -use crate::{CHECKSUM_FLAG, CrabError, Position, Result}; - -const CONTINUATION_MAGIC: [u8; 8] = *b"CRABLTC1"; -const CONTINUATION_VERSION: u32 = 1; -const CONTINUATION_LEN: usize = 36; -const CHECKSUM_FILE_SUFFIX: &str = ".crab-ltx-checksums"; -const CONTINUATION_FILE_SUFFIX: &str = ".crab-ltx-continuation"; - -/// One capture session's continuation, as recorded beside its database. -pub(crate) struct Continuation { - pub(crate) position: Position, - pub(crate) page_size: u32, - pub(crate) pages: u32, -} - -impl Continuation { - pub(crate) fn encode(&self) -> [u8; CONTINUATION_LEN] { - let mut bytes = [0u8; CONTINUATION_LEN]; - bytes[..8].copy_from_slice(&CONTINUATION_MAGIC); - bytes[8..12].copy_from_slice(&CONTINUATION_VERSION.to_be_bytes()); - bytes[12..20].copy_from_slice(&self.position.txid.to_be_bytes()); - bytes[20..28].copy_from_slice(&self.position.checksum.to_be_bytes()); - bytes[28..32].copy_from_slice(&self.page_size.to_be_bytes()); - bytes[32..36].copy_from_slice(&self.pages.to_be_bytes()); - bytes - } - - pub(crate) fn decode(bytes: &[u8]) -> Result { - if bytes.len() != CONTINUATION_LEN || bytes[..8] != CONTINUATION_MAGIC { - return Err(CrabError::InvalidState("unrecognized capture continuation")); - } - let version = u32::from_be_bytes( - bytes[8..12] - .try_into() - .map_err(|_| CrabError::InvalidState("capture continuation version"))?, - ); - if version != CONTINUATION_VERSION { - return Err(CrabError::InvalidState( - "unsupported capture continuation version", - )); - } - let position = Position { - txid: u64::from_be_bytes( - bytes[12..20] - .try_into() - .map_err(|_| CrabError::InvalidState("capture continuation txid"))?, - ), - checksum: u64::from_be_bytes( - bytes[20..28] - .try_into() - .map_err(|_| CrabError::InvalidState("capture continuation checksum"))?, - ), - }; - let page_size = u32::from_be_bytes( - bytes[28..32] - .try_into() - .map_err(|_| CrabError::InvalidState("capture continuation page size"))?, - ); - let pages = u32::from_be_bytes( - bytes[32..36] - .try_into() - .map_err(|_| CrabError::InvalidState("capture continuation page count"))?, - ); - if !crate::ltx::is_valid_page_size(page_size) - || pages == 0 - || position.txid == 0 - || position.checksum & CHECKSUM_FLAG == 0 - { - return Err(CrabError::InvalidState("invalid capture continuation")); - } - Ok(Self { - position, - page_size, - pages, - }) - } -} - -/// Returns the dense page-checksum sidecar for a database file. -pub(crate) fn checksum_path(database: &Path) -> PathBuf { - suffixed(database, CHECKSUM_FILE_SUFFIX) -} - -pub(crate) fn continuation_path(database: &Path) -> PathBuf { - suffixed(database, CONTINUATION_FILE_SUFFIX) -} - -fn suffixed(database: &Path, suffix: &str) -> PathBuf { - let mut path = database.as_os_str().to_owned(); - path.push(suffix); - path.into() -} - -pub(crate) fn wal_path(database: &Path) -> PathBuf { - suffixed(database, "-wal") -} - -fn journal_path(database: &Path) -> PathBuf { - suffixed(database, "-journal") -} - -/// Reads the continuation recorded beside `database`. -pub(crate) fn read_continuation(host: &crate::Host, database: &Path) -> Result { - let path = continuation_path(database); - let mut file = host.filesystem.open(&path)?; - if file.file_len()? != CONTINUATION_LEN as u64 { - return Err(CrabError::InvalidState("capture continuation length")); - } - Continuation::decode(&file.read_exact_at(0, CONTINUATION_LEN)?) -} - -fn remove_if_present(host: &crate::Host, path: &Path) -> Result<()> { - match host.filesystem.remove_file(path) { - Ok(()) => Ok(()), - Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(()), - Err(error) => Err(error.into()), - } -} - -/// Writes the dense checksum sidecar that a later open seeds from. -pub(crate) fn write_checksums( - host: &crate::Host, - database: &Path, - checksums: &crate::pages::PageChecksums, -) -> Result<()> { - let destination = checksum_path(database); - let scratch = scratch_path(&destination)?; - let mut file = host.filesystem.create(&scratch)?; - let written = checksums - .write_dense(file.as_mut()) - .and_then(|()| file.sync_all().map_err(Into::into)); - drop(file); - if let Err(error) = written { - let _ = host.filesystem.remove_file(&scratch); - return Err(error); - } - // The sidecar is a cache of the live index, so replacing it is safe only - // after the new file is complete and durable. - let installed = remove_if_present(host, &destination).and_then(|()| { - host.filesystem - .persist_file_new(&scratch, &destination) - .map_err(Into::into) - }); - if installed.is_err() { - let _ = host.filesystem.remove_file(&scratch); - } - installed -} - -/// Writes the continuation that a later [`crate::Db::open_resumed`] continues. -pub(crate) fn write_continuation( - host: &crate::Host, - database: &Path, - continuation: &Continuation, -) -> Result<()> { - let path = continuation_path(database); - remove_if_present(host, &path)?; - host.filesystem.persist_new(&path, &continuation.encode())?; - Ok(()) -} - -fn scratch_path(destination: &Path) -> Result { - let parent = destination - .parent() - .filter(|path| !path.as_os_str().is_empty()) - .unwrap_or(Path::new(".")); - let filename = destination - .file_name() - .ok_or(CrabError::InvalidState("missing checksum filename"))?; - static NEXT: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); - let mut name = filename.to_owned(); - name.push(format!( - ".crab-resume-{}-{}", - std::process::id(), - NEXT.fetch_add(1, std::sync::atomic::Ordering::Relaxed) - )); - Ok(parent.join(name)) -} - -/// Requires a database that a clean close left checkpointed and complete. -fn require_checkpointed(host: &crate::Host, database: &Path) -> Result<()> { - let wal = wal_path(database); - match host.filesystem.file_len(&wal) { - Ok(0) => {} - Ok(_) => { - return Err(CrabError::InvalidState( - "resume requires a checkpointed database", - )); - } - Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} - Err(error) => return Err(error.into()), - } - if host.filesystem.exists(&journal_path(database))? { - return Err(CrabError::InvalidState( - "resume requires a database without a rollback journal", - )); - } - Ok(()) -} - -/// Moves a resumable database and both sidecars onto a fresh path. -/// -/// The caller must already have matched the runtime's resume record against the -/// authoritative control: after this returns, the origin objects are never read. -pub(crate) fn move_resumed(source: &Path, destination: &Path, host: &crate::Host) -> Result<()> { - require_checkpointed(host, source)?; - let files = [source, &checksum_path(source), &continuation_path(source)]; - let installed = [ - destination, - &checksum_path(destination), - &continuation_path(destination), - ]; - files - .iter() - .zip(installed) - .try_for_each(|(from, to)| host.filesystem.rename(from, to).map_err(Into::into)) -} - -/// Removes a resumable database and both sidecars; missing files are ignored. -pub(crate) fn discard_resumed(database: &Path, host: &crate::Host) -> Result<()> { - remove_if_present(host, &continuation_path(database))?; - remove_if_present(host, &checksum_path(database))?; - remove_if_present(host, database) -} diff --git a/crates/crab-ltx/src/types.rs b/crates/crab-ltx/src/types.rs deleted file mode 100644 index 009880489..000000000 --- a/crates/crab-ltx/src/types.rs +++ /dev/null @@ -1,355 +0,0 @@ -// Contains adapted Celld lib.rs types at the revision in UPSTREAM.md. -// Apache-2.0; modified by Crab contributors. See LICENSE. -//! Shared LTX types: positions, limits, segment info, and read origins. - -use std::path::{Path, PathBuf}; - -use crate::{CrabError, Result}; - -/// An LTX position in a checksum-linked lineage, not a Git revision or owner epoch. -#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] -#[cfg_attr(feature = "replica", derive(serde::Serialize, serde::Deserialize))] -pub struct Position { - /// Transaction id of the commit this position names. - pub txid: u64, - /// Post-apply checksum that commit published. - pub checksum: u64, -} - -/// Admission bounds for local capture and recovery, not an RSS quota. -#[derive(Debug, Clone, Copy)] -pub struct Limits { - /// Largest database file local capture and recovery admit. - pub max_database_bytes: u64, - /// Largest single capture admitted, across every cut it publishes. - pub max_capture_bytes: u64, - /// Largest immutable LTX file admitted. - pub max_file_bytes: u64, - /// Largest recovery plan admitted. - pub max_plan_bytes: u64, - /// Largest number of segments one plan may reference. - pub max_segments: usize, -} - -impl Default for Limits { - fn default() -> Self { - Self { - max_database_bytes: 512 << 20, - max_capture_bytes: 64 << 20, - max_file_bytes: 512 << 20, - max_plan_bytes: 1 << 30, - max_segments: 1024, - } - } -} - -impl Limits { - pub(crate) fn validate(self) -> Result { - if self.max_database_bytes < 512 - || self.max_capture_bytes < 128 - || self.max_file_bytes < 128 - || self.max_segments == 0 - || self.max_capture_bytes > self.max_file_bytes - || self.max_file_bytes > self.max_plan_bytes - || self.max_plan_bytes > (usize::MAX / 8) as u64 - || self.max_database_bytes > (usize::MAX / 8) as u64 - { - return Err(CrabError::InvalidState("invalid resource limits")); - } - Ok(self) - } -} - -/// Immutable-file expectations to record in the server's authoritative manifest. -#[derive(Debug, Clone, PartialEq, Eq)] -#[cfg_attr(feature = "replica", derive(serde::Serialize, serde::Deserialize))] -pub struct SegmentInfo { - /// First transaction id the segment contains. - pub min_txid: u64, - /// Last transaction id the segment contains. - pub max_txid: u64, - /// SQLite page size the segment was captured with. - pub page_size: u32, - /// Database page count after the segment's commit. - pub database_pages: u32, - /// Checksum the database carried before the segment was applied. - pub pre_checksum: u64, - /// Checksum the segment published. - pub post_checksum: u64, - /// Exact byte length of the immutable segment file. - pub size_bytes: u64, - /// BLAKE3 digest of the immutable segment file. - pub blake3: [u8; 32], -} - -impl SegmentInfo { - /// Returns the position this segment publishes. - #[must_use] - pub fn position(&self) -> Position { - Position { - txid: self.max_txid, - checksum: self.post_checksum, - } - } - - #[cfg(all(test, feature = "replica"))] - pub(crate) fn from_decoded(bytes: &[u8], file: &crate::ltx::DecodedFile) -> Self { - Self::from_inspected(file, bytes.len() as u64, *blake3::hash(bytes).as_bytes()) - } - - pub(crate) fn from_inspected( - file: &crate::ltx::DecodedFile, - size_bytes: u64, - blake3: [u8; 32], - ) -> Self { - Self { - min_txid: file.header.min_txid.0, - max_txid: file.header.max_txid.0, - page_size: file.header.page_size, - database_pages: file.header.commit, - pre_checksum: file.header.pre_apply_checksum, - post_checksum: file.trailer.post_apply_checksum, - size_bytes, - blake3, - } - } -} - -/// A caller-selected local file plus manifest expectations; not yet verified. -#[derive(Clone)] -pub struct LocalSegment { - path: PathBuf, - info: SegmentInfo, - #[cfg(feature = "replica")] - captured_index: Option, -} - -impl std::fmt::Debug for LocalSegment { - fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - formatter - .debug_struct("LocalSegment") - .field("path", &self.path) - .field("info", &self.info) - .finish() - } -} - -impl LocalSegment { - /// Binds a caller-selected local file to the expectations it must satisfy. - #[must_use] - pub fn new(path: PathBuf, info: SegmentInfo) -> Self { - Self { - path, - info, - #[cfg(feature = "replica")] - captured_index: None, - } - } - /// Returns the local path of the segment file. - #[must_use] - pub fn path(&self) -> &Path { - &self.path - } - /// Returns the manifest expectations the file must satisfy. - #[must_use] - pub fn info(&self) -> &SegmentInfo { - &self.info - } - - #[cfg(feature = "replica")] - pub(crate) fn with_captured_index(mut self, index: Vec) -> Self { - self.captured_index = Some(index.into()); - self - } - - #[cfg(feature = "replica")] - pub(crate) fn captured_index(&self) -> Option { - self.captured_index.clone() - } -} - -/// All cuts produced by one capture, including checkpoint-boundary cuts. -#[derive(Debug, Clone)] -pub struct CaptureBatch { - /// Immutable cuts the capture published, in lineage order. - pub segments: Vec, - /// Position the batch's last cut publishes. - pub position: Position, - /// Bounded observations recorded while capturing. - pub timing: CaptureTiming, -} - -/// Bounded, in-memory observations for one capture operation. -/// -/// The ledger is not persisted and never participates in capture, checkpoint, -/// or fencing decisions. Durations are nanoseconds from the host's monotonic -/// clock; byte fields distinguish logical work, physical reads, and allocation. -#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] -pub struct CaptureTiming { - /// Total elapsed time for the capture phase represented by this batch. - pub total_nanos: u64, - /// Time spent preparing the managed capture before WAL synchronization. - pub preparation_nanos: u64, - /// Time spent validating the managed control-table schema. - pub schema_check_nanos: u64, - /// Time spent confirming that the WAL contains a readable frame. - pub wal_existence_nanos: u64, - /// Time spent resolving the current checksum-linked WAL position. - pub position_resolution_nanos: u64, - /// Time spent reading and parsing the WAL image. - pub wal_read_nanos: u64, - /// Time spent collecting the committed page map from the WAL image. - pub page_collection_nanos: u64, - /// Time spent validating WAL or produced LTX data. - pub verification_nanos: u64, - /// Time spent encoding LTX bytes and page records. - pub encode_nanos: u64, - /// Time spent writing LTX and temporary index bytes locally. - pub local_write_nanos: u64, - /// Time spent syncing completed LTX file contents. - pub fsync_nanos: u64, - /// Time spent publishing the LTX name and, for immediate captures, syncing - /// its parent directory. - pub parent_sync_nanos: u64, - /// Time spent in checkpoint maintenance associated with the capture. - pub checkpoint_nanos: u64, - /// Logical WAL bytes consumed by the capture. - pub wal_bytes: u64, - /// Largest physical WAL file length observed by the capture. - pub wal_file_bytes: u64, - /// Physical WAL bytes transferred into capture memory. - pub wal_read_bytes: u64, - /// Database bytes represented by the captured commit. - pub database_bytes: u64, - /// LTX bytes inspected for the returned segments. - pub ltx_bytes: u64, - /// Number of segments inspected for the returned batch. - pub segment_count: u32, - /// Number of sparse WAL image reads selected. - pub wal_sparse_reads: u32, - /// Number of complete WAL image reads selected. - pub wal_full_reads: u32, - /// Number of complete images selected before incremental WAL parsing. - pub wal_snapshot_reads: u32, - /// Number of sparse WAL reads that required a complete-image retry. - pub wal_fallback_reads: u32, - /// Peak allocated bytes in one WAL image used by this capture. - pub wal_image_bytes: u64, - /// Number of SQLite checkpoint pragmas executed. - pub checkpoint_runs: u32, - /// Number of checkpoint pragmas that reported a busy reader or writer. - pub checkpoint_busy: u32, - /// Number of checkpoint pragmas that failed with SQLITE_BUSY or SQLITE_LOCKED. - pub checkpoint_busy_errors: u32, - /// WAL frames reported by completed checkpoint pragmas. - pub checkpoint_frames: u64, - /// WAL frames backfilled by completed checkpoint pragmas. - pub checkpoint_backfilled: u64, - /// Number of checkpoints that restarted the WAL lineage. - pub checkpoint_restarts: u32, -} - -impl CaptureTiming { - pub(crate) fn merge(&mut self, other: Self) { - self.total_nanos = self.total_nanos.saturating_add(other.total_nanos); - self.preparation_nanos = self - .preparation_nanos - .saturating_add(other.preparation_nanos); - self.schema_check_nanos = self - .schema_check_nanos - .saturating_add(other.schema_check_nanos); - self.wal_existence_nanos = self - .wal_existence_nanos - .saturating_add(other.wal_existence_nanos); - self.position_resolution_nanos = self - .position_resolution_nanos - .saturating_add(other.position_resolution_nanos); - self.wal_read_nanos = self.wal_read_nanos.saturating_add(other.wal_read_nanos); - self.page_collection_nanos = self - .page_collection_nanos - .saturating_add(other.page_collection_nanos); - self.verification_nanos = self - .verification_nanos - .saturating_add(other.verification_nanos); - self.encode_nanos = self.encode_nanos.saturating_add(other.encode_nanos); - self.local_write_nanos = self - .local_write_nanos - .saturating_add(other.local_write_nanos); - self.fsync_nanos = self.fsync_nanos.saturating_add(other.fsync_nanos); - self.parent_sync_nanos = self - .parent_sync_nanos - .saturating_add(other.parent_sync_nanos); - self.checkpoint_nanos = self.checkpoint_nanos.saturating_add(other.checkpoint_nanos); - self.wal_bytes = self.wal_bytes.saturating_add(other.wal_bytes); - self.wal_file_bytes = self.wal_file_bytes.max(other.wal_file_bytes); - self.wal_read_bytes = self.wal_read_bytes.saturating_add(other.wal_read_bytes); - self.database_bytes = self.database_bytes.saturating_add(other.database_bytes); - self.ltx_bytes = self.ltx_bytes.saturating_add(other.ltx_bytes); - self.segment_count = self.segment_count.saturating_add(other.segment_count); - self.wal_sparse_reads = self.wal_sparse_reads.saturating_add(other.wal_sparse_reads); - self.wal_full_reads = self.wal_full_reads.saturating_add(other.wal_full_reads); - self.wal_snapshot_reads = self - .wal_snapshot_reads - .saturating_add(other.wal_snapshot_reads); - self.wal_fallback_reads = self - .wal_fallback_reads - .saturating_add(other.wal_fallback_reads); - self.wal_image_bytes = self.wal_image_bytes.max(other.wal_image_bytes); - self.checkpoint_runs = self.checkpoint_runs.saturating_add(other.checkpoint_runs); - self.checkpoint_busy = self.checkpoint_busy.saturating_add(other.checkpoint_busy); - self.checkpoint_busy_errors = self - .checkpoint_busy_errors - .saturating_add(other.checkpoint_busy_errors); - self.checkpoint_frames = self - .checkpoint_frames - .saturating_add(other.checkpoint_frames); - self.checkpoint_backfilled = self - .checkpoint_backfilled - .saturating_add(other.checkpoint_backfilled); - self.checkpoint_restarts = self - .checkpoint_restarts - .saturating_add(other.checkpoint_restarts); - } -} - -// Derived from Celld's position types; private to the imported codec/engine. -pub(crate) type Checksum = u64; -/// High bit LTX sets on every checksum it publishes. -/// -/// The flag is part of the wire format: it distinguishes a rolled checksum from -/// the zero value, so every reader that accepts an LTX checksum (including the -/// Cell runtime's control validation) must test the same bit. Exported so the -/// flag has one definition across the crate boundary. -pub const CHECKSUM_FLAG: u64 = 1 << 63; -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Default)] -pub(crate) struct Txid(pub u64); -impl std::fmt::Display for Txid { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "{:016x}", self.0) - } -} -#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] -pub(crate) struct Pos { - pub txid: Txid, - pub post_apply_checksum: u64, -} -impl Pos { - pub const ZERO: Self = Self { - txid: Txid(0), - post_apply_checksum: 0, - }; - pub fn new(txid: Txid, post_apply_checksum: u64) -> Self { - Self { - txid, - post_apply_checksum, - } - } -} -impl From for Position { - fn from(pos: Pos) -> Self { - Self { - txid: pos.txid.0, - checksum: pos.post_apply_checksum, - } - } -} diff --git a/crates/crab-ltx/src/wal.rs b/crates/crab-ltx/src/wal.rs deleted file mode 100644 index fdc03e2fb..000000000 --- a/crates/crab-ltx/src/wal.rs +++ /dev/null @@ -1,397 +0,0 @@ -// Derived from denoland/celld, commit 10cb1303dac710dcb3b557e318e08c855261f68b. -// Apache-2.0; see LICENSE and UPSTREAM.md. Modified by Crab contributors. - -use std::collections::{HashMap, HashSet}; - -use crate::{WAL_FRAME_HEADER_SIZE, WAL_HEADER_SIZE, wal_checksum}; - -const WAL_VERSION: u32 = 3_007_000; - -const WAL_MAGIC_LITTLE_ENDIAN: u32 = 0x377f_0682; - -const WAL_MAGIC_BIG_ENDIAN: u32 = 0x377f_0683; - -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum WalError { - Eof, - - InvalidMagic(u32), - InvalidPageSize(u32), - - UnsupportedVersion(u32), - - BufferSize { got: usize, want: u32 }, - - OffsetTooSmall { offset: i64, header_size: i64 }, - - UnalignedOffset { offset: i64, page_size: u32 }, - - PrevFrameMismatch, -} - -impl WalError { - #[inline] - pub fn is_eof(&self) -> bool { - matches!(self, WalError::Eof) - } -} - -impl std::fmt::Display for WalError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - // Matches the Go string "EOF" (io.EOF.CrabError()). - WalError::Eof => f.write_str("EOF"), - // Go: fmt.Errorf("invalid wal header magic: %x", magic) — lowercase - // hex, no leading zeros (matches Go's %x for a uint32). - WalError::InvalidPageSize(size) => write!(f, "invalid WAL page size: {size}"), - WalError::InvalidMagic(magic) => write!(f, "invalid wal header magic: {magic:x}"), - WalError::UnsupportedVersion(v) => write!(f, "unsupported wal version: {v}"), - WalError::BufferSize { got, want } => write!( - f, - "WALReader.ReadFrame(): buffer size ({got}) must match page size ({want})" - ), - WalError::OffsetTooSmall { - offset, - header_size, - } => write!( - f, - "offset ({offset}) must be greater than the wal header size ({header_size})" - ), - WalError::UnalignedOffset { offset, page_size } => { - write!(f, "unaligned wal offset {offset} for page size {page_size}") - } - WalError::PrevFrameMismatch => f.write_str("previous frame mismatch"), - } - } -} - -impl std::error::Error for WalError {} - -impl From for crate::CrabError { - fn from(e: WalError) -> Self { - crate::CrabError::Other(Box::new(e)) - } -} - -type WalResult = std::result::Result; - -#[derive(Debug)] -pub struct WalReader<'a> { - data: &'a [u8], - tail_base: usize, - frame_n: i64, - - big_endian: bool, - page_size: u32, - salt1: u32, - salt2: u32, - chksum1: u32, - chksum2: u32, -} - -impl<'a> WalReader<'a> { - pub fn new(data: &'a [u8]) -> WalResult { - let mut r = WalReader { - data, - tail_base: 0, - frame_n: 0, - big_endian: false, - page_size: 0, - salt1: 0, - salt2: 0, - chksum1: 0, - chksum2: 0, - }; - r.read_header()?; - Ok(r) - } - - pub fn new_with_offset(data: &'a [u8], offset: i64, salt1: u32, salt2: u32) -> WalResult { - // Must not start on the first page — we need to read the previous frame. - if offset <= WAL_HEADER_SIZE as i64 { - return Err(WalError::OffsetTooSmall { - offset, - header_size: WAL_HEADER_SIZE as i64, - }); - } - let mut r = Self::new_with_offset_inner(data, 0)?; - r.seek_to_offset(offset, salt1, salt2)?; - Ok(r) - } - - fn new_with_offset_inner(data: &'a [u8], tail_base: usize) -> WalResult { - let mut r = WalReader { - data, - tail_base, - frame_n: 0, - big_endian: false, - page_size: 0, - salt1: 0, - salt2: 0, - chksum1: 0, - chksum2: 0, - }; - // Read header to determine page size & byte order. - r.read_header()?; - Ok(r) - } - - fn seek_to_offset(&mut self, offset: i64, salt1: u32, salt2: u32) -> WalResult<()> { - self.salt1 = salt1; - self.salt2 = salt2; - - let frame_size = self.page_size as i64 + WAL_FRAME_HEADER_SIZE as i64; - if (offset - WAL_HEADER_SIZE as i64) % frame_size != 0 { - return Err(WalError::UnalignedOffset { - offset, - page_size: self.page_size, - }); - } - self.frame_n = (offset - WAL_HEADER_SIZE as i64) / frame_size; - - // Read the previous frame to load the running checksum. Any failure here - // (salt/checksum mismatch surfaces as WalError::Eof from read_frame_inner) - // means the previous frame doesn't match what we expect → mismatch. - self.frame_n -= 1; - let mut buf = vec![0u8; self.page_size as usize]; - if self.read_frame_inner(&mut buf, false).is_err() { - return Err(WalError::PrevFrameMismatch); - } - Ok(()) - } - - #[inline] - pub fn salt(&self) -> (u32, u32) { - (self.salt1, self.salt2) - } - - pub fn offset(&self) -> i64 { - if self.frame_n == 0 { - return 0; - } - WAL_HEADER_SIZE as i64 - + ((self.frame_n - 1) * (WAL_FRAME_HEADER_SIZE as i64 + self.page_size as i64)) - } - - fn read_at(&self, offset: i64, n: usize) -> Option<&'a [u8]> { - if offset < 0 { - return None; - } - let offset = offset as usize; - let start = if self.tail_base == 0 || offset < WAL_HEADER_SIZE { - offset - } else { - // A tail image holds nothing between the header and the tail; - // an offset in that gap is a short read, exactly as a whole-file - // image shorter than the offset would be. - WAL_HEADER_SIZE + offset.checked_sub(self.tail_base)? - }; - let end = start.checked_add(n)?; - if end > self.data.len() { - return None; - } - Some(&self.data[start..end]) - } - - pub fn new_with_offset_over_tail( - data: &'a [u8], - tail_base: i64, - offset: i64, - salt1: u32, - salt2: u32, - ) -> WalResult { - if tail_base < WAL_HEADER_SIZE as i64 || offset < tail_base { - return Err(WalError::OffsetTooSmall { - offset, - header_size: WAL_HEADER_SIZE as i64, - }); - } - let mut r = Self::new_with_offset_inner(data, tail_base as usize)?; - r.seek_to_offset(offset, salt1, salt2)?; - Ok(r) - } - - fn read_header(&mut self) -> WalResult<()> { - // If we have a partial WAL, mark WAL as done (io.EOF). - let hdr = match self.read_at(0, WAL_HEADER_SIZE) { - Some(b) => b, - None => return Err(WalError::Eof), - }; - - // Determine byte order of checksums from the magic (always read - // big-endian, like Go's binary.BigEndian.Uint32(hdr[0:])). - let magic = be_u32(&hdr[0..]); - self.big_endian = match magic { - WAL_MAGIC_LITTLE_ENDIAN => false, - WAL_MAGIC_BIG_ENDIAN => true, - _ => return Err(WalError::InvalidMagic(magic)), - }; - - // If the header checksum doesn't match then we may have failed with a - // partial WAL header write during checkpointing => io.EOF. - let chksum1 = be_u32(&hdr[24..]); - let chksum2 = be_u32(&hdr[28..]); - let (v0, v1) = wal_checksum(self.big_endian, 0, 0, &hdr[..24]); - if v0 != chksum1 || v1 != chksum2 { - return Err(WalError::Eof); - } - - // Verify version is correct. - let version = be_u32(&hdr[4..]); - if version != WAL_VERSION { - return Err(WalError::UnsupportedVersion(version)); - } - - self.page_size = be_u32(&hdr[8..]); - if !crate::ltx::is_valid_page_size(self.page_size) { - return Err(WalError::InvalidPageSize(self.page_size)); - } - self.salt1 = be_u32(&hdr[16..]); - self.salt2 = be_u32(&hdr[20..]); - self.chksum1 = chksum1; - self.chksum2 = chksum2; - - Ok(()) - } - - pub fn read_frame(&mut self, data: &mut [u8]) -> WalResult<(u32, u32)> { - self.read_frame_inner(data, true) - } - - fn read_frame_inner( - &mut self, - data: &mut [u8], - verify_checksum: bool, - ) -> WalResult<(u32, u32)> { - if data.len() != self.page_size as usize { - return Err(WalError::BufferSize { - got: data.len(), - want: self.page_size, - }); - } - - let frame_size = self.page_size as i64 + WAL_FRAME_HEADER_SIZE as i64; - let offset = WAL_HEADER_SIZE as i64 + (self.frame_n * frame_size); - - // Read WAL frame header. A short read is io.EOF. - let hdr = match self.read_at(offset, WAL_FRAME_HEADER_SIZE) { - Some(b) => b, - None => return Err(WalError::Eof), - }; - - // Read WAL page data. A short read is io.EOF. - let page = match self.read_at(offset + WAL_FRAME_HEADER_SIZE as i64, data.len()) { - Some(b) => b, - None => return Err(WalError::Eof), - }; - data.copy_from_slice(page); - - // Verify salt matches the salt in the header; otherwise end of valid WAL. - let salt1 = be_u32(&hdr[8..]); - let salt2 = be_u32(&hdr[12..]); - if self.salt1 != salt1 || self.salt2 != salt2 { - return Err(WalError::Eof); - } - - // Verify the cumulative checksum. If verification is disabled, it is - // because we are jumping to an offset and not checksumming from the - // beginning, so we simply adopt the frame's stored checksum. - let chksum1 = be_u32(&hdr[16..]); - let chksum2 = be_u32(&hdr[20..]); - if verify_checksum { - let (c0, c1) = wal_checksum(self.big_endian, self.chksum1, self.chksum2, &hdr[..8]); - let (c0, c1) = wal_checksum(self.big_endian, c0, c1, data); - self.chksum1 = c0; - self.chksum2 = c1; - if self.chksum1 != chksum1 || self.chksum2 != chksum2 { - return Err(WalError::Eof); - } - } else { - self.chksum1 = chksum1; - self.chksum2 = chksum2; - } - - let pgno = be_u32(&hdr[0..]); - let commit = be_u32(&hdr[4..]); - - self.frame_n += 1; - - Ok((pgno, commit)) - } - - pub fn page_map(&mut self) -> WalResult<(HashMap, i64, u32)> { - let mut m: HashMap = HashMap::new(); - let mut tx_map: HashMap = HashMap::new(); - let mut commit: u32 = 0; - let mut data = vec![0u8; self.page_size as usize]; - - loop { - let (pgno, fcommit) = match self.read_frame(&mut data) { - Ok(v) => v, - Err(e) if e.is_eof() => break, - Err(e) => return Err(e), - }; - - // Update latest offset for this page within the current transaction. - // Not promoted to the full map until the txn commits. - let offset = self.offset(); - tx_map.insert(pgno, offset); - - // On a commit record, transfer the txn offsets into the full map and - // record the new DB size. - if fcommit != 0 { - for (p, o) in tx_map.drain() { - m.insert(p, o); - } - commit = fcommit; - } - } - - // Remove pages that exceed the final commit size (DB shrank mid-WAL). - m.retain(|&pgno, _| pgno <= commit); - - // No complete transactions => original (zero) offset. - if m.is_empty() { - return Ok((m, 0, 0)); - } - - // Highest page offset, extended to the end of that frame. - let mut end: i64 = 0; - for &offset in m.values() { - if end == 0 || offset > end { - end = offset; - } - } - end += WAL_FRAME_HEADER_SIZE as i64 + self.page_size as i64; - - Ok((m, end, commit)) - } - - pub fn frame_salts_until(&self, until: (u32, u32)) -> HashSet<(u32, u32)> { - let mut m = HashSet::new(); - let step = WAL_FRAME_HEADER_SIZE as i64 + self.page_size as i64; - let mut offset = WAL_HEADER_SIZE as i64; - // The loop ends either when a frame-header read runs short (the Go - // `n != len(hdr)` => break) or when we reach the `until` salt below. - while let Some(hdr) = self.read_at(offset, WAL_FRAME_HEADER_SIZE) { - let salt1 = be_u32(&hdr[8..]); - let salt2 = be_u32(&hdr[12..]); - - // Track unique salts. - m.insert((salt1, salt2)); - - // Stop once we've seen the salt we were asked to read up to. - if salt1 == until.0 && salt2 == until.1 { - break; - } - - offset += step; - } - m - } -} - -#[inline] -fn be_u32(b: &[u8]) -> u32 { - u32::from_be_bytes([b[0], b[1], b[2], b[3]]) -} diff --git a/crates/crab-ltx/src/writable_vfs.rs b/crates/crab-ltx/src/writable_vfs.rs deleted file mode 100644 index 502894c0b..000000000 --- a/crates/crab-ltx/src/writable_vfs.rs +++ /dev/null @@ -1,804 +0,0 @@ -//! Sparse-file and immutable snapshot adaptation of Celld paged_vfs.rs; see UPSTREAM.md. -//! A static VFS and per-open Arc ownership keep SQLite discovery memory-safe. - -use crate::{CrabError, Result, paged_io::Io}; -use rusqlite::{Connection, ffi}; -use std::{ - collections::{HashMap, hash_map::Entry}, - ffi::{CStr, CString, c_char, c_int, c_void}, - path::{Path, PathBuf}, - sync::{Arc, Mutex, OnceLock, Weak}, -}; - -mod hydration; - -pub use hydration::{HydrationBatch, HydrationRead}; - -/// Progress resolving an inherited cut: locally materialized or superseded pages. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct Hydration { - /// Pages already materialized locally. - pub resolved: u32, - /// Pages the inherited cut covers. - pub total: u32, - /// Local page faults taken while hydrating. - pub faults: u64, -} - -impl Hydration { - /// Reports whether every page of the cut is materialized. - #[must_use] - pub fn complete(self) -> bool { - self.resolved == self.total - } -} - -struct State { - present: Vec, - ceiling: u32, - resolved: u32, - faults: u64, -} - -struct App { - io: Io, - page_size: u32, - count: u32, - local_disk: crate::DiskReservation, - read_only: bool, - state: Mutex, - error: Mutex>, -} - -fn views() -> &'static Mutex>> { - static VIEWS: OnceLock>>> = OnceLock::new(); - VIEWS.get_or_init(Mutex::default) -} - -struct ViewClaim { - path: PathBuf, - cleanup: Option, -} - -impl Drop for ViewClaim { - fn drop(&mut self) { - if let Some(host) = &self.cleanup { - // Keep the path claimed until this view's placeholder is removed. - let _ = host.filesystem.remove_file(&self.path); - } - if let Ok(mut registry) = views().lock() { - // The registry owns only discovery references. Releasing a claim - // cannot destroy an App or join its I/O worker under this lock. - registry.remove(&self.path); - } - } -} - -pub(crate) struct Registration { - _claim: ViewClaim, - app: Arc, - cursor: u32, - vfs: &'static str, -} - -impl Registration { - pub(crate) fn new(database: crate::paged_io::Database, path: &Path) -> Result { - let host = database.host(); - let vfs = register_for(host.sqlite_vfs.as_deref())?; - let parent = path - .parent() - .filter(|p| !p.as_os_str().is_empty()) - .unwrap_or(Path::new(".")); - let path = host.filesystem.canonicalize(parent)?.join( - path.file_name() - .ok_or(CrabError::InvalidState("missing database filename"))?, - ); - if path.to_str().is_none() { - return Err(CrabError::InvalidState("database path must be UTF-8")); - } - crate::recovery::reject_sidecars(&path, &host)?; - let count = database.page_count(); - let page_size = database.page_size(); - let mut claim = { - let mut registry = views() - .lock() - .map_err(|_| CrabError::InvalidState("sparse registry poisoned"))?; - let Entry::Vacant(entry) = registry.entry(path.clone()) else { - return Err(CrabError::InvalidState( - "sparse activation already registered", - )); - }; - // Claim the path before setup without making an incomplete App - // discoverable by SQLite. Failure drops only this setup's claim. - entry.insert(Weak::new()); - ViewClaim { - path, - cleanup: None, - } - }; - let path = &claim.path; - let read_only = database.read_only(); - // Writable setup failures leave a quarantined sparse file. Immutable - // snapshots own only a placeholder and remove it on every exit path. - let mut file = host.filesystem.create(path)?; - claim.cleanup = read_only.then(|| host.clone()); - if !read_only { - file.set_len(u64::from(count) * u64::from(page_size))?; - } - // These pages are derived from the pinned root. Durable warm reuse - // has its own handoff; syncing this placeholder adds no command durability. - drop(file); - let app = Arc::new(App { - io: Io::new(database)?, - page_size, - count, - local_disk: host.reserve_local_disk(0)?, - read_only, - state: Mutex::new(State { - present: if read_only { - Vec::new() - } else { - vec![false; count as usize] - }, - ceiling: count, - resolved: 0, - faults: 0, - }), - error: Mutex::new(None), - }); - views() - .lock() - .map_err(|_| CrabError::InvalidState("sparse registry poisoned"))? - .insert(path.clone(), Arc::downgrade(&app)); - Ok(Self { - _claim: claim, - app, - cursor: 1, - vfs, - }) - } - - pub(crate) fn vfs(&self) -> &'static str { - self.vfs - } - - pub(crate) fn hydration(&self) -> Result { - let state = self - .app - .state - .lock() - .map_err(|_| CrabError::InvalidState("sparse state poisoned"))?; - Ok(Hydration { - resolved: state.resolved, - total: self.app.count - - u32::from(crate::ltx::lock_pgno(self.app.page_size) <= self.app.count), - faults: state.faults, - }) - } - - pub(crate) fn step(&mut self, connection: &Connection, pages: u32) -> Result { - let start = self.hydration()?.resolved; - while self.cursor <= self.app.count && self.hydration()?.resolved - start < pages { - let page = self.cursor; - let present = self - .app - .state - .lock() - .map_err(|_| CrabError::InvalidState("sparse state poisoned"))? - .present[page as usize - 1]; - if !present && page != crate::ltx::lock_pgno(self.app.page_size) { - crate::db::read_main( - connection, - u64::from(page - 1) * u64::from(self.app.page_size), - self.app.page_size as usize, - ) - .map_err(|error| self.take_error().unwrap_or(error))?; - } - // Failed reads leave the cursor on the missing page, so retries - // cannot falsely declare hydration complete. - self.cursor += 1; - } - self.hydration() - } - - pub(crate) fn take_error(&self) -> Option { - self.app.error.lock().ok().and_then(|mut e| e.take()) - } -} - -#[repr(C)] -struct File { - methods: *const ffi::sqlite3_io_methods, - base: *mut ffi::sqlite3_file, - app: *const App, -} - -fn sqlite(rc: c_int) -> Result<()> { - if rc == ffi::SQLITE_OK { - Ok(()) - } else { - Err(rusqlite::Error::SqliteFailure(ffi::Error::new(rc), None).into()) - } -} - -fn mark(app: &App, state: &mut State, page: u32) { - if page == 0 || page > app.count || page == crate::ltx::lock_pgno(app.page_size) { - return; - } - if !state.present[page as usize - 1] { - state.present[page as usize - 1] = true; - state.resolved += 1; - } -} - -unsafe fn hydrate(file: *mut File, first: u32, last: u32) -> Result<()> { - // SAFETY: xOpen owns an Arc for this file until xClose. The base SQLite - // file remains open and its callbacks use SQLite's exact ABI. - unsafe { - let app = &*(*file).app; - for page in first..=last.min(app.count) { - if page == crate::ltx::lock_pgno(app.page_size) { - continue; - } - { - let state = app - .state - .lock() - .map_err(|_| CrabError::InvalidState("sparse state poisoned"))?; - if page > state.ceiling || state.present[page as usize - 1] { - continue; - } - } - let bytes = app.io.page(page)?; - install_page(file, page, &bytes)?; - } - Ok(()) - } -} - -unsafe fn install_page(file: *mut File, page: u32, bytes: &[u8]) -> Result<()> { - // SAFETY: the caller retains the live wrapper and its activation Arc while - // exclusively borrowing the owning SQLite connection. - unsafe { - let app = &*(*file).app; - let mut state = app - .state - .lock() - .map_err(|_| CrabError::InvalidState("sparse state poisoned"))?; - state.faults += 1; - // Owner writes and truncation can supersede pages while async fetch runs. - // Use the same gate as xWrite/xTruncate so inherited bytes never win. - if page > state.ceiling || state.present[page as usize - 1] { - return Ok(()); - } - if bytes.len() != app.page_size as usize { - return Err(CrabError::LTXCorrupted); - } - app.local_disk.try_grow(u64::from(app.page_size))?; - let base = (*file).base; - let write = (*(*base).pMethods) - .xWrite - .ok_or(CrabError::InvalidState("base VFS lacks xWrite"))?; - sqlite(write( - base, - bytes.as_ptr().cast(), - bytes.len() as c_int, - i64::from(page - 1) * i64::from(app.page_size), - ))?; - mark(app, &mut state, page); - Ok(()) - } -} - -fn page_range(amount: c_int, offset: i64, page_size: u32) -> Result<(u32, u32)> { - if amount <= 0 || offset < 0 { - return Err(CrabError::LTXCorrupted); - } - let end = offset - .checked_add(i64::from(amount) - 1) - .ok_or(CrabError::LTXCorrupted)?; - Ok(( - u32::try_from(offset / i64::from(page_size) + 1).map_err(|_| CrabError::LTXCorrupted)?, - u32::try_from(end / i64::from(page_size) + 1).map_err(|_| CrabError::LTXCorrupted)?, - )) -} - -unsafe fn guarded( - file: *mut File, - operation: impl FnOnce() -> Result<()>, - failure: c_int, -) -> c_int { - match std::panic::catch_unwind(std::panic::AssertUnwindSafe(operation)) { - Ok(Ok(())) => ffi::SQLITE_OK, - Ok(Err(error)) => { - // SAFETY: the file's Arc is live throughout this callback. - unsafe { - if !(*file).app.is_null() - && let Ok(mut slot) = (*(*file).app).error.lock() - { - *slot = Some(error); - } - } - failure - } - Err(_) => failure, - } -} - -fn read_snapshot(app: &App, offset: u64, output: &mut [u8]) -> Result { - let size = u64::from(app.count) * u64::from(app.page_size); - let available = size.saturating_sub(offset).min(output.len() as u64) as usize; - // SQLite requires a zero-filled tail on short reads, including reads at EOF. - output[available..].fill(0); - let mut copied = 0; - while copied < available { - let position = offset + copied as u64; - let page = (position / u64::from(app.page_size) + 1) as u32; - let within = (position % u64::from(app.page_size)) as usize; - let bytes = app.io.page(page)?; - if bytes.len() != app.page_size as usize { - return Err(CrabError::LTXCorrupted); - } - let count = (available - copied).min(bytes.len() - within); - output[copied..copied + count].copy_from_slice(&bytes[within..within + count]); - copied += count; - } - Ok(available < output.len()) -} - -unsafe extern "C" fn x_read( - file: *mut ffi::sqlite3_file, - buffer: *mut c_void, - amount: c_int, - offset: i64, -) -> c_int { - // SAFETY: SQLite supplies live file and amount bytes of writable storage. - unsafe { - let file = file.cast::(); - if amount == 0 { - return ffi::SQLITE_OK; - } - if !(*file).app.is_null() { - let app = &*(*file).app; - if app.read_only { - if amount < 0 || offset < 0 { - return ffi::SQLITE_IOERR_READ; - } - let output = std::slice::from_raw_parts_mut(buffer.cast::(), amount as usize); - let mut short = false; - let rc = guarded( - file, - || { - short = read_snapshot(app, offset as u64, output)?; - Ok(()) - }, - ffi::SQLITE_IOERR_READ, - ); - return if rc == ffi::SQLITE_OK && short { - ffi::SQLITE_IOERR_SHORT_READ - } else { - rc - }; - } - let rc = guarded( - file, - || { - let (first, last) = page_range(amount, offset, (*(*file).app).page_size)?; - hydrate(file, first, last) - }, - ffi::SQLITE_IOERR_READ, - ); - if rc != ffi::SQLITE_OK { - return rc; - } - } - let base = (*file).base; - match (*(*base).pMethods).xRead { - Some(call) => call(base, buffer, amount, offset), - None => ffi::SQLITE_IOERR_READ, - } - } -} - -unsafe extern "C" fn x_write( - file: *mut ffi::sqlite3_file, - buffer: *const c_void, - amount: c_int, - offset: i64, -) -> c_int { - // SAFETY: SQLite supplies a live file and amount initialized input bytes. - unsafe { - let file = file.cast::(); - let base = (*file).base; - if (*file).app.is_null() { - return match (*(*base).pMethods).xWrite { - Some(call) => call(base, buffer, amount, offset), - None => ffi::SQLITE_IOERR_WRITE, - }; - } - if (*(*file).app).read_only { - return ffi::SQLITE_READONLY; - } - if amount == 0 { - return ffi::SQLITE_OK; - } - guarded( - file, - || { - let app = &*(*file).app; - let (first, last) = page_range(amount, offset, app.page_size)?; - // SQLite may update only part of page 1. Resolve its untouched bytes - // before marking the page present; full-page writes need no fetch. - if !(offset as u64).is_multiple_of(u64::from(app.page_size)) { - hydrate(file, first, first)?; - } - if !(offset as u64 + amount as u64).is_multiple_of(u64::from(app.page_size)) { - hydrate(file, last, last)?; - } - let mut state = app - .state - .lock() - .map_err(|_| CrabError::InvalidState("sparse state poisoned"))?; - let missing = (first..=last.min(app.count)) - .filter(|page| { - *page != crate::ltx::lock_pgno(app.page_size) - && !state.present[*page as usize - 1] - }) - .count() as u64; - app.local_disk.try_grow( - missing - .checked_mul(u64::from(app.page_size)) - .ok_or(CrabError::Limit(crate::LimitKind::LocalDiskBytes))?, - )?; - let write = (*(*base).pMethods) - .xWrite - .ok_or(CrabError::InvalidState("base VFS lacks xWrite"))?; - sqlite(write(base, buffer, amount, offset))?; - for page in first..=last.min(app.count) { - mark(app, &mut state, page); - } - Ok(()) - }, - ffi::SQLITE_IOERR_WRITE, - ) - } -} - -unsafe extern "C" fn x_truncate(file: *mut ffi::sqlite3_file, size: i64) -> c_int { - // SAFETY: SQLite owns the open wrapper and its base file. - unsafe { - let file = file.cast::(); - let base = (*file).base; - let Some(truncate) = (*(*base).pMethods).xTruncate else { - return ffi::SQLITE_IOERR_TRUNCATE; - }; - if (*file).app.is_null() { - return truncate(base, size); - } - if (*(*file).app).read_only { - return ffi::SQLITE_READONLY; - } - guarded( - file, - || { - let app = &*(*file).app; - if size < 0 || !(size as u64).is_multiple_of(u64::from(app.page_size)) { - return Err(CrabError::LTXCorrupted); - } - let keep = u32::try_from(size / i64::from(app.page_size)) - .map_err(|_| CrabError::LTXCorrupted)?; - let mut state = app - .state - .lock() - .map_err(|_| CrabError::InvalidState("sparse state poisoned"))?; - sqlite(truncate(base, size))?; - // Old cut pages beyond a successful truncate are permanently - // superseded. Regrowth must never resurrect their old remote bytes. - for page in keep.saturating_add(1)..=state.ceiling { - mark(app, &mut state, page); - } - state.ceiling = state.ceiling.min(keep); - Ok(()) - }, - ffi::SQLITE_IOERR_TRUNCATE, - ) - } -} - -unsafe extern "C" fn x_close(file: *mut ffi::sqlite3_file) -> c_int { - // SAFETY: xOpen allocated the base with sqlite3_malloc and transferred one - // Arc into main files. SQLite invokes xClose once per successful open. - unsafe { - let file = file.cast::(); - let base = (*file).base; - let rc = match (*(*base).pMethods).xClose { - Some(call) => call(base), - None => ffi::SQLITE_OK, - }; - ffi::sqlite3_free(base.cast()); - if !(*file).app.is_null() { - drop(Arc::from_raw((*file).app)); - } - (*file).methods = std::ptr::null(); - rc - } -} - -macro_rules! forward_io { - ($name:ident, $field:ident, ($($arg:ident : $ty:ty),*), $failure:expr) => { - unsafe extern "C" fn $name(file: *mut ffi::sqlite3_file, $($arg:$ty),*) -> c_int { - // SAFETY: file wraps an open SQLite base; arguments preserve its ABI. - unsafe { let base = (*file.cast::()).base; - match (*(*base).pMethods).$field { Some(call) => call(base, $($arg),*), None => $failure } - } - } - }; -} -forward_io!(x_sync, xSync, (flags: c_int), ffi::SQLITE_IOERR_FSYNC); -unsafe extern "C" fn x_size(file: *mut ffi::sqlite3_file, size: *mut i64) -> c_int { - // SAFETY: SQLite supplies an open wrapper and writable output pointer. - unsafe { - let file = file.cast::(); - if !(*file).app.is_null() && (*(*file).app).read_only { - let app = &*(*file).app; - *size = i64::from(app.count) * i64::from(app.page_size); - return ffi::SQLITE_OK; - } - let base = (*file).base; - match (*(*base).pMethods).xFileSize { - Some(call) => call(base, size), - None => ffi::SQLITE_IOERR_FSTAT, - } - } -} -forward_io!(x_lock, xLock, (lock: c_int), ffi::SQLITE_IOERR_LOCK); -forward_io!(x_unlock, xUnlock, (lock: c_int), ffi::SQLITE_IOERR_UNLOCK); -forward_io!(x_reserved, xCheckReservedLock, (out: *mut c_int), ffi::SQLITE_IOERR_CHECKRESERVEDLOCK); -forward_io!(x_control, xFileControl, (op: c_int, arg: *mut c_void), ffi::SQLITE_NOTFOUND); -forward_io!(x_sector, xSectorSize, (), 4096); -forward_io!(x_shm_map, xShmMap, (page: c_int, size: c_int, extend: c_int, out: *mut *mut c_void), ffi::SQLITE_IOERR_SHMMAP); -forward_io!(x_shm_lock, xShmLock, (offset: c_int, n: c_int, flags: c_int), ffi::SQLITE_IOERR_SHMLOCK); -forward_io!(x_shm_unmap, xShmUnmap, (delete: c_int), ffi::SQLITE_IOERR); - -unsafe extern "C" fn x_characteristics(file: *mut ffi::sqlite3_file) -> c_int { - // SAFETY: the open file owns its registration's Arc until xClose. - unsafe { - let app = (*file.cast::()).app; - if !app.is_null() && (*app).read_only { - return ffi::SQLITE_IOCAP_IMMUTABLE; - } - } - // Sparse faults can write during reads. Do not inherit atomic/batch-write - // optimizations that can bypass the wrapper's bookkeeping. - 0 -} - -unsafe extern "C" fn x_shm_barrier(file: *mut ffi::sqlite3_file) { - // SAFETY: the wrapper's base is open and owns its shared-memory region. - unsafe { - let base = (*file.cast::()).base; - if let Some(call) = (*(*base).pMethods).xShmBarrier { - call(base); - } - } -} - -static METHODS: ffi::sqlite3_io_methods = ffi::sqlite3_io_methods { - iVersion: 2, - xClose: Some(x_close), - xRead: Some(x_read), - xWrite: Some(x_write), - xTruncate: Some(x_truncate), - xSync: Some(x_sync), - xFileSize: Some(x_size), - xLock: Some(x_lock), - xUnlock: Some(x_unlock), - xCheckReservedLock: Some(x_reserved), - xFileControl: Some(x_control), - xSectorSize: Some(x_sector), - xDeviceCharacteristics: Some(x_characteristics), - xShmMap: Some(x_shm_map), - xShmLock: Some(x_shm_lock), - xShmBarrier: Some(x_shm_barrier), - xShmUnmap: Some(x_shm_unmap), - xFetch: None, - xUnfetch: None, -}; - -unsafe extern "C" fn x_open( - vfs: *mut ffi::sqlite3_vfs, - name: *const c_char, - file: *mut ffi::sqlite3_file, - flags: c_int, - out: *mut c_int, -) -> c_int { - // SAFETY: SQLite provides zeroed szOsFile storage and a valid nullable name. - // The base VFS is a process-lifetime registration pinned in pAppData. - unsafe { - (*file).pMethods = std::ptr::null(); - let app = if flags & ffi::SQLITE_OPEN_MAIN_DB != 0 { - if name.is_null() { - return ffi::SQLITE_CANTOPEN; - } - let Ok(path) = CStr::from_ptr(name).to_str() else { - return ffi::SQLITE_CANTOPEN; - }; - let Ok(registry) = views().lock() else { - return ffi::SQLITE_CANTOPEN; - }; - let Some(app) = registry.get(Path::new(path)).and_then(Weak::upgrade) else { - return ffi::SQLITE_CANTOPEN; - }; - let mode = if app.read_only { - ffi::SQLITE_OPEN_READONLY - } else { - ffi::SQLITE_OPEN_READWRITE - }; - if flags & (ffi::SQLITE_OPEN_READONLY | ffi::SQLITE_OPEN_READWRITE) != mode { - return ffi::SQLITE_CANTOPEN; - } - Some(app) - } else { - None - }; - let base = (*vfs).pAppData.cast::(); - let Some(open) = (*base).xOpen else { - return ffi::SQLITE_CANTOPEN; - }; - let base_file = ffi::sqlite3_malloc((*base).szOsFile).cast::(); - if base_file.is_null() { - return ffi::SQLITE_NOMEM; - } - std::ptr::write_bytes(base_file.cast::(), 0, (*base).szOsFile as usize); - let rc = open(base, name, base_file, flags, out); - if rc != ffi::SQLITE_OK { - if !(*base_file).pMethods.is_null() - && let Some(close) = (*(*base_file).pMethods).xClose - { - close(base_file); - } - ffi::sqlite3_free(base_file.cast()); - return rc; - } - let wrapper = file.cast::(); - (*wrapper).base = base_file; - (*wrapper).app = app.map_or(std::ptr::null(), Arc::into_raw); - (*wrapper).methods = &METHODS; - ffi::SQLITE_OK - } -} - -macro_rules! forward_vfs { - ($name:ident, $field:ident, ($($arg:ident : $ty:ty),*), $failure:expr) => { - unsafe extern "C" fn $name(vfs: *mut ffi::sqlite3_vfs, $($arg:$ty),*) -> c_int { - // SAFETY: pAppData pins the host's process-lifetime base VFS; exact ABI. - unsafe { let base = (*vfs).pAppData.cast::(); - match (*base).$field { Some(call) => call(base, $($arg),*), None => $failure } - } - } - }; -} -forward_vfs!(x_delete, xDelete, (name: *const c_char, sync: c_int), ffi::SQLITE_IOERR_DELETE); -forward_vfs!(x_access, xAccess, (name: *const c_char, flags: c_int, out: *mut c_int), ffi::SQLITE_IOERR_ACCESS); -forward_vfs!(x_full_path, xFullPathname, (name: *const c_char, size: c_int, out: *mut c_char), ffi::SQLITE_CANTOPEN); -forward_vfs!(x_randomness, xRandomness, (size: c_int, out: *mut c_char), 0); -forward_vfs!(x_sleep, xSleep, (micros: c_int), 0); -forward_vfs!(x_current_time, xCurrentTime, (out: *mut f64), ffi::SQLITE_ERROR); - -fn register_for(base: Option<&str>) -> Result<&'static str> { - static REGISTRATIONS: OnceLock, &'static str>>> = OnceLock::new(); - let mut registrations = REGISTRATIONS - .get_or_init(Mutex::default) - .lock() - .map_err(|_| CrabError::InvalidState("sparse VFS registry poisoned"))?; - let key = base.map(str::to_owned); - if let Some(name) = registrations.get(&key) { - return Ok(name); - } - let name = match base { - None => "crab-ltx-writable-v1".to_owned(), - Some(_) => format!("crab-ltx-writable-v1-{}", registrations.len()), - }; - let c_name = - CString::new(name.as_str()).map_err(|_| CrabError::InvalidState("invalid VFS name"))?; - let c_base = base - .map(CString::new) - .transpose() - .map_err(|_| CrabError::InvalidState("invalid base VFS name"))?; - sqlite(register(c_base.as_deref(), &c_name))?; - // SQLite stores zName, not a copy. One registration per host VFS remains - // process-lifetime, so independent connections cannot outlive its memory. - let _ = c_name.into_raw(); - let name = Box::leak(name.into_boxed_str()); - registrations.insert(key, name); - Ok(name) -} - -fn register(base_name: Option<&CStr>, name: &CStr) -> c_int { - // SAFETY: optional callbacks are zeroed; mandatory version-1 callbacks are - // installed before registration. Both VFS structs remain process-lifetime. - unsafe { - let rc = ffi::sqlite3_initialize(); - if rc != ffi::SQLITE_OK { - return rc; - } - let base = ffi::sqlite3_vfs_find(base_name.map_or(std::ptr::null(), CStr::as_ptr)); - if base.is_null() || !ffi::sqlite3_vfs_find(name.as_ptr()).is_null() { - return ffi::SQLITE_ERROR; - } - let mut vfs: Box = Box::new(std::mem::zeroed()); - vfs.iVersion = 1; - vfs.szOsFile = std::mem::size_of::() as c_int; - vfs.mxPathname = (*base).mxPathname; - vfs.pAppData = base.cast(); - vfs.zName = name.as_ptr(); - vfs.xOpen = Some(x_open); - vfs.xDelete = Some(x_delete); - vfs.xAccess = Some(x_access); - vfs.xFullPathname = Some(x_full_path); - vfs.xRandomness = Some(x_randomness); - vfs.xSleep = Some(x_sleep); - vfs.xCurrentTime = Some(x_current_time); - let rc = ffi::sqlite3_vfs_register(&mut *vfs, 0); - if rc == ffi::SQLITE_OK { - let _ = Box::into_raw(vfs); - } - rc - } -} - -#[cfg(test)] -mod tests { - use super::page_range; - use crate::CrabError; - - #[test] - fn page_range_rejects_non_positive_amounts() { - assert!(matches!( - page_range(0, 0, 4096), - Err(CrabError::LTXCorrupted) - )); - assert!(matches!( - page_range(-1, 0, 4096), - Err(CrabError::LTXCorrupted) - )); - } - - #[test] - fn page_range_rejects_negative_offsets() { - assert!(matches!( - page_range(1, -1, 4096), - Err(CrabError::LTXCorrupted) - )); - } - - #[test] - fn page_range_keeps_a_partial_range_inside_one_page() { - assert_eq!(page_range(1, 0, 4096).unwrap(), (1, 1)); - assert_eq!(page_range(4096, 0, 4096).unwrap(), (1, 1)); - } - - #[test] - fn page_range_includes_both_ends_of_a_crossing_range() { - assert_eq!(page_range(1, 4096, 4096).unwrap(), (2, 2)); - assert_eq!(page_range(2, 4095, 4096).unwrap(), (1, 2)); - assert_eq!(page_range(4096, 4096, 4096).unwrap(), (2, 2)); - } - - #[test] - fn page_range_rejects_a_range_that_overflows_the_offset() { - assert!(matches!( - page_range(2, i64::MAX, 4096), - Err(CrabError::LTXCorrupted) - )); - } - - #[test] - fn page_range_rejects_a_page_number_past_the_cartesian_ceiling() { - let offset = 4096 * i64::from(u32::MAX); - assert!(matches!( - page_range(1, offset, 4096), - Err(CrabError::LTXCorrupted) - )); - } -} diff --git a/crates/crab-ltx/src/writable_vfs/hydration.rs b/crates/crab-ltx/src/writable_vfs/hydration.rs deleted file mode 100644 index a120109d7..000000000 --- a/crates/crab-ltx/src/writable_vfs/hydration.rs +++ /dev/null @@ -1,133 +0,0 @@ -//! Bounded asynchronous fetch and owner-thread installation of inherited pages. - -use super::*; - -/// A bounded fetch plan for missing pages of one sparse activation. -/// -/// Fetch outside the SQLite worker, then install its result on the same Db. -/// Dropping a plan or fetched batch leaves hydration progress unchanged. -pub struct HydrationRead { - app: Arc, - pages: Vec, - next: u32, -} - -/// Authenticated inherited pages awaiting installation on their owning Db. -pub struct HydrationBatch { - app: Arc, - pages: crate::paged_io::Pages, - next: u32, -} - -impl HydrationRead { - /// Returns the maximum page payload retained by the fetched batch. - #[must_use] - pub fn retained_bytes(&self) -> usize { - self.pages.len() * self.app.page_size as usize - } - - /// Fetches authenticated pages without accessing the SQLite connection. - /// - /// The caller bounds the wait and retains payload admission until the - /// resulting batch is installed or dropped. - pub async fn fetch(self) -> Result { - let mut pages = Vec::with_capacity(self.pages.len()); - let mut index = 0; - while let Some(&first) = self.pages.get(index) { - let mut count = 1; - while index + count < self.pages.len() - && self.pages.get(index + count) == first.checked_add(count as u32).as_ref() - { - count += 1; - } - let fetched = self.app.io.hydration_pages(first, count as u32).await?; - if fetched.is_empty() || fetched.len() > count { - return Err(CrabError::LTXCorrupted); - } - for (page, bytes) in fetched { - if self.pages.get(index) != Some(&page) - || bytes.len() != self.app.page_size as usize - { - return Err(CrabError::LTXCorrupted); - } - pages.push((page, bytes)); - index += 1; - } - } - Ok(HydrationBatch { - app: self.app, - pages, - next: self.next, - }) - } -} - -impl Registration { - pub(crate) fn prepare_hydration(&self, limit: u32) -> Result { - if !(1..=64).contains(&limit) { - return Err(CrabError::InvalidState( - "hydration batch must contain 1..64 pages", - )); - } - let state = self - .app - .state - .lock() - .map_err(|_| CrabError::InvalidState("sparse state poisoned"))?; - let mut pages = Vec::with_capacity(limit as usize); - let mut next = self.cursor; - while next <= self.app.count && pages.len() < limit as usize { - if next <= state.ceiling - && !state.present[next as usize - 1] - && next != crate::ltx::lock_pgno(self.app.page_size) - { - pages.push(next); - } - next = next.checked_add(1).ok_or(CrabError::LTXCorrupted)?; - } - Ok(HydrationRead { - app: self.app.clone(), - pages, - next, - }) - } - - pub(crate) fn install_hydration( - &mut self, - connection: &Connection, - batch: HydrationBatch, - ) -> Result { - if !Arc::ptr_eq(&self.app, &batch.app) { - return Err(CrabError::InvalidState( - "hydration batch belongs to another activation", - )); - } - let mut file: *mut ffi::sqlite3_file = std::ptr::null_mut(); - // SAFETY: the connection is exclusively borrowed on its owning worker. - // Check the wrapper and activation before using its base SQLite handle. - unsafe { - sqlite(ffi::sqlite3_file_control( - connection.handle(), - c"main".as_ptr(), - ffi::SQLITE_FCNTL_FILE_POINTER, - (&mut file as *mut *mut ffi::sqlite3_file).cast(), - ))?; - if file.is_null() || !std::ptr::eq((*file).pMethods, &METHODS) { - return Err(CrabError::InvalidState( - "sparse SQLite main file unavailable", - )); - } - let file = file.cast::(); - if !std::ptr::eq((*file).app, Arc::as_ptr(&self.app)) { - return Err(CrabError::InvalidState( - "hydration connection belongs to another activation", - )); - } - for (page, bytes) in batch.pages { - install_page(file, page, &bytes)?; - } - } - self.cursor = self.cursor.max(batch.next); - self.hydration() - } -} diff --git a/crates/crab-ltx/tests-allow-list.txt b/crates/crab-ltx/tests-allow-list.txt deleted file mode 100644 index 2851f2b4b..000000000 --- a/crates/crab-ltx/tests-allow-list.txt +++ /dev/null @@ -1,20 +0,0 @@ -capture/wal.rs # asserts private WAL encoding and index handling -capture/timing.rs # asserts the private capture timing recorder -cell_layout.rs # asserts private cell layout helpers -codec.rs # asserts private LTX codec internals -db/tests.rs # drives the private managed database and admissions -db.rs # declares crate-private test modules -environment/directory_cache.rs # declares crate-private test modules -environment/tests.rs # drives the private disk budget, cache, and executor hooks -environment.rs # declares crate-private test modules -error.rs # asserts private error classification -format_tests.rs # independent literal LTX vectors over private encoders -host.rs # asserts private host file helpers -paged_io.rs # drives private sparse page I/O -pages.rs # asserts private page checksum handling -replica/cache.rs # asserts the private directory cache -replica/directory/tests.rs # asserts private radix directory internals -replica/directory.rs # declares crate-private test modules -replica/root.rs # asserts private root codec bounds -replica/upload.rs # asserts private capture upload sources -writable_vfs.rs # asserts private page-range arithmetic diff --git a/crates/crab-ltx/tests/cell.rs b/crates/crab-ltx/tests/cell.rs deleted file mode 100644 index 9ac5f1cdb..000000000 --- a/crates/crab-ltx/tests/cell.rs +++ /dev/null @@ -1,10 +0,0 @@ -//! Cell-scoped LTX replication tests: exact roots, parallel restore, local replication. - -#![cfg(feature = "replica")] - -mod cell { - pub mod bundle; - pub mod replication; - pub mod restore; - pub mod roots; -} diff --git a/crates/crab-ltx/tests/cell/bundle.rs b/crates/crab-ltx/tests/cell/bundle.rs deleted file mode 100644 index f4db0d0b0..000000000 --- a/crates/crab-ltx/tests/cell/bundle.rs +++ /dev/null @@ -1,119 +0,0 @@ -//! Bundle envelope fences: magic, layout, identity, and limit rejects. - -use std::path::Path; - -use crab_ltx::bundle::{Bundle, BundleBuilder, BundleEntry, BundleRow}; -use crab_ltx::{Db, Limits, SegmentInfo}; - -fn segment(directory: &Path, marker: u8) -> (SegmentInfo, Vec) { - let database_path = directory.join(format!("database-{marker}")); - let mut database = Db::open(&database_path, Limits::default()).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - database.capture().unwrap(); - database - .transaction(|transaction| { - transaction.execute("INSERT INTO values_ VALUES (1)", [])?; - Ok(()) - }) - .unwrap(); - let capture = database.capture().unwrap(); - let segment = capture.segments.first().unwrap(); - let info = segment.info().clone(); - let bytes = std::fs::read(segment.path()).unwrap(); - database.close().unwrap(); - (info, bytes) -} - -fn row_info(min_txid: u64, max_txid: u64, size_bytes: u64) -> SegmentInfo { - SegmentInfo { - min_txid, - max_txid, - page_size: 4096, - database_pages: 1, - pre_checksum: 0, - post_checksum: 0, - size_bytes, - blake3: [0; 32], - } -} - -fn envelope(rows: &[BundleRow], payload: Vec) -> Vec { - let footer = serde_json::to_vec(rows).unwrap(); - let mut bytes = payload; - bytes.extend_from_slice(&footer); - bytes.extend_from_slice(&u32::try_from(footer.len()).unwrap().to_le_bytes()); - bytes.extend_from_slice(b"CRB1"); - bytes -} - -#[test] -fn decode_rejects_a_body_without_the_crb1_magic() { - assert!(Bundle::decode(vec![0; 16], Limits::default()).is_err()); -} - -#[test] -fn decode_rejects_an_empty_body() { - assert!(Bundle::decode(Vec::new(), Limits::default()).is_err()); -} - -#[test] -fn decode_rejects_rows_that_do_not_cover_the_payload() { - let rows = [BundleRow { - repository: "aa".to_string(), - epoch: "bb".to_string(), - info: row_info(1, 1, 8), - offset: 0, - }]; - let bytes = envelope(&rows, vec![0; 16]); - assert!(Bundle::decode(bytes, Limits::default()).is_err()); -} - -#[test] -fn decode_rejects_overlapping_rows() { - let rows = [ - BundleRow { - repository: "aa".to_string(), - epoch: "bb".to_string(), - info: row_info(1, 1, 8), - offset: 0, - }, - BundleRow { - repository: "aa".to_string(), - epoch: "bb".to_string(), - info: row_info(2, 2, 8), - offset: 4, - }, - ]; - let bytes = envelope(&rows, vec![0; 16]); - assert!(Bundle::decode(bytes, Limits::default()).is_err()); -} - -#[test] -fn encode_rejects_a_duplicate_segment_identity() { - let directory = tempfile::tempdir().unwrap(); - let (info, bytes) = segment(directory.path(), 1); - let entry = || BundleEntry::for_cell([1; 32], [1; 16], info.clone(), bytes.clone()); - assert!(Bundle::encode(vec![entry(), entry()], Limits::default()).is_err()); -} - -#[test] -fn builder_rejects_more_than_the_segment_limit() { - let directory = tempfile::tempdir().unwrap(); - let limits = Limits { - max_segments: 1, - ..Limits::default() - }; - let mut builder = BundleBuilder::new_temp(directory.path(), limits).unwrap(); - let (info, bytes) = segment(directory.path(), 2); - builder - .push(BundleEntry::for_cell([2; 32], [2; 16], info, bytes)) - .unwrap(); - let (info, bytes) = segment(directory.path(), 3); - assert!( - builder - .push(BundleEntry::for_cell([3; 32], [3; 16], info, bytes)) - .is_err() - ); -} diff --git a/crates/crab-ltx/tests/cell/replication.rs b/crates/crab-ltx/tests/cell/replication.rs deleted file mode 100644 index 7ca4e302e..000000000 --- a/crates/crab-ltx/tests/cell/replication.rs +++ /dev/null @@ -1,320 +0,0 @@ -use crab_ltx::rusqlite::Connection; -use crab_ltx::{CrabError, Db, Limits, LocalSegment, VerifiedPlan, compact_exact, restore_exact}; -use tempfile::TempDir; - -fn insert(db: &mut Db, value: &str) { - db.transaction(|tx| { - tx.execute( - "CREATE TABLE IF NOT EXISTS messages (id INTEGER PRIMARY KEY, body TEXT NOT NULL)", - [], - )?; - tx.execute("INSERT INTO messages(body) VALUES (?1)", [value])?; - Ok(()) - }) - .unwrap(); -} - -fn rows(path: &std::path::Path) -> Vec { - let conn = Connection::open(path).unwrap(); - let integrity: String = conn - .query_row("PRAGMA integrity_check", [], |row| row.get(0)) - .unwrap(); - assert_eq!(integrity, "ok"); - conn.prepare("SELECT body FROM messages ORDER BY id") - .unwrap() - .query_map([], |row| row.get(0)) - .unwrap() - .collect::, _>>() - .unwrap() -} - -#[test] -fn committed_sql_survives_cold_restore_and_compaction() { - let source = TempDir::new().unwrap(); - let remote = TempDir::new().unwrap(); - let restore = TempDir::new().unwrap(); - let limits = Limits::default(); - let mut db = Db::open(&source.path().join("repo.sqlite"), limits).unwrap(); - let mut files = Vec::new(); - let mut position = Default::default(); - for text in ["initial", "second", "Unicode: 🦀 中文"] { - insert(&mut db, text); - let batch = db.capture().unwrap(); - assert!( - batch - .segments - .iter() - .all(|file| file.info().post_checksum != 0) - ); - position = batch.position; - for file in batch.segments { - let path = remote.path().join(file.path().file_name().unwrap()); - std::fs::copy(file.path(), &path).unwrap(); - files.push(LocalSegment::new(path, file.info().clone())); - } - } - db.close().unwrap(); - source.close().unwrap(); - let plan = VerifiedPlan::new(&files, position, limits).unwrap(); - let direct = restore.path().join("direct.sqlite"); - restore_exact(&plan, &direct).unwrap(); - let compacted = compact_exact(&plan, &remote.path().join("snapshot.ltx")).unwrap(); - let compact_plan = VerifiedPlan::new(&[compacted], position, limits).unwrap(); - let compact_db = restore.path().join("compact.sqlite"); - restore_exact(&compact_plan, &compact_db).unwrap(); - assert_eq!( - std::fs::read(&direct).unwrap(), - std::fs::read(&compact_db).unwrap() - ); - assert_eq!(rows(&direct), ["initial", "second", "Unicode: 🦀 中文"]); -} - -#[test] -fn snapshot_matches_delta_restore_byte_for_byte() { - let temp = TempDir::new().unwrap(); - let mut db = Db::open(&temp.path().join("repo.sqlite"), Limits::default()).unwrap(); - insert(&mut db, "one"); - let mut batch = db.capture().unwrap(); - insert(&mut db, "two"); - let next = db.capture().unwrap(); - batch.segments.extend(next.segments); - let (snapshot, _) = db.snapshot(&temp.path().join("snapshot.ltx")).unwrap(); - assert_eq!(snapshot.info().position(), next.position); - let plan = VerifiedPlan::new(&batch.segments, next.position, Limits::default()).unwrap(); - let snapshot_plan = VerifiedPlan::new(&[snapshot], next.position, Limits::default()).unwrap(); - let a = temp.path().join("a.sqlite"); - let b = temp.path().join("b.sqlite"); - restore_exact(&plan, &a).unwrap(); - restore_exact(&snapshot_plan, &b).unwrap(); - assert_eq!(std::fs::read(a).unwrap(), std::fs::read(b).unwrap()); -} - -#[test] -fn rolled_back_sql_is_not_restored() { - let temp = TempDir::new().unwrap(); - let mut db = Db::open(&temp.path().join("repo.sqlite"), Limits::default()).unwrap(); - insert(&mut db, "keep"); - let result = db.transaction(|tx| { - tx.execute("INSERT INTO messages(body) VALUES ('rollback')", [])?; - Err::<(), _>(crab_ltx::rusqlite::Error::InvalidQuery) - }); - assert!(result.is_err()); - let batch = db.capture().unwrap(); - let plan = VerifiedPlan::new(&batch.segments, batch.position, Limits::default()).unwrap(); - let path = temp.path().join("restored.sqlite"); - restore_exact(&plan, &path).unwrap(); - assert_eq!(rows(&path), ["keep"]); -} - -#[test] -fn checkpoint_threshold_captures_every_cut_and_growth_page() { - let temp = TempDir::new().unwrap(); - let mut db = Db::open(&temp.path().join("repo.sqlite"), Limits::default()).unwrap(); - let mut segments = Vec::new(); - let mut target = Default::default(); - for round in 0..4 { - db.transaction(|tx| { - tx.execute( - "CREATE TABLE IF NOT EXISTS blobs (id INTEGER PRIMARY KEY, payload BLOB)", - [], - )?; - for _ in 0..400 { - tx.execute("INSERT INTO blobs(payload) VALUES (zeroblob(12000))", [])?; - } - if round % 2 == 1 { - tx.execute("DELETE FROM blobs WHERE id % 2 = 0", [])?; - } - Ok(()) - }) - .unwrap(); - let batch = db.capture().unwrap(); - segments.extend(batch.segments); - target = batch.position; - } - let plan = VerifiedPlan::new(&segments, target, Limits::default()).unwrap(); - let path = temp.path().join("restored.sqlite"); - restore_exact(&plan, &path).unwrap(); - let conn = Connection::open(path).unwrap(); - let check: String = conn - .query_row("PRAGMA integrity_check", [], |row| row.get(0)) - .unwrap(); - assert_eq!(check, "ok"); - let count: i64 = conn - .query_row("SELECT count(*) FROM blobs", [], |row| row.get(0)) - .unwrap(); - assert_eq!(count, 800); -} - -#[test] -fn exact_plan_rejects_gap_overlap_wrong_target_and_manifest_mutation() { - let temp = TempDir::new().unwrap(); - let limits = Limits::default(); - let mut db = Db::open(&temp.path().join("repo.sqlite"), limits).unwrap(); - let mut segments = Vec::new(); - let mut target = Default::default(); - for value in ["a", "b", "c"] { - insert(&mut db, value); - let batch = db.capture().unwrap(); - segments.extend(batch.segments); - target = batch.position; - } - let mut gap = segments.clone(); - gap.remove(1); - assert!(VerifiedPlan::new(&gap, target, limits).is_err()); - let mut overlap = segments.clone(); - overlap.insert(1, segments[0].clone()); - assert!(VerifiedPlan::new(&overlap, target, limits).is_err()); - let mut wrong_target = target; - wrong_target.txid += 1; - assert!(VerifiedPlan::new(&segments, wrong_target, limits).is_err()); - let mut info = segments[0].info().clone(); - info.post_checksum ^= 1; - let mut wrong_info = segments.clone(); - wrong_info[0] = LocalSegment::new(segments[0].path().to_owned(), info); - assert!(VerifiedPlan::new(&wrong_info, target, limits).is_err()); - let mut corrupted = std::fs::read(segments[0].path()).unwrap(); - corrupted[120] ^= 1; - std::fs::write(segments[0].path(), corrupted).unwrap(); - assert!(VerifiedPlan::new(&segments, target, limits).is_err()); -} - -#[test] -fn verified_plan_owns_image_and_never_overwrites_destination() { - let temp = TempDir::new().unwrap(); - let mut db = Db::open(&temp.path().join("repo.sqlite"), Limits::default()).unwrap(); - insert(&mut db, "retained"); - let batch = db.capture().unwrap(); - let plan = VerifiedPlan::new(&batch.segments, batch.position, Limits::default()).unwrap(); - for file in batch.segments { - std::fs::remove_file(file.path()).unwrap(); - } - let compacted = compact_exact(&plan, &temp.path().join("compacted.ltx")).unwrap(); - let compact_plan = VerifiedPlan::new(&[compacted], plan.position(), Limits::default()).unwrap(); - let destination = temp.path().join("restore.sqlite"); - restore_exact(&plan, &destination).unwrap(); - assert!(restore_exact(&plan, &destination).is_err()); - assert_eq!(rows(&destination), ["retained"]); - let compact_destination = temp.path().join("compact-restore.sqlite"); - restore_exact(&compact_plan, &compact_destination).unwrap(); - assert_eq!(rows(&compact_destination), ["retained"]); -} - -#[test] -fn capture_failure_fences_writer_and_stale_sessions_are_refused() { - let temp = TempDir::new().unwrap(); - let path = temp.path().join("repo.sqlite"); - let mut db = Db::open( - &path, - Limits { - max_capture_bytes: 32768, - // The escalated full image cannot fit either bound, so this commit - // is uncapturable and the session must fence. - max_file_bytes: 32768, - ..Limits::default() - }, - ) - .unwrap(); - assert!(Db::open(&path, Limits::default()).is_err()); - db.transaction(|tx| { - tx.execute("CREATE TABLE large (body BLOB)", [])?; - tx.execute("INSERT INTO large VALUES (randomblob(65536))", [])?; - Ok(()) - }) - .unwrap(); - assert!(db.capture().is_err()); - assert!(matches!(db.transaction(|_| Ok(())), Err(CrabError::Fenced))); - db.close().unwrap(); - assert!(Db::open(&path, Limits::default()).is_err()); -} - -#[test] -fn restore_admission_rejects_database_and_chain_byte_limits() { - let temp = TempDir::new().unwrap(); - let mut db = Db::open(&temp.path().join("repo.sqlite"), Limits::default()).unwrap(); - insert(&mut db, "bounded"); - let batch = db.capture().unwrap(); - let small = Limits { - max_database_bytes: 512, - ..Limits::default() - }; - assert!(VerifiedPlan::new(&batch.segments, batch.position, small).is_err()); - let small = Limits { - max_capture_bytes: 128, - max_file_bytes: 128, - max_plan_bytes: 128, - ..Limits::default() - }; - assert!(VerifiedPlan::new(&batch.segments, batch.position, small).is_err()); -} - -#[test] -fn corrupt_committed_wal_cannot_acknowledge_an_earlier_valid_cut() { - use std::io::{Read, Seek, SeekFrom, Write}; - let temp = TempDir::new().unwrap(); - let path = temp.path().join("repo.sqlite"); - let mut db = Db::open(&path, Limits::default()).unwrap(); - insert(&mut db, "already captured"); - db.capture().unwrap(); - insert(&mut db, "valid but not captured yet"); - insert(&mut db, "damaged commit"); - let wal_path = temp.path().join("repo.sqlite-wal"); - let mut wal = std::fs::OpenOptions::new() - .read(true) - .write(true) - .open(wal_path) - .unwrap(); - let len = wal.metadata().unwrap().len(); - wal.seek(SeekFrom::Start(len - 1)).unwrap(); - let mut byte = [0]; - wal.read_exact(&mut byte).unwrap(); - byte[0] ^= 1; - wal.seek(SeekFrom::Start(len - 1)).unwrap(); - wal.write_all(&byte).unwrap(); - wal.sync_all().unwrap(); - assert!(db.capture().is_err()); - assert!(matches!(db.transaction(|_| Ok(())), Err(CrabError::Fenced))); -} - -#[test] -fn recovered_database_starts_a_new_epoch_and_can_capture_again() { - let temp = TempDir::new().unwrap(); - let mut db = Db::open(&temp.path().join("old.sqlite"), Limits::default()).unwrap(); - insert(&mut db, "before takeover"); - let batch = db.capture().unwrap(); - let plan = VerifiedPlan::new(&batch.segments, batch.position, Limits::default()).unwrap(); - db.close().unwrap(); - let path = temp.path().join("new.sqlite"); - restore_exact(&plan, &path).unwrap(); - let mut new = Db::open(&path, Limits::default()).unwrap(); - insert(&mut new, "after takeover"); - let batch = new.capture().unwrap(); - assert_eq!(batch.segments[0].info().min_txid, 1); - let plan = VerifiedPlan::new(&batch.segments, batch.position, Limits::default()).unwrap(); - let path = temp.path().join("restored.sqlite"); - restore_exact(&plan, &path).unwrap(); - assert_eq!(rows(&path), ["before takeover", "after takeover"]); -} - -#[test] -fn sqlite_sidecars_prevent_restore() { - let temp = TempDir::new().unwrap(); - let mut db = Db::open(&temp.path().join("repo.sqlite"), Limits::default()).unwrap(); - insert(&mut db, "safe"); - let batch = db.capture().unwrap(); - let plan = VerifiedPlan::new(&batch.segments, batch.position, Limits::default()).unwrap(); - let path = temp.path().join("restored.sqlite"); - std::fs::write(temp.path().join("restored.sqlite-wal"), b"stale").unwrap(); - assert!(restore_exact(&plan, &path).is_err()); - assert!(!path.exists()); -} - -#[cfg(unix)] -#[test] -fn database_symlink_cannot_claim_a_second_capture_session() { - let temp = TempDir::new().unwrap(); - let path = temp.path().join("repo.sqlite"); - let db = Db::open(&path, Limits::default()).unwrap(); - let alias = temp.path().join("alias.sqlite"); - std::os::unix::fs::symlink(db.path(), &alias).unwrap(); - assert!(Db::open(&alias, Limits::default()).is_err()); -} diff --git a/crates/crab-ltx/tests/cell/restore.rs b/crates/crab-ltx/tests/cell/restore.rs deleted file mode 100644 index 0845d2a40..000000000 --- a/crates/crab-ltx/tests/cell/restore.rs +++ /dev/null @@ -1,694 +0,0 @@ -#![cfg(feature = "replica")] - -use std::{ - fmt, - sync::{ - Arc, Mutex, - atomic::{AtomicU8, AtomicUsize, Ordering}, - }, - time::Duration, -}; - -use async_trait::async_trait; -use bytes::Bytes; -use crab_ltx::CellStorageLayout; -use crab_ltx::{ - CaptureBatch, CaptureTiming, CellReplica, Db, Host, Limits, LtxPhase, LtxReadOrigin, - LtxRequestOutcome, LtxTelemetry, RootRef, VerifiedPlan, restore_exact, -}; -use crab_storage::Store; -use futures_util::{StreamExt as _, stream::BoxStream}; -use object_store::{ - GetOptions, GetRange, GetResult, GetResultPayload, ListResult, MultipartUpload, ObjectMeta, - ObjectStore, PutMultipartOptions, PutOptions, PutPayload, PutResult, memory::InMemory, - path::Path, -}; - -mod read_ahead; -mod upload; - -const NO_FAULT: u8 = 0; -const SHORT_RANGE: u8 = 1; -const CORRUPT_RANGE: u8 = 2; -const TIMEOUT_RANGE: u8 = 3; -const PUT_FAILURE: u8 = 4; -const PUT_RESPONSE_LOST: u8 = 5; - -#[derive(Default)] -struct RecordingTelemetry { - phases: Mutex>, - logical_reads: Mutex>, - reads: Mutex>, -} - -impl LtxTelemetry for RecordingTelemetry { - fn phase(&self, phase: LtxPhase, _: Duration, _: bool) { - self.phases.lock().unwrap().push(phase); - } - - fn logical_read(&self, origin: LtxReadOrigin) { - self.logical_reads.lock().unwrap().push(origin); - } - - fn origin_request(&self, origin: LtxReadOrigin, outcome: LtxRequestOutcome, bytes: u64) { - self.reads.lock().unwrap().push((origin, outcome, bytes)); - } -} - -struct ReadStats { - ranges: Mutex)>>, - active: AtomicUsize, - peak: AtomicUsize, - active_ranges: AtomicUsize, - peak_ranges: AtomicUsize, - active_range_bytes: AtomicUsize, - peak_range_bytes: AtomicUsize, - maximum_range_bytes: AtomicUsize, - body_requests: AtomicUsize, -} - -impl ReadStats { - fn new() -> Self { - Self { - ranges: Mutex::new(Vec::new()), - active: AtomicUsize::new(0), - peak: AtomicUsize::new(0), - active_ranges: AtomicUsize::new(0), - peak_ranges: AtomicUsize::new(0), - active_range_bytes: AtomicUsize::new(0), - peak_range_bytes: AtomicUsize::new(0), - maximum_range_bytes: AtomicUsize::new(0), - body_requests: AtomicUsize::new(0), - } - } - - fn begin(self: &Arc, range_bytes: Option) -> ReadGuard { - self.body_requests.fetch_add(1, Ordering::SeqCst); - let active = self.active.fetch_add(1, Ordering::SeqCst) + 1; - self.peak.fetch_max(active, Ordering::SeqCst); - if let Some(bytes) = range_bytes { - let active = self.active_ranges.fetch_add(1, Ordering::SeqCst) + 1; - self.peak_ranges.fetch_max(active, Ordering::SeqCst); - let active_bytes = self.active_range_bytes.fetch_add(bytes, Ordering::SeqCst) + bytes; - self.peak_range_bytes - .fetch_max(active_bytes, Ordering::SeqCst); - } - ReadGuard { - stats: Arc::clone(self), - range_bytes, - } - } - - fn reset(&self) { - self.ranges.lock().unwrap().clear(); - assert_eq!(self.active.load(Ordering::SeqCst), 0); - assert_eq!(self.active_ranges.load(Ordering::SeqCst), 0); - assert_eq!(self.active_range_bytes.load(Ordering::SeqCst), 0); - self.peak.store(0, Ordering::SeqCst); - self.peak_ranges.store(0, Ordering::SeqCst); - self.peak_range_bytes.store(0, Ordering::SeqCst); - self.maximum_range_bytes.store(0, Ordering::SeqCst); - self.body_requests.store(0, Ordering::SeqCst); - } -} - -struct ReadGuard { - stats: Arc, - range_bytes: Option, -} - -impl Drop for ReadGuard { - fn drop(&mut self) { - self.stats.active.fetch_sub(1, Ordering::SeqCst); - if let Some(bytes) = self.range_bytes { - self.stats.active_ranges.fetch_sub(1, Ordering::SeqCst); - self.stats - .active_range_bytes - .fetch_sub(bytes, Ordering::SeqCst); - } - } -} - -struct InstrumentedStore { - inner: Arc, - delay: Duration, - fault: AtomicU8, - stats: Arc, - puts: AtomicUsize, - multipart: AtomicUsize, -} - -impl InstrumentedStore { - fn new(inner: Arc, delay: Duration) -> Arc { - Arc::new(Self { - inner, - delay, - fault: AtomicU8::new(NO_FAULT), - stats: Arc::new(ReadStats::new()), - puts: AtomicUsize::new(0), - multipart: AtomicUsize::new(0), - }) - } - - fn arm(&self, fault: u8) { - self.fault.store(fault, Ordering::SeqCst); - } - - fn reset(&self) { - self.stats.reset(); - } - - /// Fails an upload the way a transient provider failure arrives. - fn check_upload_fault(&self) -> object_store::Result<()> { - if self.fault.load(Ordering::SeqCst) != PUT_FAILURE { - return Ok(()); - } - Err(object_store::Error::Generic { - store: "InstrumentedStore", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::ConnectionReset, - "injected upload failure", - )), - }) - } -} - -impl fmt::Debug for InstrumentedStore { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter.write_str("InstrumentedStore") - } -} - -impl fmt::Display for InstrumentedStore { - fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - formatter.write_str("InstrumentedStore") - } -} - -#[async_trait] -impl ObjectStore for InstrumentedStore { - async fn put_opts( - &self, - location: &Path, - payload: PutPayload, - options: PutOptions, - ) -> object_store::Result { - self.puts.fetch_add(1, Ordering::SeqCst); - self.check_upload_fault()?; - let result = self.inner.put_opts(location, payload, options).await?; - if location.extension() == Some("ltx") - && self - .fault - .compare_exchange( - PUT_RESPONSE_LOST, - NO_FAULT, - Ordering::SeqCst, - Ordering::SeqCst, - ) - .is_ok() - { - return Err(object_store::Error::Generic { - store: "InstrumentedStore", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::ConnectionReset, - "injected response loss after commit", - )), - }); - } - Ok(result) - } - - async fn put_multipart_opts( - &self, - location: &Path, - options: PutMultipartOptions, - ) -> object_store::Result> { - self.multipart.fetch_add(1, Ordering::SeqCst); - self.check_upload_fault()?; - self.inner.put_multipart_opts(location, options).await - } - - async fn get_opts( - &self, - location: &Path, - options: GetOptions, - ) -> object_store::Result { - let is_range = options.range.is_some(); - let range_bytes = if let Some(GetRange::Bounded(range)) = &options.range { - self.stats - .ranges - .lock() - .unwrap() - .push((location.clone(), range.clone())); - let bytes = usize::try_from(range.end - range.start).unwrap_or(usize::MAX); - self.stats - .maximum_range_bytes - .fetch_max(bytes, Ordering::SeqCst); - Some(bytes) - } else { - None - }; - let _guard = (!options.head).then(|| self.stats.begin(range_bytes)); - if !options.head { - tokio::time::sleep(self.delay).await; - } - if is_range && self.fault.load(Ordering::SeqCst) == TIMEOUT_RANGE { - return Err(object_store::Error::Generic { - store: "InstrumentedStore", - source: Box::new(std::io::Error::new( - std::io::ErrorKind::TimedOut, - "injected range timeout", - )), - }); - } - let result = self.inner.get_opts(location, options).await?; - let fault = self.fault.load(Ordering::SeqCst); - if !is_range || fault == NO_FAULT { - return Ok(result); - } - - let meta = result.meta.clone(); - let range = result.range.clone(); - let attributes = result.attributes.clone(); - let extensions = result.extensions.clone(); - let bytes = result.bytes().await?; - let bytes = match fault { - SHORT_RANGE => bytes.slice(..bytes.len().saturating_sub(1)), - CORRUPT_RANGE => { - let mut corrupted = bytes.to_vec(); - if let Some(first) = corrupted.first_mut() { - *first ^= 1; - } - Bytes::from(corrupted) - } - _ => bytes, - }; - Ok(GetResult { - payload: GetResultPayload::Stream( - futures_util::stream::once(async move { Ok(bytes) }).boxed(), - ), - meta, - range, - attributes, - extensions, - }) - } - - fn delete_stream( - &self, - locations: BoxStream<'static, object_store::Result>, - ) -> BoxStream<'static, object_store::Result> { - self.inner.delete_stream(locations) - } - - fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { - self.inner.list(prefix) - } - - async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { - self.inner.list_with_delimiter(prefix).await - } - - async fn copy_opts( - &self, - from: &Path, - to: &Path, - options: object_store::CopyOptions, - ) -> object_store::Result<()> { - self.inner.copy_opts(from, to, options).await - } -} - -struct Fixture { - backend: Arc, - cell: [u8; 32], - incarnation: [u8; 16], - root: RootRef, - expected: Vec, -} - -async fn fixture(extra_segments: usize) -> Fixture { - let directory = tempfile::TempDir::new().unwrap(); - let source = directory.path().join("source.sqlite"); - let mut writer = Db::open(&source, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL);\ - INSERT INTO counter VALUES(0);\ - CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(3000000))", - ) - }) - .unwrap(); - let first = writer.capture().unwrap(); - let mut segments = first.segments; - let mut position = first.position; - for _ in 0..extra_segments { - writer - .transaction(|transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(()) - }) - .unwrap(); - let next = writer.capture().unwrap(); - segments.extend(next.segments); - position = next.position; - } - writer.close().unwrap(); - let batch = CaptureBatch { - segments, - position, - timing: CaptureTiming::default(), - }; - - let expected_path = directory.path().join("expected.sqlite"); - let plan = VerifiedPlan::new(&batch.segments, batch.position, Limits::default()).unwrap(); - restore_exact(&plan, &expected_path).unwrap(); - let expected = std::fs::read(expected_path).unwrap(); - - let backend = Arc::new(InMemory::new()); - let cell = [141; 32]; - let incarnation = [142; 16]; - let replica = cell_replica( - Store::new(backend.clone()), - cell, - incarnation, - Host::default(), - ); - let root = replica.prepare(None, &batch, 1, 1).await.unwrap().root(); - Fixture { - backend, - cell, - incarnation, - root, - expected, - } -} - -fn cell_replica(store: Store, cell: [u8; 32], incarnation: [u8; 16], host: Host) -> CellReplica { - CellReplica::new( - CellStorageLayout::new(store, Path::from("parallel"), [143; 16]), - cell, - incarnation, - Limits::default(), - ) - .unwrap() - .with_host(host) -} - -fn delayed_replica( - fixture: &Fixture, - delay: Duration, - io_slots: usize, -) -> (Arc, CellReplica) { - let store = InstrumentedStore::new(fixture.backend.clone(), delay); - let replica = cell_replica( - Store::new(store.clone()), - fixture.cell, - fixture.incarnation, - Host::default().with_io_slots(Arc::new(tokio::sync::Semaphore::new(io_slots))), - ); - (store, replica) -} - -fn p95(samples: &mut [Duration]) -> Duration { - samples.sort_unstable(); - samples[(samples.len() * 95).div_ceil(100) - 1] -} - -#[tokio::test] -async fn failed_upload_returns_no_proposal_and_the_retry_is_identical() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("source.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL);\ - INSERT INTO counter VALUES(0)", - ) - }) - .unwrap(); - let first = writer.capture().unwrap(); - writer - .transaction(|transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(()) - }) - .unwrap(); - let next = writer.capture().unwrap(); - let mut segments = first.segments; - segments.extend(next.segments); - let batch = CaptureBatch { - segments, - position: next.position, - timing: CaptureTiming::default(), - }; - - // A transient upload failure must surface as a retryable failure and must - // never hand back a root proposal. - let backend = Arc::new(InMemory::new()); - let store = InstrumentedStore::new(backend.clone(), Duration::ZERO); - store.arm(PUT_FAILURE); - let replica = cell_replica( - Store::new(store.clone()), - [151; 32], - [152; 16], - Host::default(), - ); - let error = match replica.prepare(None, &batch, 1, 1).await { - Ok(_) => panic!("a failed upload must not return a root proposal"), - Err(error) => error, - }; - assert_eq!( - error.classify(), - crab_ltx::FailureClass::Retryable { after: None } - ); - - // Clearing the fault and retrying the same inputs yields the canonical - // proposal: an independent replica over a fresh backend agrees byte for - // byte, so the failed attempt left no partial root behind. - store.arm(NO_FAULT); - let retried = replica.prepare(None, &batch, 1, 1).await.unwrap(); - let fresh_backend = Arc::new(InMemory::new()); - let fresh = cell_replica( - Store::new(fresh_backend), - [151; 32], - [152; 16], - Host::default(), - ); - let canonical = fresh.prepare(None, &batch, 1, 1).await.unwrap(); - assert_eq!(retried.root(), canonical.root()); - assert_eq!(retried.root().position, batch.position); -} - -#[tokio::test(start_paused = true)] -async fn cold_open_and_restore_improve_p95_under_object_latency() { - let fixture = fixture(192).await; - for delay_ms in [5, 20, 100] { - let delay = Duration::from_millis(delay_ms); - let (store, replica) = delayed_replica(&fixture, delay, 8); - replica.open_root(&fixture.root).await.unwrap(); - - let mut open_samples = Vec::new(); - for _ in 0..5 { - // Each independent Store identity exercises the cold origin path; - // repeated opens through one Store now reuse authenticated metadata. - let (cold_store, cold_replica) = delayed_replica(&fixture, delay, 8); - let started = tokio::time::Instant::now(); - cold_replica.open_root(&fixture.root).await.unwrap(); - open_samples.push(started.elapsed()); - assert!(cold_store.stats.peak.load(Ordering::SeqCst) >= 3); - } - assert!(p95(&mut open_samples) < delay * 4); - - store.reset(); - replica.open_root(&fixture.root).await.unwrap(); - assert_eq!(store.stats.body_requests.load(Ordering::SeqCst), 0); - - let opened = replica.open_root(&fixture.root).await.unwrap(); - opened.paged().read_page(1).await.unwrap(); - let mut warm_read_samples = Vec::new(); - for _ in 0..5 { - store.reset(); - let started = tokio::time::Instant::now(); - opened.paged().read_page(1).await.unwrap(); - warm_read_samples.push(started.elapsed()); - assert_eq!(store.stats.body_requests.load(Ordering::SeqCst), 1); - } - assert!(p95(&mut warm_read_samples) <= delay); - - let warm = tempfile::TempDir::new().unwrap(); - opened - .restore(&warm.path().join("warm.sqlite")) - .await - .unwrap(); - let mut restore_samples = Vec::new(); - let mut minimum_requests = usize::MAX; - for sample in 0..5 { - store.reset(); - let output = tempfile::TempDir::new().unwrap(); - let destination = output.path().join(format!("restore-{sample}.sqlite")); - let started = tokio::time::Instant::now(); - opened.restore(&destination).await.unwrap(); - restore_samples.push(started.elapsed()); - minimum_requests = - minimum_requests.min(store.stats.body_requests.load(Ordering::SeqCst)); - assert_eq!(std::fs::read(destination).unwrap(), fixture.expected); - assert!(store.stats.peak_ranges.load(Ordering::SeqCst) > 1); - assert!(store.stats.maximum_range_bytes.load(Ordering::SeqCst) <= 1 << 20); - } - assert!(p95(&mut restore_samples) < delay * minimum_requests as u32); - } -} - -#[tokio::test(start_paused = true)] -async fn simultaneous_restores_share_host_io_admission() { - let fixture = fixture(0).await; - let (store, replica) = delayed_replica(&fixture, Duration::from_millis(20), 3); - let opened = replica.open_root(&fixture.root).await.unwrap(); - let output = tempfile::TempDir::new().unwrap(); - let first = output.path().join("first.sqlite"); - let second = output.path().join("second.sqlite"); - store.reset(); - - let (left, right) = tokio::join!(opened.restore(&first), opened.restore(&second)); - - left.unwrap(); - right.unwrap(); - assert!(store.stats.peak.load(Ordering::SeqCst) > 1); - assert!(store.stats.peak.load(Ordering::SeqCst) <= 3); - assert!(store.stats.peak_range_bytes.load(Ordering::SeqCst) <= 3 << 20); - assert_eq!(std::fs::read(first).unwrap(), fixture.expected); - assert_eq!(std::fs::read(second).unwrap(), fixture.expected); -} - -#[tokio::test] -async fn replica_telemetry_attributes_prepare_cold_sparse_and_restore_work() { - let fixture = fixture(4).await; - let store = InstrumentedStore::new(fixture.backend.clone(), Duration::ZERO); - let telemetry = Arc::new(RecordingTelemetry::default()); - let host = Host::default().with_ltx_telemetry(telemetry.clone()); - let replica = cell_replica( - Store::new(store), - fixture.cell, - fixture.incarnation, - host.clone(), - ); - - let opened = replica.open_root(&fixture.root).await.unwrap(); - opened.paged().read_page(1).await.unwrap(); - let output = tempfile::TempDir::new().unwrap(); - opened - .restore(&output.path().join("telemetry.sqlite")) - .await - .unwrap(); - let mut writer = Db::open(&output.path().join("new.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - let cuts = writer.capture().unwrap(); - let append = cell_replica( - Store::new(Arc::new(InMemory::new())), - [151; 32], - [152; 16], - host, - ); - append.prepare(None, &cuts, 1, 1).await.unwrap(); - writer.close().unwrap(); - - let phases = telemetry.phases.lock().unwrap(); - for expected in [ - LtxPhase::RootPreparation, - LtxPhase::RootOpen, - LtxPhase::Directory, - LtxPhase::FrameFetch, - LtxPhase::RestoreWrite, - ] { - assert!(phases.contains(&expected), "missing {expected:?}"); - } - let reads = telemetry.reads.lock().unwrap(); - for expected in [LtxReadOrigin::Cold, LtxReadOrigin::Sparse] { - assert!( - reads - .iter() - .any(|(origin, outcome, bytes)| *origin == expected - && *outcome == LtxRequestOutcome::Succeeded - && *bytes > 0), - "missing {expected:?}" - ); - } - let logical_reads = telemetry.logical_reads.lock().unwrap(); - assert!(logical_reads.contains(&LtxReadOrigin::Cold)); - assert!(logical_reads.contains(&LtxReadOrigin::Sparse)); -} - -#[tokio::test] -async fn failed_provider_attempt_is_not_counted_as_a_logical_retry() { - let fixture = fixture(0).await; - let store = InstrumentedStore::new(fixture.backend.clone(), Duration::ZERO); - let telemetry = Arc::new(RecordingTelemetry::default()); - let host = Host::default().with_ltx_telemetry(telemetry.clone()); - let replica = cell_replica( - Store::new(store.clone()), - fixture.cell, - fixture.incarnation, - host, - ); - let opened = replica.open_root(&fixture.root).await.unwrap(); - telemetry.logical_reads.lock().unwrap().clear(); - telemetry.reads.lock().unwrap().clear(); - store.arm(TIMEOUT_RANGE); - - assert!(opened.paged().read_page(1).await.is_err()); - - assert_eq!( - *telemetry.logical_reads.lock().unwrap(), - vec![LtxReadOrigin::Sparse] - ); - assert!( - telemetry - .reads - .lock() - .unwrap() - .iter() - .any(|(origin, outcome, bytes)| *origin == LtxReadOrigin::Sparse - && *outcome == LtxRequestOutcome::Failed - && *bytes == 0) - ); -} - -#[tokio::test(start_paused = true)] -async fn failed_or_cancelled_parallel_restore_never_publishes_destination() { - let fixture = fixture(0).await; - let (store, replica) = delayed_replica(&fixture, Duration::from_millis(100), 8); - let opened = replica.open_root(&fixture.root).await.unwrap(); - let output = tempfile::TempDir::new().unwrap(); - - for (name, fault) in [ - ("short.sqlite", SHORT_RANGE), - ("corrupt.sqlite", CORRUPT_RANGE), - ("timeout.sqlite", TIMEOUT_RANGE), - ] { - let destination = output.path().join(name); - store.arm(fault); - assert!(opened.restore(&destination).await.is_err()); - assert!(!destination.exists()); - } - - store.arm(NO_FAULT); - let destination = output.path().join("cancelled.sqlite"); - assert!( - tokio::time::timeout(Duration::from_millis(10), opened.restore(&destination)) - .await - .is_err() - ); - assert!(!destination.exists()); - assert!(!std::fs::read_dir(output.path()).unwrap().any(|entry| { - entry - .unwrap() - .file_name() - .to_string_lossy() - .contains(".crab-restore-") - })); -} diff --git a/crates/crab-ltx/tests/cell/restore/read_ahead.rs b/crates/crab-ltx/tests/cell/restore/read_ahead.rs deleted file mode 100644 index 84dc5c5c7..000000000 --- a/crates/crab-ltx/tests/cell/restore/read_ahead.rs +++ /dev/null @@ -1,143 +0,0 @@ -//! Demand-read work on fragmented immutable roots. - -use super::*; - -struct IsolatedExecutor; - -impl crab_ltx::environment::Executor for IsolatedExecutor { - fn dispatch(&self, job: Box) -> std::io::Result<()> { - tokio::task::spawn_blocking(job); - Ok(()) - } - - fn start_worker( - &self, - job: Box, - ) -> std::io::Result> { - std::thread::Builder::new() - .spawn(job) - .map(|thread| Box::new(thread) as Box) - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn fragmented_snapshot_demand_reads_do_not_refetch_cached_frames() { - verify_read_ahead(Arc::new(InMemory::new()), "read-ahead").await; -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires the RustFS environment documented in examples/README.md"] -async fn rustfs_fragmented_snapshot_demand_reads_do_not_refetch_cached_frames() { - let endpoint = std::env::var("CRAB_LTX_TEST_ENDPOINT").unwrap(); - let store = crab_storage::build_explicit_store( - &std::env::var("CRAB_LTX_TEST_BUCKET").unwrap(), - crab_storage::ObjectStoreCredentials::Aws { - access_key_id: std::env::var("AWS_ACCESS_KEY_ID").unwrap(), - secret_access_key: std::env::var("AWS_SECRET_ACCESS_KEY").unwrap(), - session_token: None, - region: "us-east-1".into(), - }, - Some(&endpoint), - endpoint.starts_with("http://"), - ) - .unwrap(); - let run = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_nanos(); - let prefix = format!("crab-ltx-tests/read-ahead/{run}"); - verify_read_ahead(store.inner().clone(), &prefix).await; - eprintln!("RustFS demand read-ahead passed: {prefix}"); -} - -async fn verify_read_ahead(backend: Arc, prefix: &str) { - for page_size in [512, 4096] { - let directory = tempfile::TempDir::new().unwrap(); - let source = directory.path().join("source.sqlite"); - let connection = crab_ltx::rusqlite::Connection::open(&source).unwrap(); - connection - .pragma_update(None, "page_size", page_size) - .unwrap(); - connection.execute_batch("VACUUM").unwrap(); - drop(connection); - let mut writer = Db::open(&source, Limits::default()).unwrap(); - let expected = writer - .transaction(|tx| { - tx.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL); \ - CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES(0)", - )?; - let mut expected = blake3::Hasher::new(); - for row in 0u32..128 { - let mut bytes = vec![0; 4096]; - blake3::Hasher::new() - .update(&row.to_be_bytes()) - .finalize_xof() - .fill(&mut bytes); - tx.execute("INSERT INTO payload VALUES(?1)", [bytes.as_slice()])?; - expected.update(&bytes); - } - Ok(expected.finalize()) - }) - .unwrap(); - let store = InstrumentedStore::new(backend.clone(), Duration::ZERO); - // A private driver keeps unrelated concurrent tests from evicting this - // sub-MiB working set while its exact origin work is measured. - let host = Host::default().with_executor(Arc::new(IsolatedExecutor)); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(store.clone()), - Path::from(format!("{prefix}/{page_size}")), - [161; 16], - ), - [162; 32], - [163; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap(); - writer - .transaction(|tx| tx.execute_batch("UPDATE counter SET value = value + 1")) - .unwrap(); - let next = replica - .prepare(Some(&root.root()), &writer.capture().unwrap(), 2, 1) - .await - .unwrap(); - writer.close().unwrap(); - let verified = replica.open_root(&next.root()).await.unwrap(); - assert_eq!(verified.paged().page_size(), page_size); - tokio::task::spawn_blocking(move || { - store.reset(); - let view = verified.open_read_only(&directory.path().join("view.sqlite")).unwrap(); - assert_eq!(payload_hash(&view.connection().unwrap()), expected); - let ranges = store.stats.ranges.lock().unwrap().clone(); - assert!(!ranges.is_empty()); - for (index, (path, range)) in ranges.iter().enumerate() { - assert!(ranges[..index].iter().all(|(previous_path, previous)| { - previous_path != path || previous.end <= range.start || range.end <= previous.start - }), "read-ahead refetched a cached immutable frame at {page_size}-byte pages: {ranges:?}"); - } - let bytes: u64 = ranges.iter().map(|(_, range)| range.end - range.start).sum(); - eprintln!("Demand scan page_size={page_size} ranges={} bytes={bytes}", ranges.len()); - store.reset(); - assert_eq!(payload_hash(&view.connection().unwrap()), expected); - assert!(store.stats.ranges.lock().unwrap().is_empty()); - }).await.unwrap(); - } -} - -fn payload_hash(connection: &crab_ltx::rusqlite::Connection) -> blake3::Hash { - let mut statement = connection - .prepare("SELECT value FROM payload ORDER BY rowid") - .unwrap(); - let mut rows = statement.query([]).unwrap(); - let mut hash = blake3::Hasher::new(); - while let Some(row) = rows.next().unwrap() { - hash.update(&row.get::<_, Vec>(0).unwrap()); - } - hash.finalize() -} diff --git a/crates/crab-ltx/tests/cell/restore/upload.rs b/crates/crab-ltx/tests/cell/restore/upload.rs deleted file mode 100644 index 72607fcc8..000000000 --- a/crates/crab-ltx/tests/cell/restore/upload.rs +++ /dev/null @@ -1,118 +0,0 @@ -//! Small-body transport selection, exact restore, and ambiguous upload retries. - -use super::*; - -#[tokio::test] -async fn native_compacted_and_bundled_uploads_use_single_put_only_for_small_bodies() { - for (payload, expected_multipart) in [(4096, 0), (300_000, 1)] { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); - writer - .transaction(|tx| { - tx.execute_batch("CREATE TABLE t(value BLOB)")?; - tx.execute("INSERT INTO t VALUES(randomblob(?1))", [payload])?; - Ok(()) - }) - .unwrap(); - let batch = writer.capture_deferred().unwrap(); - let store = InstrumentedStore::new(Arc::new(InMemory::new()), Duration::ZERO); - let replica = cell_replica( - Store::new(store.clone()), - [151; 32], - [152; 16], - Host::default(), - ); - let prepared = replica.prepare(None, &batch, 1, 1).await.unwrap(); - assert_eq!(store.multipart.load(Ordering::SeqCst), expected_multipart); - let cost = replica.take_publication_cost(); - assert_eq!( - store.puts.load(Ordering::SeqCst) + expected_multipart, - cost.objects as usize - ); - - let compacted = replica - .prepare_compaction( - &prepared.root(), - 0..prepared.verified().segment_count(), - 9, - directory.path(), - ) - .await - .unwrap(); - assert_eq!( - store.multipart.load(Ordering::SeqCst), - expected_multipart * 2 - ); - let before = directory.path().join("before"); - let after = directory.path().join("after"); - prepared.verified().restore(&before).await.unwrap(); - compacted.verified().restore(&after).await.unwrap(); - assert_eq!( - std::fs::read(before).unwrap(), - std::fs::read(&after).unwrap() - ); - let bundle = crab_ltx::bundle::Bundle::encode( - batch - .segments - .iter() - .map(|segment| { - crab_ltx::bundle::BundleEntry::for_cell( - [151; 32], - [152; 16], - segment.info().clone(), - std::fs::read(segment.path()).unwrap(), - ) - }) - .collect(), - Limits::default(), - ) - .unwrap(); - let bundled = replica.prepare_bundle(None, &bundle, 1, 1).await.unwrap(); - assert_eq!( - store.multipart.load(Ordering::SeqCst), - expected_multipart * 3 - ); - let bundle_restore = directory.path().join("bundle"); - bundled.verified().restore(&bundle_restore).await.unwrap(); - assert_eq!( - std::fs::read(bundle_restore).unwrap(), - std::fs::read(after).unwrap() - ); - writer.close().unwrap(); - } -} - -#[tokio::test] -async fn small_body_upload_reconciles_response_loss_without_changing_the_root() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("source"), Limits::default()).unwrap(); - writer - .transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) - .unwrap(); - let batch = writer.capture_deferred().unwrap(); - let store = InstrumentedStore::new(Arc::new(InMemory::new()), Duration::ZERO); - store.arm(PUT_RESPONSE_LOST); - let replica = cell_replica( - Store::new(store.clone()), - [153; 32], - [154; 16], - Host::default(), - ); - let prepared = replica.prepare(None, &batch, 1, 1).await.unwrap(); - assert_eq!(store.fault.load(Ordering::SeqCst), NO_FAULT); - assert_eq!(store.multipart.load(Ordering::SeqCst), 0); - assert_eq!( - store.puts.load(Ordering::SeqCst) as u64, - replica.take_publication_cost().objects + 1 - ); - let restored = directory.path().join("restored"); - prepared.verified().restore(&restored).await.unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT v FROM t", [], |row| row.get::<_, u32>(0)) - .unwrap(), - 7 - ); - writer.close().unwrap(); -} diff --git a/crates/crab-ltx/tests/cell/roots.rs b/crates/crab-ltx/tests/cell/roots.rs deleted file mode 100644 index 1b4050f43..000000000 --- a/crates/crab-ltx/tests/cell/roots.rs +++ /dev/null @@ -1,37 +0,0 @@ -#![cfg(feature = "replica")] - -use std::sync::{ - Arc, - atomic::{AtomicU64, Ordering}, -}; - -use bytes::Bytes; -use crab_ltx::{ - CaptureBatch, CaptureTiming, CellObjectKind, CellReplica, CellStorageLayout, Db, DiskBudget, - Host, Limits, RecoveryOverlay, RootRef, VerifiedPlan, - bundle::{Bundle, BundleEntry}, - restore_exact, -}; -use crab_storage::{StorageReadKind, Store}; -use object_store::{ObjectStoreExt as _, memory::InMemory, path::Path}; - -fn checksum_path(database: &std::path::Path) -> std::path::PathBuf { - let mut path = database.as_os_str().to_owned(); - path.push(".crab-ltx-checksums"); - path.into() -} - -fn replica(store: Store, cell: [u8; 32], incarnation: [u8; 16]) -> CellReplica { - CellReplica::new( - CellStorageLayout::new(store, Path::from("runtime"), [3; 16]), - cell, - incarnation, - Limits::default(), - ) - .unwrap() -} - -mod compaction; -mod directory; -mod lifecycle; -mod sparse; diff --git a/crates/crab-ltx/tests/cell/roots/compaction.rs b/crates/crab-ltx/tests/cell/roots/compaction.rs deleted file mode 100644 index 8c73d690f..000000000 --- a/crates/crab-ltx/tests/cell/roots/compaction.rs +++ /dev/null @@ -1,418 +0,0 @@ -//! Scheduled compaction, large-frame streaming, and corruption refusal. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn range_compaction_reads_and_reserves_only_the_selected_data() { - verify_range_compaction(Arc::new(InMemory::new()), Path::from("runtime")).await; -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires the RustFS environment documented in README.md"] -async fn rustfs_range_compaction_preserves_exact_native_and_bundled_roots() { - let bucket = std::env::var("CRAB_LTX_TEST_BUCKET").unwrap(); - let endpoint = std::env::var("CRAB_LTX_TEST_ENDPOINT").unwrap(); - let store = crab_storage::build_explicit_store( - &bucket, - crab_storage::ObjectStoreCredentials::Aws { - access_key_id: std::env::var("AWS_ACCESS_KEY_ID").unwrap(), - secret_access_key: std::env::var("AWS_SECRET_ACCESS_KEY").unwrap(), - session_token: None, - region: "us-east-1".into(), - }, - Some(&endpoint), - endpoint.starts_with("http://"), - ) - .unwrap(); - let run = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_nanos(); - verify_range_compaction( - store.inner().clone(), - Path::from(format!( - "crab-ltx-tests/range-compaction/{run}-{}", - std::process::id() - )), - ) - .await; -} - -async fn verify_range_compaction(backend: Arc, prefix: Path) { - for (page_size, rows, payload) in [(4096, 4096, 3000), (512, 70000, 400)] { - let prefix = prefix.clone().join(page_size.to_string()); - let source = tempfile::TempDir::new().unwrap(); - let path = source.path().join("range.sqlite"); - let initial = crab_ltx::rusqlite::Connection::open(&path).unwrap(); - initial.execute_batch(&format!("PRAGMA page_size={page_size}; CREATE TABLE t(k INTEGER PRIMARY KEY, v BLOB NOT NULL)")).unwrap(); - drop(initial); - let mut writer = Db::open(&path, Limits::default()).unwrap(); - writer - .transaction(|tx| { - tx.execute( - "WITH RECURSIVE n(k) AS (VALUES(1) UNION ALL SELECT k+1 FROM n WHERE k 16 << 20); - - for (representation, root) in [("native", root), ("bundle", bundled)] { - let scratch_mib = 1; - let slots = Arc::new(tokio::sync::Semaphore::new(scratch_mib)); - let observed = read_bytes.clone(); - let cold = CellReplica::new( - CellStorageLayout::new( - Store::new(backend.clone()).with_read_byte_observer(Arc::new(move |bytes| { - observed.fetch_add(bytes, Ordering::Relaxed); - })), - prefix.clone(), - [3; 16], - ), - [233; 32], - [234; 16], - Limits::default(), - ) - .unwrap(); - let limited = cold.with_host(Host::default().with_scratch_slots(slots.clone())); - read_bytes.store(0, Ordering::Relaxed); - let compacted = limited - .prepare_compaction(&root, start..end, 1, scratch.path()) - .await - .unwrap(); - let cost = limited.take_publication_cost(); - assert!( - cost.objects <= 10, - "range compaction rewrote {} objects", - cost.objects - ); - assert!( - read_bytes.load(Ordering::Relaxed) < 128 << 10, - "a tiny range downloaded {} bytes of unrelated metadata", - read_bytes.load(Ordering::Relaxed) - ); - assert_eq!(slots.available_permits(), scratch_mib); - eprintln!( - "range compaction: {page_size}-byte pages, {representation}, {} read bytes, {} uploaded objects, {scratch_mib} MiB scratch available", - read_bytes.load(Ordering::Relaxed), - cost.objects, - ); - assert_eq!(compacted.predecessor(), Some(root)); - assert_eq!(compacted.root().position, root.position); - assert_eq!(compacted.root().commit_sequence, root.commit_sequence); - let destination = scratch - .path() - .join(format!("range-{representation}.sqlite")); - replica - .open_root(&compacted.root()) - .await - .unwrap() - .restore(&destination) - .await - .unwrap(); - assert_eq!(std::fs::read(destination).unwrap(), expected); - } - } -} - -#[tokio::test] -async fn compaction_scratch_exhaustion_refuses_cleanly() { - let source = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&source.path().join("scratch.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction - .execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(randomblob(1048576))") - }) - .unwrap(); - let batch = writer.capture().unwrap(); - let replica = replica(Store::new(Arc::new(InMemory::new())), [231; 32], [232; 16]); - let root = replica.prepare(None, &batch, 1, 1).await.unwrap().root(); - - // One scratch MiB cannot hold this input and its output, so the request - // must be refused as capacity instead of starting work it cannot finish. - let constrained = replica - .clone() - .with_host(Host::default().with_scratch_slots(Arc::new(tokio::sync::Semaphore::new(1)))); - let scratch = tempfile::TempDir::new().unwrap(); - let error = match constrained - .prepare_compaction(&root, 0..1, 9, scratch.path()) - .await - { - Ok(_) => panic!("an unadmitted compaction must not return a proposal"), - Err(error) => error, - }; - assert_eq!(error.classify(), crab_ltx::FailureClass::Capacity); - assert_eq!( - std::fs::read_dir(scratch.path()).unwrap().count(), - 0, - "a refused compaction must not leave scratch files" - ); - - // The pinned root is untouched and still verifies. - let verified = replica.open_root(&root).await.unwrap(); - assert_eq!(verified.root().position, batch.position); - - // The same request succeeds once the host can admit the scratch it needs. - let compacted = replica - .prepare_compaction(&root, 0..1, 9, scratch.path()) - .await - .unwrap(); - assert_eq!(compacted.root().position, batch.position); - writer.close().unwrap(); -} - -#[tokio::test] -async fn scheduled_cell_compaction_promotes_fanout_and_preserves_root() { - let source = tempfile::TempDir::new().unwrap(); - let database = source.path().join("scheduled.sqlite"); - let mut writer = Db::open(&database, Limits::default()).unwrap(); - let store = Store::new(Arc::new(InMemory::new())); - let replica = replica(store, [41; 32], [42; 16]); - let mut root = None; - for sequence in 1_u64..=8 { - writer - .transaction(|transaction| { - if sequence == 1 { - transaction.execute_batch( - "CREATE TABLE events(sequence INTEGER PRIMARY KEY, value TEXT NOT NULL)", - )?; - } - transaction.execute( - "INSERT INTO events(sequence, value) VALUES (?1, ?2)", - (sequence, format!("event-{sequence}")), - )?; - Ok(()) - }) - .unwrap(); - let batch = writer.capture().unwrap(); - root = Some( - replica - .prepare(root.as_ref(), &batch, sequence, 1) - .await - .unwrap() - .root(), - ); - } - writer.close().unwrap(); - let root = root.unwrap(); - assert_eq!(replica.open_root(&root).await.unwrap().segment_count(), 8); - - let scratch = tempfile::TempDir::new().unwrap(); - let compacted = replica - .prepare_scheduled_compaction(&root, scratch.path()) - .await - .unwrap() - .unwrap(); - assert_eq!(compacted.predecessor(), Some(root)); - assert_eq!(compacted.root().position, root.position); - assert_eq!(compacted.root().commit_sequence, root.commit_sequence); - assert_eq!(compacted.verified().segment_count(), 1); - assert!( - replica - .prepare_scheduled_compaction(&compacted.root(), scratch.path()) - .await - .unwrap() - .is_none() - ); - - let restored = tempfile::TempDir::new().unwrap(); - let before = restored.path().join("before.sqlite"); - let after = restored.path().join("after.sqlite"); - replica - .open_root(&root) - .await - .unwrap() - .restore(&before) - .await - .unwrap(); - replica - .open_root(&compacted.root()) - .await - .unwrap() - .restore(&after) - .await - .unwrap(); - assert_eq!( - std::fs::read(before).unwrap(), - std::fs::read(after).unwrap() - ); -} -#[tokio::test(flavor = "multi_thread")] -async fn compaction_streams_large_frames_and_cleans_scratch() { - let source = tempfile::TempDir::new().unwrap(); - let database = source.path().join("large.sqlite"); - let mut writer = Db::open(&database, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(3000000))", - ) - }) - .unwrap(); - let maximum_read = Arc::new(AtomicU64::new(0)); - let total_read = Arc::new(AtomicU64::new(0)); - let observed = Arc::clone(&maximum_read); - let observed_total = Arc::clone(&total_read); - let store = - Store::new(Arc::new(InMemory::new())).with_read_byte_observer(Arc::new(move |bytes| { - observed.fetch_max(bytes, Ordering::SeqCst); - observed_total.fetch_add(bytes, Ordering::SeqCst); - })); - let replica = replica(store, [91; 32], [92; 16]); - let cuts = writer.capture().unwrap(); - let source_bytes = cuts - .segments - .iter() - .map(|segment| segment.info().size_bytes) - .sum::(); - let root = replica.prepare(None, &cuts, 1, 1).await.unwrap().root(); - writer.close().unwrap(); - let scratch = tempfile::TempDir::new().unwrap(); - maximum_read.store(0, Ordering::SeqCst); - total_read.store(0, Ordering::SeqCst); - - let compacted = replica - .prepare_compaction(&root, 0..1, 9, scratch.path()) - .await - .unwrap(); - - assert_eq!(compacted.root().position, root.position); - assert!( - maximum_read.load(Ordering::SeqCst) <= 1 << 20, - "compaction must not download the complete LTX body" - ); - assert!( - total_read.load(Ordering::SeqCst) <= source_bytes.saturating_add(2 << 20), - "compaction must fetch each selected body only once" - ); - assert_eq!(std::fs::read_dir(scratch.path()).unwrap().count(), 0); - let restored = scratch.path().join("restored.sqlite"); - compacted.verified().restore(&restored).await.unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT length(value) FROM payload", [], |row| { - row.get::<_, u64>(0) - }) - .unwrap(), - 3_000_000 - ); -} -#[tokio::test] -async fn compaction_rejects_selected_body_corruption_outside_page_frames() { - let source = tempfile::TempDir::new().unwrap(); - let database = source.path().join("corrupt.sqlite"); - let mut writer = Db::open(&database, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE messages(body TEXT NOT NULL);\ - INSERT INTO messages VALUES ('verified')", - ) - }) - .unwrap(); - let batch = writer.capture().unwrap(); - let info = batch.segments[0].info().clone(); - let mut corrupted = std::fs::read(batch.segments[0].path()).unwrap(); - corrupted[0] ^= 1; - let inner = Arc::new(InMemory::new()); - let store = Store::new(inner.clone()); - let cell = [93; 32]; - let incarnation = [94; 16]; - let layout = CellStorageLayout::new(store.clone(), Path::from("runtime"), [3; 16]); - let replica = CellReplica::new(layout.clone(), cell, incarnation, Limits::default()).unwrap(); - let root = replica.prepare(None, &batch, 1, 1).await.unwrap().root(); - writer.close().unwrap(); - let object = - layout.incarnation_object_path(&cell, &incarnation, &info.blake3, CellObjectKind::Ltx); - inner - .put(&object, Bytes::from(corrupted).into()) - .await - .unwrap(); - let scratch = tempfile::TempDir::new().unwrap(); - - let error = match replica - .prepare_compaction(&root, 0..1, 9, scratch.path()) - .await - { - Ok(_) => panic!("corrupt selected LTX must not compact"), - Err(error) => error, - }; - - assert!(matches!(error, crab_ltx::CrabError::ChecksumMismatch)); - assert_eq!(std::fs::read_dir(scratch.path()).unwrap().count(), 0); -} diff --git a/crates/crab-ltx/tests/cell/roots/directory.rs b/crates/crab-ltx/tests/cell/roots/directory.rs deleted file mode 100644 index 90830a365..000000000 --- a/crates/crab-ltx/tests/cell/roots/directory.rs +++ /dev/null @@ -1,505 +0,0 @@ -//! Directory growth, sharing, cache survival, and truncate/regrow fencing. - -use super::*; - -#[tokio::test] -async fn changed_cut_loads_only_touched_directory_nodes() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL);\ - INSERT INTO counter VALUES (0);\ - CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(20000000))", - ) - }) - .unwrap(); - let read_bytes = Arc::new(AtomicU64::new(0)); - let observed = read_bytes.clone(); - let store = - Store::new(Arc::new(InMemory::new())).with_read_byte_observer(Arc::new(move |bytes| { - observed.fetch_add(bytes, Ordering::SeqCst); - })); - let replica = replica(store, [41; 32], [42; 16]); - let first = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - - writer - .transaction(|transaction| { - transaction.execute("UPDATE counter SET value = value + 1", [])?; - Ok(()) - }) - .unwrap(); - let next = writer.capture().unwrap(); - read_bytes.store(0, Ordering::SeqCst); - let second = replica.prepare(Some(&first), &next, 2, 1).await.unwrap(); - - assert!( - read_bytes.load(Ordering::SeqCst) < 100_000, - "an incremental root must not reload the full 20 MB snapshot index" - ); - assert_eq!(second.root().position, next.position); - writer.close().unwrap(); -} -#[tokio::test] -async fn directory_nodes_are_shared_across_exact_root_views() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(2000000))", - ) - }) - .unwrap(); - let metadata_reads = Arc::new(AtomicU64::new(0)); - let observed = Arc::clone(&metadata_reads); - let store = - Store::new(Arc::new(InMemory::new())).with_read_request_observer(Arc::new(move |kind| { - if kind == StorageReadKind::Get { - observed.fetch_add(1, Ordering::SeqCst); - } - })); - let replica = replica(store, [81; 32], [82; 16]); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - - metadata_reads.store(0, Ordering::SeqCst); - let first = replica.open_root(&root).await.unwrap(); - assert_eq!(first.directory_height(), 1); - first.paged().read_page(1).await.unwrap(); - let after_first_fault = metadata_reads.load(Ordering::SeqCst); - first.paged().read_page(1).await.unwrap(); - assert_eq!(metadata_reads.load(Ordering::SeqCst), after_first_fault); - - let second = replica.clone().open_root(&root).await.unwrap(); - let before_second_fault = metadata_reads.load(Ordering::SeqCst); - second.paged().read_page(1).await.unwrap(); - assert_eq!(metadata_reads.load(Ordering::SeqCst), before_second_fault); -} -#[tokio::test] -async fn directory_cache_survives_replica_restart_without_directory_origin_read() { - let directory = tempfile::TempDir::new().unwrap(); - let cache_root = directory.path().join("directory-cache"); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(2000000))", - ) - }) - .unwrap(); - - let backend = Arc::new(InMemory::new()); - let cell = [85; 32]; - let incarnation = [86; 16]; - let cache_host = Host::default() - .with_local_disk_budget(DiskBudget::new(64 * 1024 * 1024)) - .with_directory_cache(cache_root.clone()) - .await - .unwrap(); - let first = - replica(Store::new(backend.clone()), cell, incarnation).with_host(cache_host.clone()); - let root = first - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - let warm_replica = - replica(Store::new(backend.clone()), cell, incarnation).with_host(cache_host.clone()); - let warm = warm_replica.open_root(&root).await.unwrap(); - assert_eq!(warm.directory_height(), 1); - warm.paged().read_page(1).await.unwrap(); - cache_host.drain_cache_fills().await; - assert!(cache_host.directory_cache_stats().unwrap().entries() >= 1); - drop(first); - drop(warm_replica); - - let uncached_reads = Arc::new(AtomicU64::new(0)); - let observed = Arc::clone(&uncached_reads); - let uncached_store = - Store::new(backend.clone()).with_read_byte_observer(Arc::new(move |bytes| { - observed.fetch_add(bytes, Ordering::SeqCst); - })); - let uncached = replica(uncached_store, cell, incarnation); - let uncached_root = uncached.open_root(&root).await.unwrap(); - uncached_reads.store(0, Ordering::SeqCst); - uncached_root.paged().read_page(1).await.unwrap(); - let uncached_bytes = uncached_reads.load(Ordering::SeqCst); - assert!(uncached_bytes > 0); - - let cached_reads = Arc::new(AtomicU64::new(0)); - let observed = Arc::clone(&cached_reads); - let cached_store = Store::new(backend).with_read_byte_observer(Arc::new(move |bytes| { - observed.fetch_add(bytes, Ordering::SeqCst); - })); - let cached_host = Host::default() - .with_local_disk_budget(DiskBudget::new(64 * 1024 * 1024)) - .with_directory_cache(cache_root) - .await - .unwrap(); - let cached = replica(cached_store, cell, incarnation).with_host(cached_host); - let cached_root = cached.open_root(&root).await.unwrap(); - cached_reads.store(0, Ordering::SeqCst); - cached_root.paged().read_page(1).await.unwrap(); - let cached_bytes = cached_reads.load(Ordering::SeqCst); - - assert!( - cached_bytes < uncached_bytes, - "a restarted replica must avoid the persisted directory-node origin read" - ); -} -#[tokio::test] -async fn directory_growth_adds_authenticated_parent_level() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(700000))", - ) - }) - .unwrap(); - let replica = replica(Store::new(Arc::new(InMemory::new())), [61; 32], [62; 16]); - let first = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap(); - assert_eq!(first.verified().directory_height(), 0); - - writer - .transaction(|transaction| { - transaction.execute("INSERT INTO payload VALUES(randomblob(700000))", [])?; - Ok(()) - }) - .unwrap(); - let second = replica - .prepare(Some(&first.root()), &writer.capture().unwrap(), 2, 1) - .await - .unwrap(); - assert_eq!(second.verified().directory_height(), 1); - let last = second.verified().paged().page_count(); - assert_eq!( - second - .verified() - .paged() - .read_page(last) - .await - .unwrap() - .len(), - second.verified().page_size() as usize - ); - writer.close().unwrap(); -} -#[tokio::test] -async fn truncate_regrow_cannot_reuse_old_locator() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("cell.sqlite"); - let initial = crab_ltx::rusqlite::Connection::open(&path).unwrap(); - initial - .execute_batch("PRAGMA auto_vacuum = FULL; VACUUM") - .unwrap(); - drop(initial); - let mut writer = Db::open(&path, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(zeroblob(4000000))", - ) - }) - .unwrap(); - let replica = replica(Store::new(Arc::new(InMemory::new())), [51; 32], [52; 16]); - let first_batch = writer.capture().unwrap(); - let first_pages = first_batch.segments.last().unwrap().info().database_pages; - let first = replica - .prepare(None, &first_batch, 1, 1) - .await - .unwrap() - .root(); - - writer - .transaction(|transaction| { - transaction.execute("DELETE FROM payload", [])?; - Ok(()) - }) - .unwrap(); - let truncated_batch = writer.capture().unwrap(); - let truncated_pages = truncated_batch - .segments - .last() - .unwrap() - .info() - .database_pages; - assert!(truncated_pages < first_pages); - let truncated = replica - .prepare(Some(&first), &truncated_batch, 2, 1) - .await - .unwrap() - .root(); - - writer - .transaction(|transaction| { - transaction.execute("INSERT INTO payload VALUES(randomblob(4000000))", [])?; - Ok(()) - }) - .unwrap(); - let regrown_batch = writer.capture().unwrap(); - let regrown = replica - .prepare(Some(&truncated), ®rown_batch, 3, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - - let restored = directory.path().join("restored.sqlite"); - let writable = replica - .open_root(®rown) - .await - .unwrap() - .paged() - .prepare_writable(&restored) - .await - .unwrap(); - let restored_for_open = restored.clone(); - let mut replacement = - tokio::task::spawn_blocking(move || writable.open_writable(&restored_for_open)) - .await - .unwrap() - .unwrap(); - let (length, is_zero): (u32, bool) = replacement - .transaction(|transaction| { - transaction.query_row( - "SELECT length(value), value = zeroblob(length(value)) FROM payload", - [], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - }) - .unwrap(); - assert_eq!(length, 4_000_000); - assert!(!is_zero); - replacement.close().unwrap(); -} -#[tokio::test] -async fn initial_streaming_directory_merges_truncation_and_regrowth() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("cell.sqlite"); - let initial = crab_ltx::rusqlite::Connection::open(&path).unwrap(); - initial - .execute_batch("PRAGMA page_size = 512; PRAGMA auto_vacuum = FULL; VACUUM") - .unwrap(); - drop(initial); - let mut writer = Db::open(&path, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(zeroblob(4000000))", - ) - }) - .unwrap(); - let first = writer.capture().unwrap(); - writer - .transaction(|transaction| { - transaction.execute("DELETE FROM payload", [])?; - Ok(()) - }) - .unwrap(); - let truncated = writer.capture().unwrap(); - let truncated_position = truncated.position; - let truncated_count = first.segments.len() + truncated.segments.len(); - writer - .transaction(|transaction| { - transaction.execute("INSERT INTO payload VALUES(randomblob(4000000))", [])?; - Ok(()) - }) - .unwrap(); - let regrown = writer.capture().unwrap(); - let mut segments = first.segments; - segments.extend(truncated.segments); - segments.extend(regrown.segments); - let batch = CaptureBatch { - segments, - position: regrown.position, - timing: CaptureTiming::default(), - }; - let segment_count = batch.segments.len(); - - let replica = replica(Store::new(Arc::new(InMemory::new())), [53; 32], [54; 16]); - let prepared = replica.prepare(None, &batch, 1, 1).await.unwrap(); - assert_eq!(prepared.verified().segment_count(), segment_count); - writer.close().unwrap(); - - let restored = directory.path().join("streamed.sqlite"); - prepared.verified().restore(&restored).await.unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(&restored).unwrap(); - let (length, is_zero): (u32, bool) = connection - .query_row( - "SELECT length(value), value = zeroblob(length(value)) FROM payload", - [], - |row| Ok((row.get(0)?, row.get(1)?)), - ) - .unwrap(); - assert_eq!(length, 4_000_000); - assert!(!is_zero); - drop(connection); - let expected = std::fs::read(&restored).unwrap(); - // The old image spans multiple merge jobs. Compacting through its - // truncation must yield even when a batch contains no surviving pages. - for (range, level) in [(0..2, 1), (0..segment_count, 9)] { - let scratch = tempfile::TempDir::new().unwrap(); - let compacted = replica - .prepare_compaction(&prepared.root(), range, level, scratch.path()) - .await - .unwrap(); - let destination = scratch.path().join("compacted.sqlite"); - compacted.verified().restore(&destination).await.unwrap(); - assert_eq!(std::fs::read(destination).unwrap(), expected); - } - - // Compact the original image while the exact endpoint is still truncated. - // Its old suffix must be discarded rather than reintroduced by relocation. - let truncated_batch = CaptureBatch { - segments: batch.segments[..truncated_count].to_vec(), - position: truncated_position, - timing: CaptureTiming::default(), - }; - let truncated = replica.prepare(None, &truncated_batch, 1, 1).await.unwrap(); - let expected_path = directory.path().join("truncated.sqlite"); - truncated.verified().restore(&expected_path).await.unwrap(); - let scratch = tempfile::TempDir::new().unwrap(); - let compacted = replica - .prepare_compaction(&truncated.root(), 0..1, 1, scratch.path()) - .await - .unwrap(); - let destination = scratch.path().join("compacted.sqlite"); - compacted.verified().restore(&destination).await.unwrap(); - assert_eq!( - std::fs::read(destination).unwrap(), - std::fs::read(expected_path).unwrap() - ); -} - -#[tokio::test] -async fn writable_activation_preserves_leaf_order_across_parent_branches() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("source.sqlite"); - let initial = crab_ltx::rusqlite::Connection::open(&path).unwrap(); - initial - .execute_batch("PRAGMA page_size=512; CREATE TABLE payload(value BLOB)") - .unwrap(); - drop(initial); - let mut writer = Db::open(&path, Limits::default()).unwrap(); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO payload VALUES(zeroblob(40000000))")) - .unwrap(); - let replica = replica(Store::new(Arc::new(InMemory::new())), [101; 32], [102; 16]); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - let verified = replica.open_root(&root).await.unwrap(); - assert_eq!(verified.directory_height(), 2); - let destination = directory.path().join("restored.sqlite"); - let prepared = verified - .paged() - .prepare_writable(&destination) - .await - .unwrap(); - let mut restored = prepared.open_writable(&destination).unwrap(); - let length: u64 = restored - .query_with(|db| db.query_row("SELECT length(value) FROM payload", [], |row| row.get(0))) - .unwrap(); - assert_eq!(length, 40_000_000); - restored.close().unwrap(); - writer.close().unwrap(); -} - -#[tokio::test] -async fn writable_activation_rejects_a_corrupt_late_leaf_and_cleans_its_file() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("source.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|tx| { - tx.execute_batch( - "CREATE TABLE payload(value); INSERT INTO payload VALUES(zeroblob(10000000))", - ) - }) - .unwrap(); - let backend = Arc::new(InMemory::new()); - let layout = CellStorageLayout::new( - Store::new(backend.clone()), - Path::from("activation-corruption"), - [103; 16], - ); - let cell = [104; 32]; - let incarnation = [105; 16]; - let source = CellReplica::new(layout.clone(), cell, incarnation, Limits::default()).unwrap(); - let root = source - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - let mut leaves = Vec::new(); - for object in source.reachable_objects(&root).await.unwrap() { - if object.kind == CellObjectKind::Directory { - let path = - layout.incarnation_object_path(&cell, &incarnation, &object.digest, object.kind); - let bytes = backend.get(&path).await.unwrap().bytes().await.unwrap(); - // CRBDIR01 kind and first page locate a late leaf independently of - // object digest order, so earlier prefetched leaves can succeed. - if bytes[10] == 0 { - let first = u32::from_be_bytes(bytes[32..36].try_into().unwrap()); - leaves.push((first, path, bytes)); - } - } - } - let (first, path, original) = leaves - .into_iter() - .max_by_key(|(first, _, _)| *first) - .unwrap(); - assert!(first > 256); - let io = Arc::new(tokio::sync::Semaphore::new(4)); - let reader = CellReplica::new( - CellStorageLayout::new( - Store::new(backend.clone()), - Path::from("activation-corruption"), - [103; 16], - ), - cell, - incarnation, - Limits::default(), - ) - .unwrap() - .with_host(Host::default().with_io_slots(io.clone())); - let paged = reader.open_root(&root).await.unwrap().paged(); - let mut corrupt = original.to_vec(); - *corrupt.last_mut().unwrap() ^= 1; - backend - .put(&path, Bytes::from(corrupt).into()) - .await - .unwrap(); - let destination = directory.path().join("active.sqlite"); - assert!(matches!( - paged.clone().prepare_writable(&destination).await, - Err(crab_ltx::CrabError::ChecksumMismatch) - )); - assert!(!checksum_path(&destination).exists()); - assert_eq!(io.available_permits(), 4); - backend.put(&path, original.into()).await.unwrap(); - paged.prepare_writable(&destination).await.unwrap(); - writer.close().unwrap(); -} diff --git a/crates/crab-ltx/tests/cell/roots/lifecycle.rs b/crates/crab-ltx/tests/cell/roots/lifecycle.rs deleted file mode 100644 index e06b7277f..000000000 --- a/crates/crab-ltx/tests/cell/roots/lifecycle.rs +++ /dev/null @@ -1,683 +0,0 @@ -//! Exact-root preparation, inventory verification, and overlay recovery. - -use super::*; - -#[tokio::test] -async fn read_only_roots_keep_exact_snapshots_without_materializing_pages() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("source.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction - .execute_batch("CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES(1)") - }) - .unwrap(); - let first = writer.capture().unwrap(); - let disk = DiskBudget::new(8 << 20); - let replica = replica(Store::new(Arc::new(InMemory::new())), [41; 32], [42; 16]) - .with_host(Host::default().with_local_disk_budget(disk.clone())); - let first_root = replica.prepare(None, &first, 1, 1).await.unwrap().root(); - let first_path = directory.path().join("first.sqlite"); - let first_view = replica - .open_root(&first_root) - .await - .unwrap() - .open_read_only(&first_path) - .unwrap(); - assert_eq!(first_view.root(), first_root); - assert_eq!(disk.used(), 0); - assert_eq!(std::fs::metadata(&first_path).unwrap().len(), 0); - assert!( - replica - .open_root(&first_root) - .await - .unwrap() - .open_read_only(&first_path) - .is_err() - ); - assert!(first_path.exists()); - { - let connection = first_view.connection().unwrap(); - assert_eq!( - connection - .query_row("PRAGMA cache_size", [], |row| row.get::<_, i64>(0)) - .unwrap(), - -64 - ); - let value: i64 = connection - .query_row("SELECT value FROM counter", [], |row| row.get(0)) - .unwrap(); - assert_eq!(value, 1); - connection.pragma_update(None, "query_only", false).unwrap(); - assert!( - connection - .is_readonly(rusqlite::DatabaseName::Main) - .unwrap() - ); - assert!( - connection - .execute("UPDATE counter SET value=99", []) - .is_err() - ); - } - - writer - .transaction(|transaction| transaction.execute_batch("UPDATE counter SET value=2")) - .unwrap(); - let second = writer.capture().unwrap(); - let second_root = replica - .prepare(Some(&first_root), &second, 2, 1) - .await - .unwrap() - .root(); - let second_path = directory.path().join("second.sqlite"); - let second_view = replica - .open_root(&second_root) - .await - .unwrap() - .open_read_only(&second_path) - .unwrap(); - let old: i64 = first_view - .connection() - .unwrap() - .query_row("SELECT value FROM counter", [], |row| row.get(0)) - .unwrap(); - let current: i64 = second_view - .connection() - .unwrap() - .query_row("SELECT value FROM counter", [], |row| row.get(0)) - .unwrap(); - assert_eq!((old, current), (1, 2)); - - for path in [&first_path, &second_path] { - assert_eq!(std::fs::metadata(path).unwrap().len(), 0); - for suffix in ["-wal", "-shm", "-journal", ".crab-ltx"] { - assert!(!std::path::PathBuf::from(format!("{}{suffix}", path.display())).exists()); - } - } - drop(first_view); - drop(second_view); - assert!(!first_path.exists()); - assert!(!second_path.exists()); - assert_eq!(disk.used(), 0); - writer.close().unwrap(); -} - -#[tokio::test] -async fn small_appends_report_a_bounded_publication_cost() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(randomblob(4096))") - }) - .unwrap(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - [3; 16], - ), - [211; 32], - [212; 16], - Limits::default(), - ) - .unwrap(); - - // A bootstrap root uploads one body, one index, the changed directory - // nodes, and the root document. The bound documents the per-command - // object-store amplification the runtime budgets against. - let first = writer.capture().unwrap(); - let root = replica.prepare(None, &first, 1, 1).await.unwrap().root(); - let initial = replica.take_publication_cost(); - assert!( - (4..=12).contains(&initial.objects), - "bootstrap objects: {initial:?}" - ); - assert!( - initial.bytes >= first.segments[0].info().size_bytes, - "{initial:?}" - ); - - // One appended command pays at least a body and index, and no more than the - // same bounded set of metadata objects. - writer - .transaction(|transaction| { - transaction.execute_batch("INSERT INTO t VALUES(randomblob(4096))") - }) - .unwrap(); - let second = writer.capture().unwrap(); - replica.prepare(Some(&root), &second, 2, 1).await.unwrap(); - let append = replica.take_publication_cost(); - assert!( - (3..=12).contains(&append.objects), - "append objects: {append:?}" - ); - assert!( - append.bytes >= second.segments[0].info().size_bytes, - "{append:?}" - ); - assert_eq!(replica.publication_cost().objects, 0); -} - -#[tokio::test] -async fn oversized_commit_publishes_as_a_full_image_root() { - let directory = tempfile::TempDir::new().unwrap(); - // The incremental bound is far below the commit and the database, so the - // capture escalates and the publication must admit the full image. - let limits = Limits { - max_capture_bytes: 8 * 1024, - ..Limits::default() - }; - let mut writer = Db::open(&directory.path().join("cell.sqlite"), limits).unwrap(); - writer - .transaction(|transaction| transaction.execute_batch("CREATE TABLE payload(value BLOB)")) - .unwrap(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("runtime"), - [3; 16], - ), - [201; 32], - [202; 16], - limits, - ) - .unwrap(); - let first = writer.capture().unwrap(); - let first_root = replica.prepare(None, &first, 1, 1).await.unwrap().root(); - - writer - .transaction(|transaction| { - transaction.execute_batch("INSERT INTO payload VALUES(randomblob(65536))") - }) - .unwrap(); - let second = writer.capture().unwrap(); - assert!( - second.segments[0].info().size_bytes > limits.max_capture_bytes, - "the escalated cut must exceed the incremental bound" - ); - - let prepared = replica - .prepare(Some(&first_root), &second, 2, 1) - .await - .unwrap(); - let root = prepared.root(); - assert_eq!(prepared.verified().segment_count(), 2); - assert_eq!(root.position, second.position); - - // The published root restores the escalated commit exactly. - let restored = directory.path().join("restored.sqlite"); - replica - .open_root(&root) - .await - .unwrap() - .restore(&restored) - .await - .unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(&restored).unwrap(); - let rows: i64 = connection - .query_row("SELECT count(*) FROM payload", [], |row| row.get(0)) - .unwrap(); - let bytes: i64 = connection - .query_row("SELECT length(value) FROM payload", [], |row| row.get(0)) - .unwrap(); - assert_eq!((rows, bytes), (1, 65536)); - writer.close().unwrap(); -} - -#[tokio::test] -async fn prepared_cell_handles_release_dirty_admission() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let dirty = Arc::new(tokio::sync::Semaphore::new(1)); - let host = Host::default().with_dirty_slots(dirty.clone()); - let replica = - replica(Store::new(Arc::new(InMemory::new())), [101; 32], [102; 16]).with_host(host); - - let prepared = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap(); - assert_eq!(dirty.available_permits(), 1); - let writable = prepared - .verified() - .paged() - .prepare_writable(&directory.path().join("active.sqlite")) - .await - .unwrap(); - assert_eq!(dirty.available_permits(), 1); - - drop(writable); - writer.close().unwrap(); -} - -#[tokio::test] -async fn prepared_root_reopens_without_a_mutable_head() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE messages(id INTEGER PRIMARY KEY, body TEXT);\ - INSERT INTO messages(body) VALUES ('first');\ - CREATE TABLE payload(value BLOB);\ - INSERT INTO payload VALUES(randomblob(2000000))", - ) - }) - .unwrap(); - let read_bytes = Arc::new(AtomicU64::new(0)); - let observed = read_bytes.clone(); - let store = - Store::new(Arc::new(InMemory::new())).with_read_byte_observer(Arc::new(move |bytes| { - observed.fetch_add(bytes, Ordering::SeqCst); - })); - let replica = replica(store.clone(), [1; 32], [2; 16]); - let first = writer.capture().unwrap(); - let prepared = replica.prepare(None, &first, 1, 7).await.unwrap(); - let first_root = prepared.root(); - assert_eq!(prepared.predecessor(), None); - assert_eq!(prepared.verified().schema(), 7); - - writer - .transaction(|transaction| { - transaction.execute("INSERT INTO messages(body) VALUES ('second')", [])?; - Ok(()) - }) - .unwrap(); - let second = writer.capture().unwrap(); - let prepared = replica - .prepare(Some(&first_root), &second, 2, 7) - .await - .unwrap(); - assert_eq!(prepared.predecessor(), Some(first_root)); - assert_eq!(prepared.verified().segment_count(), 2); - assert_eq!(prepared.root().position, second.position); - - let expected_dir = tempfile::TempDir::new().unwrap(); - let expected_path = expected_dir.path().join("expected.sqlite"); - let segments = first - .segments - .iter() - .chain(&second.segments) - .cloned() - .collect::>(); - let plan = VerifiedPlan::new(&segments, second.position, Limits::default()).unwrap(); - restore_exact(&plan, &expected_path).unwrap(); - let expected = std::fs::read(expected_path).unwrap(); - writer.close().unwrap(); - directory.close().unwrap(); - - read_bytes.store(0, Ordering::SeqCst); - let reopened = replica.open_root(&prepared.root()).await.unwrap(); - assert_eq!(reopened.root(), prepared.root()); - assert_eq!( - reopened.database_pages(), - prepared.verified().database_pages() - ); - assert!(reopened.page_size().is_power_of_two()); - assert!( - read_bytes.load(Ordering::SeqCst) < 100_000, - "activation must not fetch the LTX bodies or every directory leaf" - ); - let restored_path = expected_dir.path().join("streamed.sqlite"); - assert_eq!( - reopened.restore(&restored_path).await.unwrap(), - prepared.root().position - ); - assert_eq!(std::fs::read(&restored_path).unwrap(), expected); - read_bytes.store(0, Ordering::SeqCst); - assert!(reopened.restore(&restored_path).await.is_err()); - assert_eq!(read_bytes.load(Ordering::SeqCst), 0); - assert_eq!(std::fs::read(restored_path).unwrap(), expected); - - let scratch = Arc::new(tokio::sync::Semaphore::new(64)); - let limited = replica - .clone() - .with_host(Host::default().with_scratch_slots(scratch.clone())) - .open_root(&prepared.root()) - .await - .unwrap(); - read_bytes.store(0, Ordering::SeqCst); - let rejected_path = expected_dir.path().join("scratch-rejected.sqlite"); - assert!(matches!( - limited.restore(&rejected_path).await, - Err(crab_ltx::CrabError::Limit( - crab_ltx::LimitKind::ScratchDiskBytes - )) - )); - assert_eq!(scratch.available_permits(), 64); - assert_eq!(read_bytes.load(Ordering::SeqCst), 0); - assert!(!rejected_path.exists()); -} -#[tokio::test] -async fn exact_root_inventory_verifies_every_remote_dependency() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(2000000))", - ) - }) - .unwrap(); - let backend = Arc::new(InMemory::new()); - let store = Store::new(backend.clone()); - let cell = [31; 32]; - let incarnation = [32; 16]; - let layout = CellStorageLayout::new(store.clone(), Path::from("runtime"), [3; 16]); - let replica = CellReplica::new(layout.clone(), cell, incarnation, Limits::default()).unwrap(); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - - let objects = replica.reachable_objects(&root).await.unwrap(); - assert!(objects.windows(2).all(|pair| pair[0] < pair[1])); - for kind in [ - CellObjectKind::Ltx, - CellObjectKind::Index, - CellObjectKind::Directory, - CellObjectKind::Root, - ] { - assert!(objects.iter().any(|object| object.kind == kind)); - } - for object in &objects { - let path = layout.incarnation_object_path(&cell, &incarnation, &object.digest, object.kind); - store.head(&path).await.unwrap(); - } - - let missing = objects - .iter() - .find(|object| object.kind == CellObjectKind::Directory) - .unwrap(); - let missing_path = - layout.incarnation_object_path(&cell, &incarnation, &missing.digest, missing.kind); - backend.delete(&missing_path).await.unwrap(); - assert!(replica.reachable_objects(&root).await.is_err()); -} -#[tokio::test] -async fn warm_root_cache_does_not_mask_missing_metadata() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let backend = Arc::new(InMemory::new()); - let cell = [41; 32]; - let incarnation = [42; 16]; - let layout = - CellStorageLayout::new(Store::new(backend.clone()), Path::from("runtime"), [3; 16]); - let replica = CellReplica::new(layout.clone(), cell, incarnation, Limits::default()).unwrap(); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - let path = - layout.incarnation_object_path(&cell, &incarnation, &root.digest, CellObjectKind::Root); - let root_bytes = backend.get(&path).await.unwrap().bytes().await.unwrap(); - let segment_page = replica - .reachable_objects(&root) - .await - .unwrap() - .into_iter() - .find(|object| object.kind == CellObjectKind::Root && object.digest != root.digest) - .unwrap(); - backend.delete(&path).await.unwrap(); - - assert!(replica.reachable_objects(&root).await.is_err()); - writer - .transaction(|transaction| transaction.execute_batch("INSERT INTO values_ VALUES(2)")) - .unwrap(); - let next = writer.capture().unwrap(); - assert!(replica.prepare(Some(&root), &next, 2, 1).await.is_err()); - backend.put(&path, root_bytes.into()).await.unwrap(); - let segment_path = layout.incarnation_object_path( - &cell, - &incarnation, - &segment_page.digest, - CellObjectKind::Root, - ); - backend.delete(&segment_path).await.unwrap(); - assert!(replica.prepare(Some(&root), &next, 2, 1).await.is_err()); - writer.close().unwrap(); -} -#[tokio::test] -async fn root_scope_and_commit_sequence_are_fenced() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let batch = writer.capture().unwrap(); - let store = Store::new(Arc::new(InMemory::new())); - let replica = replica(store, [1; 32], [2; 16]); - let root = replica.prepare(None, &batch, 4, 1).await.unwrap().root(); - assert!(replica.prepare(Some(&root), &batch, 4, 1).await.is_err()); - let wrong = RootRef { - cell: [9; 32], - ..root - }; - assert!(replica.open_root(&wrong).await.is_err()); - writer.close().unwrap(); -} -#[tokio::test(flavor = "multi_thread")] -async fn prepare_does_not_write_mutable_keys() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("cell.sqlite"); - let mut writer = Db::open(&path, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE messages(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ - INSERT INTO messages(body) VALUES ('first')", - ) - }) - .unwrap(); - let first = writer.capture().unwrap(); - writer - .transaction(|transaction| { - transaction.execute("INSERT INTO messages(body) VALUES ('second')", [])?; - Ok(()) - }) - .unwrap(); - let second = writer.capture().unwrap(); - - let cell = [71; 32]; - let incarnation = [72; 16]; - let first_segment = first.segments.first().unwrap(); - let second_segment = second.segments.first().unwrap(); - let bundle = Bundle::encode( - vec![ - BundleEntry::for_cell( - cell, - incarnation, - first_segment.info().clone(), - std::fs::read(first_segment.path()).unwrap(), - ), - BundleEntry::for_cell( - [73; 32], - incarnation, - first_segment.info().clone(), - std::fs::read(first_segment.path()).unwrap(), - ), - BundleEntry::for_cell( - cell, - incarnation, - second_segment.info().clone(), - std::fs::read(second_segment.path()).unwrap(), - ), - ], - Limits::default(), - ) - .unwrap(); - let bundle_file = tempfile::NamedTempFile::new().unwrap(); - std::fs::write(bundle_file.path(), bundle.read_all().unwrap()).unwrap(); - let bundle = Bundle::decode_file(bundle_file.path(), Limits::default()).unwrap(); - let store = Store::new(Arc::new(InMemory::new())); - let replica = replica(store.clone(), cell, incarnation); - let bundled = replica.prepare_bundle(None, &bundle, 2, 5).await.unwrap(); - assert_eq!(bundled.verified().segment_count(), 2); - assert_eq!(bundled.root().position, second.position); - - writer - .transaction(|transaction| { - transaction.execute("INSERT INTO messages(body) VALUES ('third')", [])?; - Ok(()) - }) - .unwrap(); - let third = writer.capture().unwrap(); - let appended = replica - .prepare(Some(&bundled.root()), &third, 3, 5) - .await - .unwrap(); - writer.close().unwrap(); - directory.close().unwrap(); - let compaction_scratch = tempfile::TempDir::new().unwrap(); - - let compacted = replica - .prepare_compaction(&appended.root(), 0..2, 1, compaction_scratch.path()) - .await - .unwrap(); - assert_eq!(compacted.predecessor(), Some(appended.root())); - assert_eq!(compacted.root().position, appended.root().position); - assert_eq!(compacted.root().commit_sequence, 3); - assert_eq!(compacted.verified().segment_count(), 2); - assert_ne!(compacted.root().digest, appended.root().digest); - - let snapshot = replica - .prepare_compaction(&compacted.root(), 0..2, 9, compaction_scratch.path()) - .await - .unwrap(); - assert_eq!(snapshot.verified().segment_count(), 1); - assert_eq!(snapshot.root().position, appended.root().position); - let written = store - .list_prefix(&Path::from("runtime/cells/v1")) - .await - .unwrap(); - assert!( - !written.is_empty() - && written - .iter() - .all(|object| object.location.as_ref().contains("/objects/")), - "root preparation must write only immutable dependency objects" - ); - assert!( - written - .iter() - .all(|object| !object.location.as_ref().contains("/.staging/")), - "bundle staging objects must not remain after preparation" - ); - let destination = tempfile::TempDir::new().unwrap(); - let path = destination.path().join("restored.sqlite"); - let writable = replica - .open_root(&snapshot.root()) - .await - .unwrap() - .paged() - .prepare_writable(&path) - .await - .unwrap(); - let mut restored = tokio::task::spawn_blocking(move || writable.open_writable(&path)) - .await - .unwrap() - .unwrap(); - assert_eq!( - restored - .transaction(|transaction| { - transaction.query_row("SELECT count(*) FROM messages", [], |row| { - row.get::<_, u32>(0) - }) - }) - .unwrap(), - 3 - ); - restored.close().unwrap(); -} -#[tokio::test] -async fn recovered_overlay_requires_exact_predecessor_and_final_position() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = - Db::open(&directory.path().join("recovery.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ - INSERT INTO events(body) VALUES ('published')", - ) - }) - .unwrap(); - let first = writer.capture().unwrap(); - let cell = [81; 32]; - let incarnation = [82; 16]; - let replica = replica(Store::new(Arc::new(InMemory::new())), cell, incarnation); - let base = replica.prepare(None, &first, 1, 6).await.unwrap().root(); - - writer - .transaction(|transaction| { - transaction.execute("INSERT INTO events(body) VALUES ('fleet-only')", [])?; - Ok(()) - }) - .unwrap(); - let tail = writer.capture().unwrap(); - let entries = tail - .segments - .iter() - .map(|segment| { - BundleEntry::for_cell( - cell, - incarnation, - segment.info().clone(), - std::fs::read(segment.path()).unwrap(), - ) - }) - .collect(); - let bundle = Bundle::encode(entries, Limits::default()).unwrap(); - let overlay = RecoveryOverlay::new(base, bundle, tail.position, 2); - let recovered = replica - .prepare_recovered_overlay(&overlay, 6) - .await - .unwrap(); - - assert_eq!(recovered.predecessor(), Some(base)); - assert_eq!(recovered.root().position, tail.position); - assert_eq!(recovered.root().commit_sequence, 2); - - let invalid = RecoveryOverlay::new( - RootRef { - cell: [83; 32], - ..base - }, - Bundle::encode( - tail.segments - .iter() - .map(|segment| { - BundleEntry::for_cell( - cell, - incarnation, - segment.info().clone(), - std::fs::read(segment.path()).unwrap(), - ) - }) - .collect(), - Limits::default(), - ) - .unwrap(), - tail.position, - 2, - ); - assert!( - replica - .prepare_recovered_overlay(&invalid, 6) - .await - .is_err() - ); - writer.close().unwrap(); -} diff --git a/crates/crab-ltx/tests/cell/roots/sparse.rs b/crates/crab-ltx/tests/cell/roots/sparse.rs deleted file mode 100644 index 4eb42fabc..000000000 --- a/crates/crab-ltx/tests/cell/roots/sparse.rs +++ /dev/null @@ -1,755 +0,0 @@ -//! Sparse writer publication and hydration coalescing. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn exact_cell_root_opens_sparse_writer_and_publishes_incrementally() { - let source = tempfile::TempDir::new().unwrap(); - let source_path = source.path().join("source.sqlite"); - let mut writer = Db::open(&source_path, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE messages(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ - INSERT INTO messages(body) VALUES ('first'), ('second');\ - CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(2000000))", - ) - }) - .unwrap(); - let store = Store::new(Arc::new(InMemory::new())); - let replica = replica(store, [8; 32], [9; 16]); - let root = replica - .prepare(None, &writer.capture().unwrap(), 0, 3) - .await - .unwrap() - .root(); - writer.close().unwrap(); - source.close().unwrap(); - - let active = tempfile::TempDir::new().unwrap(); - let active_path = active.path().join("active.sqlite"); - let writable = replica - .open_root(&root) - .await - .unwrap() - .paged() - .prepare_writable(&active_path) - .await - .unwrap(); - let checksum_bytes = std::fs::metadata(checksum_path(&active_path)) - .unwrap() - .len(); - assert!(checksum_bytes > 0 && checksum_bytes.is_multiple_of(8)); - let mut writer = tokio::task::spawn_blocking(move || writable.open_writable(&active_path)) - .await - .unwrap() - .unwrap(); - assert_eq!(writer.position(), root.position); - assert!(!writer.hydration().unwrap().unwrap().complete()); - writer - .transaction(|transaction| { - assert_eq!( - transaction.query_row("SELECT count(*) FROM messages", [], |row| row - .get::<_, u32>(0))?, - 2 - ); - transaction.execute("INSERT INTO messages(body) VALUES ('third')", [])?; - Ok(()) - }) - .unwrap(); - let next = replica - .prepare(Some(&root), &writer.capture().unwrap(), 1, 3) - .await - .unwrap() - .root(); - writer.close().unwrap(); - assert_eq!(next.commit_sequence, 1); - - let replacement = tempfile::TempDir::new().unwrap(); - let replacement_path = replacement.path().join("replacement.sqlite"); - let writable = replica - .open_root(&next) - .await - .unwrap() - .paged() - .prepare_writable(&replacement_path) - .await - .unwrap(); - let mut replacement = - tokio::task::spawn_blocking(move || writable.open_writable(&replacement_path)) - .await - .unwrap() - .unwrap(); - let count = replacement - .transaction(|transaction| { - transaction.query_row("SELECT count(*) FROM messages", [], |row| { - row.get::<_, u32>(0) - }) - }) - .unwrap(); - assert_eq!(count, 3); - replacement.close().unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn unproved_sparse_cut_and_lost_sidecar_do_not_advance_the_selected_root() { - let directory = tempfile::TempDir::new().unwrap(); - let source = directory.path().join("source.sqlite"); - let mut writer = Db::open(&source, Limits::default()).unwrap(); - writer - .transaction(|tx| { - tx.execute_batch("CREATE TABLE values_(v INTEGER); INSERT INTO values_ VALUES(1)") - }) - .unwrap(); - let replica = replica(Store::new(Arc::new(InMemory::new())), [91; 32], [92; 16]); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - - let active = directory.path().join("active.sqlite"); - let writable = replica - .open_root(&root) - .await - .unwrap() - .paged() - .prepare_writable(&active) - .await - .unwrap(); - let mut writer = writable.open_writable(&active).unwrap(); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO values_ VALUES(2)")) - .unwrap(); - let unproved = writer.capture_deferred().unwrap(); - assert!(unproved.position.txid > root.position.txid); - writer.close().unwrap(); - std::fs::remove_file(checksum_path(&active)).unwrap(); - std::fs::remove_file(&active).unwrap(); - - let recovered = directory.path().join("recovered.sqlite"); - let writable = replica - .open_root(&root) - .await - .unwrap() - .paged() - .prepare_writable(&recovered) - .await - .unwrap(); - let mut writer = writable.open_writable(&recovered).unwrap(); - assert_eq!(writer.position(), root.position); - let count: i64 = writer - .query_with(|db| db.query_row("SELECT count(*) FROM values_", [], |row| row.get(0))) - .unwrap(); - assert_eq!(count, 1); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO values_ VALUES(3)")) - .unwrap(); - assert_eq!( - writer.capture_deferred().unwrap().position.txid, - root.position.txid + 1 - ); - writer.close().unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn sparse_crash_writer() { - use std::io::{Read, Write}; - - let Ok(store_path) = std::env::var("CRAB_LTX_SPARSE_CRASH_STORE") else { - return; - }; - let active = std::path::PathBuf::from(std::env::var("CRAB_LTX_SPARSE_CRASH_ACTIVE").unwrap()); - let digest = std::env::var("CRAB_LTX_SPARSE_CRASH_DIGEST").unwrap(); - let mut root_digest = [0; 32]; - for (index, byte) in root_digest.iter_mut().enumerate() { - *byte = u8::from_str_radix(&digest[index * 2..index * 2 + 2], 16).unwrap(); - } - let root = RootRef { - cell: [91; 32], - incarnation: [92; 16], - digest: root_digest, - position: crab_ltx::Position { - txid: std::env::var("CRAB_LTX_SPARSE_CRASH_TXID") - .unwrap() - .parse() - .unwrap(), - checksum: std::env::var("CRAB_LTX_SPARSE_CRASH_CHECKSUM") - .unwrap() - .parse() - .unwrap(), - }, - commit_sequence: 1, - }; - let store = Store::new(Arc::new( - object_store::local::LocalFileSystem::new_with_prefix(store_path).unwrap(), - )); - let replica = replica(store, root.cell, root.incarnation); - let writable = replica - .open_root(&root) - .await - .unwrap() - .paged() - .prepare_writable(&active) - .await - .unwrap(); - let mut writer = writable.open_writable(&active).unwrap(); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO values_ VALUES(2)")) - .unwrap(); - let cut = writer.capture_deferred().unwrap(); - assert!(cut.position.txid > root.position.txid); - println!("SPARSE-CUT {}", cut.position.txid); - std::io::stdout().flush().unwrap(); - let mut byte = [0]; - std::io::stdin().read_exact(&mut byte).unwrap(); - panic!("parent must kill the sparse writer"); -} - -#[tokio::test(flavor = "multi_thread")] -async fn process_killed_sparse_writer_restores_selected_root_and_commits_again() { - use std::io::BufRead; - use std::process::{Command, Stdio}; - - let store_dir = tempfile::TempDir::new().unwrap(); - let source_dir = tempfile::TempDir::new().unwrap(); - let store = Store::new(Arc::new( - object_store::local::LocalFileSystem::new_with_prefix(store_dir.path()).unwrap(), - )); - let replica = replica(store, [91; 32], [92; 16]); - let mut source = Db::open(&source_dir.path().join("source.sqlite"), Limits::default()).unwrap(); - source - .transaction(|tx| { - tx.execute_batch("CREATE TABLE values_(v INTEGER); INSERT INTO values_ VALUES(1)") - }) - .unwrap(); - let root = replica - .prepare(None, &source.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - source.close().unwrap(); - - let active = source_dir.path().join("active.sqlite"); - let digest: String = root - .digest - .iter() - .map(|byte| format!("{byte:02x}")) - .collect(); - let child = Command::new(std::env::current_exe().unwrap()) - .args([ - "cell::roots::sparse::sparse_crash_writer", - "--exact", - "--nocapture", - ]) - .env("CRAB_LTX_SPARSE_CRASH_STORE", store_dir.path()) - .env("CRAB_LTX_SPARSE_CRASH_ACTIVE", &active) - .env("CRAB_LTX_SPARSE_CRASH_DIGEST", digest) - .env("CRAB_LTX_SPARSE_CRASH_TXID", root.position.txid.to_string()) - .env( - "CRAB_LTX_SPARSE_CRASH_CHECKSUM", - root.position.checksum.to_string(), - ) - .stdin(Stdio::piped()) - .stdout(Stdio::piped()) - .stderr(Stdio::inherit()) - .spawn() - .unwrap(); - let mut child = KillOnDrop(child); - let _stdin = child.0.stdin.take().unwrap(); - let stdout = child.0.stdout.take().unwrap(); - let (send, receive) = std::sync::mpsc::channel(); - std::thread::spawn(move || { - for line in std::io::BufReader::new(stdout) - .lines() - .map_while(Result::ok) - { - if line.contains("SPARSE-CUT") { - let _ = send.send(()); - break; - } - } - }); - receive - .recv_timeout(std::time::Duration::from_secs(15)) - .unwrap(); - child.0.kill().unwrap(); - assert!(!child.0.wait().unwrap().success()); - source_dir.close().unwrap(); - - let recovered_dir = tempfile::TempDir::new().unwrap(); - let recovered = recovered_dir.path().join("recovered.sqlite"); - let writable = replica - .open_root(&root) - .await - .unwrap() - .paged() - .prepare_writable(&recovered) - .await - .unwrap(); - let mut writer = writable.open_writable(&recovered).unwrap(); - let count: i64 = writer - .query_with(|db| db.query_row("SELECT count(*) FROM values_", [], |row| row.get(0))) - .unwrap(); - assert_eq!(count, 1); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO values_ VALUES(3)")) - .unwrap(); - let next = writer.capture_deferred().unwrap(); - assert_eq!(next.position.txid, root.position.txid + 1); - let published = replica.prepare(Some(&root), &next, 2, 1).await.unwrap(); - assert_eq!(published.root().position, next.position); - writer.close().unwrap(); -} - -struct KillOnDrop(std::process::Child); - -impl Drop for KillOnDrop { - fn drop(&mut self) { - let _ = self.0.kill(); - let _ = self.0.wait(); - } -} -#[tokio::test(flavor = "multi_thread")] -async fn sparse_hydration_coalesces_contiguous_cell_frames() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("cell.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(2000000))", - ) - }) - .unwrap(); - let range_reads = Arc::new(AtomicU64::new(0)); - let observed = Arc::clone(&range_reads); - let store = - Store::new(Arc::new(InMemory::new())).with_read_request_observer(Arc::new(move |kind| { - if kind == StorageReadKind::Range { - observed.fetch_add(1, Ordering::SeqCst); - } - })); - let replica = replica(store, [83; 32], [84; 16]); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - - let destination = directory.path().join("sparse.sqlite"); - let writable = replica - .open_root(&root) - .await - .unwrap() - .paged() - .prepare_writable(&destination) - .await - .unwrap(); - range_reads.store(0, Ordering::SeqCst); - let (before, after, reads) = tokio::task::spawn_blocking(move || { - let mut writer = writable.open_writable(&destination).unwrap(); - let before = writer.hydration().unwrap().unwrap(); - let requests = range_reads.load(Ordering::SeqCst); - let after = writer.hydrate_step(320).unwrap(); - let reads = range_reads.load(Ordering::SeqCst) - requests; - writer.close().unwrap(); - (before, after, reads) - }) - .await - .unwrap(); - let hydrated = after.resolved - before.resolved; - assert!(hydrated >= 256); - assert!(reads < u64::from(hydrated)); -} - -#[tokio::test(flavor = "multi_thread")] -async fn immutable_reader_faults_only_needed_pages_and_preserves_provider_errors() { - let directory = tempfile::TempDir::new().unwrap(); - let source = directory.path().join("source.sqlite"); - let mut writer = Db::open(&source, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER); INSERT INTO counter VALUES(7);\ - CREATE TABLE payload(value BLOB); INSERT INTO payload VALUES(randomblob(2000000))", - ) - }) - .unwrap(); - let expected: Vec = writer - .query_with(|connection| { - connection.query_row("SELECT value FROM payload", [], |row| row.get(0)) - }) - .unwrap(); - let batch = writer.capture().unwrap(); - let info = batch.segments[0].info().clone(); - let original = std::fs::read(batch.segments[0].path()).unwrap(); - let backend = Arc::new(InMemory::new()); - let read_bytes = Arc::new(AtomicU64::new(0)); - let observed = Arc::clone(&read_bytes); - let store = Store::new(backend.clone()).with_read_byte_observer(Arc::new(move |bytes| { - observed.fetch_add(bytes, Ordering::SeqCst); - })); - let layout = CellStorageLayout::new(store, Path::from("sparse-reader"), [3; 16]); - let cell = [83; 32]; - let incarnation = [84; 16]; - let replica = CellReplica::new(layout.clone(), cell, incarnation, Limits::default()).unwrap(); - let root = replica.prepare(None, &batch, 1, 1).await.unwrap().root(); - writer.close().unwrap(); - std::fs::remove_file(source).unwrap(); - - // One LTX job slot must remain sufficient: SQLite opening cannot hold the - // same slot that a cold directory-cache lookup needs to complete its fault. - let host = Host::default() - .with_job_slots(Arc::new(tokio::sync::Semaphore::new(1))) - .with_directory_cache(directory.path().join("directory-cache")) - .await - .unwrap(); - let verified = replica.with_host(host).open_root(&root).await.unwrap(); - read_bytes.store(0, Ordering::SeqCst); - let destination = directory.path().join("reader.sqlite"); - let view = verified.open_read_only(&destination).unwrap(); - { - let connection = view.connection().unwrap(); - let value: i64 = connection - .query_row("SELECT value FROM counter", [], |row| row.get(0)) - .unwrap(); - assert_eq!(value, 7); - assert!(read_bytes.load(Ordering::SeqCst) < 512_000); - assert_eq!(std::fs::metadata(&destination).unwrap().len(), 0); - - let expired = std::time::Instant::now() - std::time::Duration::from_secs(1); - assert!( - crab_ltx::with_paged_io_deadline(expired, || { - connection.query_row("SELECT value FROM payload", [], |row| { - row.get::<_, Vec>(0) - }) - }) - .is_err() - ); - assert!(matches!( - view.take_io_error(), - Some(crab_ltx::CrabError::Deadline) - )); - } - - let object = - layout.incarnation_object_path(&cell, &incarnation, &info.blake3, CellObjectKind::Ltx); - backend.delete(&object).await.unwrap(); - { - let connection = view.connection().unwrap(); - assert!( - connection - .query_row("SELECT value FROM payload", [], |row| row - .get::<_, Vec>(0)) - .is_err() - ); - assert!(matches!( - view.take_io_error(), - Some(crab_ltx::CrabError::Storage(_)) - )); - } - let damaged: Vec<_> = original.iter().map(|byte| byte ^ 1).collect(); - backend - .put(&object, Bytes::from(damaged).into()) - .await - .unwrap(); - { - let connection = view.connection().unwrap(); - assert!( - connection - .query_row("SELECT value FROM payload", [], |row| row - .get::<_, Vec>(0)) - .is_err() - ); - assert!(matches!( - view.take_io_error(), - Some(crab_ltx::CrabError::ChecksumMismatch) - )); - } - backend - .put(&object, Bytes::from(original).into()) - .await - .unwrap(); - let actual: Vec = view - .connection() - .unwrap() - .query_row("SELECT value FROM payload", [], |row| row.get(0)) - .unwrap(); - assert_eq!(actual, expected); - assert_eq!(std::fs::metadata(&destination).unwrap().len(), 0); - drop(view); - assert!(!destination.exists()); -} - -async fn hydration_writer(store: Store) -> (tempfile::TempDir, Db, CellReplica, RootRef) { - let directory = tempfile::TempDir::new().unwrap(); - let source_path = directory.path().join("source.sqlite"); - let initial = crab_ltx::rusqlite::Connection::open(&source_path).unwrap(); - initial - .execute_batch("PRAGMA auto_vacuum=FULL; VACUUM;") - .unwrap(); - drop(initial); - let mut source = Db::open(&source_path, Limits::default()).unwrap(); - source.transaction(|tx| tx.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL); INSERT INTO payload VALUES(randomblob(2000000))", - )).unwrap(); - let replica = replica(store, [91; 32], [92; 16]); - let root = replica - .prepare(None, &source.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - source.close().unwrap(); - let destination = directory.path().join("active.sqlite"); - let writable = replica - .open_root(&root) - .await - .unwrap() - .paged() - .prepare_writable(&destination) - .await - .unwrap(); - let writer = tokio::task::spawn_blocking(move || writable.open_writable(&destination)) - .await - .unwrap() - .unwrap(); - (directory, writer, replica, root) -} - -#[tokio::test(flavor = "multi_thread")] -async fn asynchronous_hydration_reuses_pages_prefetched_by_sqlite() { - let ranges = Arc::new(AtomicU64::new(0)); - let observed = ranges.clone(); - let store = - Store::new(Arc::new(InMemory::new())).with_read_request_observer(Arc::new(move |kind| { - if kind == StorageReadKind::Range { - observed.fetch_add(1, Ordering::SeqCst); - } - })); - let (_directory, mut writer, _, _) = hydration_writer(store).await; - let before = writer.hydration().unwrap().unwrap(); - let opening_reads = ranges.swap(0, Ordering::SeqCst); - assert!( - opening_reads > 0, - "SQLite opening must have prefetched inherited pages" - ); - let batch = writer - .prepare_hydration(8) - .unwrap() - .unwrap() - .fetch() - .await - .unwrap(); - let after = writer.install_hydration(batch).unwrap(); - writer.close().unwrap(); - assert_eq!(after.resolved - before.resolved, 8); - assert_eq!( - ranges.load(Ordering::SeqCst), - 0, - "hydration fetched pages already in the demand cache" - ); -} - -#[tokio::test(flavor = "multi_thread")] -async fn asynchronous_hydration_does_not_advance_until_installation() { - let (_directory, mut writer, _, _) = - hydration_writer(Store::new(Arc::new(InMemory::new()))).await; - let before = writer.hydration().unwrap().unwrap(); - let read = writer.prepare_hydration(64).unwrap().unwrap(); - assert!(read.retained_bytes() <= 64 * 65_536); - drop(read.fetch().await.unwrap()); - assert_eq!(writer.hydration().unwrap().unwrap(), before); - let batch = writer - .prepare_hydration(64) - .unwrap() - .unwrap() - .fetch() - .await - .unwrap(); - let after = writer.install_hydration(batch).unwrap(); - assert_eq!(after.resolved - before.resolved, 64); - while !writer.hydration().unwrap().unwrap().complete() { - let batch = writer - .prepare_hydration(64) - .unwrap() - .unwrap() - .fetch() - .await - .unwrap(); - writer.install_hydration(batch).unwrap(); - } - writer - .query_with(|connection| -> crab_ltx::rusqlite::Result<()> { - let integrity: String = - connection.query_row("PRAGMA integrity_check", [], |row| row.get(0))?; - assert_eq!(integrity, "ok"); - Ok(()) - }) - .unwrap(); - writer.close().unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn asynchronous_hydration_preserves_pages_superseded_by_checkpoint() { - let (directory, mut writer, replica, root) = - hydration_writer(Store::new(Arc::new(InMemory::new()))).await; - let batch = writer - .prepare_hydration(64) - .unwrap() - .unwrap() - .fetch() - .await - .unwrap(); - writer - .transaction(|tx| tx.execute_batch("UPDATE payload SET value = zeroblob(2000000)")) - .unwrap(); - let cut = writer - .checkpoint(crab_ltx::CheckpointMode::Truncate) - .unwrap(); - writer.install_hydration(batch).unwrap(); - let next = replica - .prepare(Some(&root), &cut, 2, 1) - .await - .unwrap() - .root(); - writer - .query_with(|connection| -> crab_ltx::rusqlite::Result<()> { - let value: Vec = - connection.query_row("SELECT value FROM payload", [], |row| row.get(0))?; - assert_eq!(value, vec![0; 2_000_000]); - Ok(()) - }) - .unwrap(); - writer.close().unwrap(); - let restored = directory.path().join("restored.sqlite"); - replica - .open_root(&next) - .await - .unwrap() - .restore(&restored) - .await - .unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - let value: Vec = connection - .query_row("SELECT value FROM payload", [], |row| row.get(0)) - .unwrap(); - assert_eq!(value, vec![0; 2_000_000]); -} - -#[tokio::test(flavor = "multi_thread")] -async fn asynchronous_hydration_rejects_a_different_activation() { - let (_directory, writer, _, _) = hydration_writer(Store::new(Arc::new(InMemory::new()))).await; - let (_other_directory, mut other, _, _) = - hydration_writer(Store::new(Arc::new(InMemory::new()))).await; - let batch = writer - .prepare_hydration(64) - .unwrap() - .unwrap() - .fetch() - .await - .unwrap(); - let before = std::fs::read(other.path()).unwrap(); - assert!(matches!( - other.install_hydration(batch), - Err(crab_ltx::CrabError::InvalidState(_)) - )); - assert_eq!(std::fs::read(other.path()).unwrap(), before); - writer.close().unwrap(); - other.close().unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn asynchronous_hydration_never_resurrects_truncated_pages_after_regrowth() { - let (_directory, mut writer, replica, root) = - hydration_writer(Store::new(Arc::new(InMemory::new()))).await; - let batch = writer - .prepare_hydration(64) - .unwrap() - .unwrap() - .fetch() - .await - .unwrap(); - let original_bytes = std::fs::metadata(writer.path()).unwrap().len(); - writer - .transaction(|tx| tx.execute_batch("DELETE FROM payload")) - .unwrap(); - let cut = writer - .checkpoint(crab_ltx::CheckpointMode::Truncate) - .unwrap(); - assert!(std::fs::metadata(writer.path()).unwrap().len() < original_bytes); - let smaller = replica - .prepare(Some(&root), &cut, 2, 1) - .await - .unwrap() - .root(); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO payload VALUES(zeroblob(2000000))")) - .unwrap(); - let cut = writer - .checkpoint(crab_ltx::CheckpointMode::Truncate) - .unwrap(); - writer.install_hydration(batch).unwrap(); - replica.prepare(Some(&smaller), &cut, 3, 1).await.unwrap(); - writer - .query_with(|connection| -> crab_ltx::rusqlite::Result<()> { - let value: Vec = - connection.query_row("SELECT value FROM payload", [], |row| row.get(0))?; - assert_eq!(value, vec![0; 2_000_000]); - Ok(()) - }) - .unwrap(); - writer.close().unwrap(); -} - -#[tokio::test(flavor = "multi_thread")] -async fn hydrated_root_resumes_without_an_intervening_application_write() { - let (directory, mut writer, replica, root) = - hydration_writer(Store::new(Arc::new(InMemory::new()))).await; - while !writer.hydration().unwrap().unwrap().complete() { - let batch = writer - .prepare_hydration(64) - .unwrap() - .unwrap() - .fetch() - .await - .unwrap(); - writer.install_hydration(batch).unwrap(); - } - for cycle in 0..2 { - writer.persist_continuation().unwrap(); - let source = writer.path().to_owned(); - writer.close().unwrap(); - let destination = directory.path().join(format!("resumed-{cycle}.sqlite")); - writer = replica.open_resumed(&source, &destination).unwrap(); - assert_eq!(writer.position(), root.position); - } - writer - .transaction(|tx| tx.execute("INSERT INTO payload VALUES (?1)", [b"resumed".as_slice()])) - .unwrap(); - let cut = writer.capture().unwrap(); - assert_eq!(cut.segments[0].info().pre_checksum, root.position.checksum); - assert_eq!(cut.segments[0].info().min_txid, root.position.txid + 1); - let next = replica.prepare(Some(&root), &cut, 2, 1).await.unwrap(); - writer.close().unwrap(); - let restored = directory.path().join("restored.sqlite"); - replica - .open_root(&next.root()) - .await - .unwrap() - .restore(&restored) - .await - .unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(&restored).unwrap(); - let lengths: Vec = connection - .prepare("SELECT length(value) FROM payload ORDER BY rowid") - .unwrap() - .query_map([], |row| row.get(0)) - .unwrap() - .collect::>() - .unwrap(); - assert_eq!(lengths, [2_000_000, 7]); -} diff --git a/crates/crab-ltx/tests/host.rs b/crates/crab-ltx/tests/host.rs deleted file mode 100644 index b3ae06307..000000000 --- a/crates/crab-ltx/tests/host.rs +++ /dev/null @@ -1,8 +0,0 @@ -//! Host hook, executor, and worker admission tests. - -#![cfg(feature = "replica")] - -mod host { - pub mod admissions; - pub mod hooks; -} diff --git a/crates/crab-ltx/tests/host/admissions.rs b/crates/crab-ltx/tests/host/admissions.rs deleted file mode 100644 index 7eac23180..000000000 --- a/crates/crab-ltx/tests/host/admissions.rs +++ /dev/null @@ -1,73 +0,0 @@ -//! Disk budget admission tests. - -use std::sync::{ - Arc, - atomic::{AtomicBool, AtomicUsize, Ordering}, -}; - -use crab_ltx::{CrabError, DiskBudget, DiskBudgetAdmission}; - -/// A hook whose install-time reconcile succeeds and whose later reconciles fail. -/// -/// `dies_on_failure` models the owner dropping between the budget's liveness -/// prune and the reconcile call itself; without it the owner stays live and its -/// rejection must keep binding the budget. -struct ScriptedAdmission { - reconciles: AtomicUsize, - dies_on_failure: bool, - live: AtomicBool, -} - -impl ScriptedAdmission { - fn new(dies_on_failure: bool) -> Self { - Self { - reconciles: AtomicUsize::new(0), - dies_on_failure, - live: AtomicBool::new(true), - } - } -} - -impl DiskBudgetAdmission for ScriptedAdmission { - fn reconcile(&self, _bytes: u64) -> crab_ltx::Result<()> { - if self.reconciles.fetch_add(1, Ordering::AcqRel) == 0 { - return Ok(()); - } - if self.dies_on_failure { - self.live.store(false, Ordering::Release); - return Err(CrabError::InvalidState("runtime ledger closed")); - } - Err(CrabError::InvalidState("live ledger rejects bytes")) - } - - fn is_live(&self) -> bool { - self.live.load(Ordering::Acquire) - } -} - -#[test] -fn hook_that_dies_during_reconcile_does_not_fail_a_live_reservation() { - let budget = DiskBudget::new(1 << 20); - budget - .install_admission(Arc::new(ScriptedAdmission::new(true))) - .expect("a live hook installs"); - - let reservation = budget - .try_reserve(64) - .expect("a hook whose owner is gone stops constraining the budget"); - assert_eq!(budget.used(), 64); - drop(reservation); - assert_eq!(budget.used(), 0); - assert!(budget.try_reserve(64).is_ok()); -} - -#[test] -fn live_hook_rejection_still_fails_the_reservation() { - let budget = DiskBudget::new(1 << 20); - budget - .install_admission(Arc::new(ScriptedAdmission::new(false))) - .expect("a live hook installs"); - - assert!(budget.try_reserve(64).is_err()); - assert_eq!(budget.used(), 0); -} diff --git a/crates/crab-ltx/tests/host/hooks.rs b/crates/crab-ltx/tests/host/hooks.rs deleted file mode 100644 index fe0f5f838..000000000 --- a/crates/crab-ltx/tests/host/hooks.rs +++ /dev/null @@ -1,331 +0,0 @@ -#[cfg(feature = "replica")] -use crab_ltx::environment::{Executor, Worker}; -#[cfg(feature = "replica")] -use crab_ltx::{CellReplica, CellStorageLayout, LocalSegment}; -use crab_ltx::{ - CheckpointMode, CrabError, Db, Host, Limits, - environment::{DirectFileSystem, FileIo, FileSystem}, -}; -#[cfg(feature = "replica")] -use crab_storage::Store; -#[cfg(feature = "replica")] -use object_store::throttle::{ThrottleConfig, ThrottledStore}; -#[cfg(feature = "replica")] -use object_store::{memory::InMemory, path::Path as ObjectPath}; -#[cfg(feature = "replica")] -use std::time::Duration; -use std::{ - collections::BTreeSet, - io, - path::{Path, PathBuf}, - sync::{ - Arc, Mutex, - atomic::{AtomicBool, AtomicUsize, Ordering}, - }, -}; - -#[cfg(feature = "replica")] -use std::sync::OnceLock; - -#[cfg(feature = "replica")] -struct DelayedExecutor { - delay: Duration, -} - -#[cfg(feature = "replica")] -struct TestWorker(std::thread::JoinHandle<()>); - -#[cfg(feature = "replica")] -impl Worker for TestWorker { - fn join(self: Box) -> io::Result<()> { - self.0 - .join() - .map_err(|_| io::Error::other("test worker panicked")) - } -} - -#[cfg(feature = "replica")] -impl Executor for DelayedExecutor { - fn dispatch(&self, job: Box) -> io::Result<()> { - let delay = self.delay; - tokio::spawn(async move { - tokio::time::sleep(delay).await; - let _ = tokio::task::spawn_blocking(job).await; - }); - Ok(()) - } - - fn start_worker(&self, job: Box) -> io::Result> { - Ok(Box::new(TestWorker(std::thread::spawn(job)))) - } -} - -struct Pause { - operation: &'static str, - entered: tokio::sync::Notify, - released: Mutex, - wake: std::sync::Condvar, -} - -impl Pause { - fn wait(&self, operation: &str) { - if self.operation == operation { - let mut released = self.released.lock().unwrap(); - self.entered.notify_one(); - while !*released { - released = self.wake.wait(released).unwrap(); - } - } - } -} - -struct Release(Arc); - -impl Drop for Release { - fn drop(&mut self) { - *self.0.released.lock().unwrap() = true; - self.0.wake.notify_all(); - } -} - -#[derive(Clone, Default)] -struct Faults { - failure: Arc>>, - planned: Arc>>, - calls: Arc>>, - largest_read: Arc, - read_calls: Arc, - checksum_reads: Arc, - checksum_read_bytes: Arc, - largest_checksum_read: Arc, - largest_write: Arc, - write_calls: Arc, - file_syncs: Arc, - parent_syncs: Arc, - track_all: Arc, - #[cfg(feature = "replica")] - create_pause: Arc>>, - forbidden_thread: Arc>>, - pause: Arc>>>, -} - -#[cfg(feature = "replica")] -struct InstallPause { - entered: std::sync::Barrier, - release: std::sync::Barrier, -} - -#[cfg(feature = "replica")] -impl InstallPause { - fn new() -> Self { - Self { - entered: std::sync::Barrier::new(2), - release: std::sync::Barrier::new(2), - } - } -} - -impl Faults { - fn arm(&self, operation: Option<&'static str>) { - *self.failure.lock().unwrap() = operation; - } - - /// Arms an ordered list of operations to fail, one injection per match. - /// - /// A plan models a sequence of failures across seams — a torn write, then a - /// failed rename — instead of one armed operation at a time. - fn plan(&self, operations: impl IntoIterator) { - *self.planned.lock().unwrap() = operations.into_iter().collect(); - } - - fn check(&self, operation: &'static str) -> io::Result<()> { - self.calls.lock().unwrap().insert(operation); - if *self.forbidden_thread.lock().unwrap() == Some(std::thread::current().id()) { - return Err(io::Error::other("filesystem work on async thread")); - } - let pause = self.pause.lock().unwrap().clone(); - if let Some(pause) = pause { - pause.wait(operation); - } - if *self.failure.lock().unwrap() == Some(operation) { - return Err(io::Error::new(io::ErrorKind::StorageFull, operation)); - } - let mut planned = self.planned.lock().unwrap(); - if planned.first() == Some(&operation) { - planned.remove(0); - return Err(io::Error::new(io::ErrorKind::StorageFull, operation)); - } - Ok(()) - } -} - -struct File { - inner: Box, - faults: Faults, - track: bool, - checksum: bool, -} - -impl FileIo for File { - fn write_all(&mut self, bytes: &[u8]) -> io::Result<()> { - if self.track { - self.faults - .largest_write - .fetch_max(bytes.len(), Ordering::Relaxed); - self.faults.write_calls.fetch_add(1, Ordering::Relaxed); - } - if let Err(error) = self.faults.check("write_all") { - self.inner.write_all(&bytes[..bytes.len() / 2])?; - return Err(error); - } - self.inner.write_all(bytes) - } - fn write_all_at(&mut self, offset: u64, bytes: &[u8]) -> io::Result<()> { - if self.track { - self.faults - .largest_write - .fetch_max(bytes.len(), Ordering::Relaxed); - self.faults.write_calls.fetch_add(1, Ordering::Relaxed); - } - if let Err(error) = self.faults.check("write_all_at") { - self.inner.write_all_at(offset, &bytes[..bytes.len() / 2])?; - return Err(error); - } - self.inner.write_all_at(offset, bytes) - } - fn read_exact_at(&mut self, offset: u64, len: usize) -> io::Result> { - if self.checksum { - self.faults.checksum_reads.fetch_add(1, Ordering::Relaxed); - self.faults - .checksum_read_bytes - .fetch_add(len, Ordering::Relaxed); - self.faults - .largest_checksum_read - .fetch_max(len, Ordering::Relaxed); - } - if self.track { - self.faults.largest_read.fetch_max(len, Ordering::Relaxed); - self.faults.read_calls.fetch_add(1, Ordering::Relaxed); - } - self.faults.check("read_exact_at")?; - self.inner.read_exact_at(offset, len) - } - fn sync_all(&mut self) -> io::Result<()> { - self.faults.check("sync_all")?; - self.faults.file_syncs.fetch_add(1, Ordering::Relaxed); - self.inner.sync_all() - } - fn file_len(&self) -> io::Result { - self.inner.file_len() - } - fn set_len(&mut self, len: u64) -> io::Result<()> { - self.faults.check("set_len")?; - self.inner.set_len(len) - } -} - -macro_rules! filesystem_operation { - ($name:ident($($arg:ident: $ty:ty),*) -> $result:ty) => { - fn $name(&self, $($arg: $ty),*) -> io::Result<$result> { - self.check(stringify!($name))?; - DirectFileSystem.$name($($arg),*) - } - }; -} - -impl FileSystem for Faults { - fn open(&self, path: &Path) -> io::Result> { - self.check("open")?; - Ok(Box::new(File { - inner: DirectFileSystem.open(path)?, - faults: self.clone(), - checksum: path.to_string_lossy().ends_with(".crab-ltx-checksums"), - track: self.track_all.load(Ordering::Relaxed) - || path.to_string_lossy().contains(".ltx"), - })) - } - fn open_rw(&self, path: &Path) -> io::Result> { - self.check("open_rw")?; - Ok(Box::new(File { - inner: DirectFileSystem.open_rw(path)?, - faults: self.clone(), - checksum: path.to_string_lossy().ends_with(".crab-ltx-checksums"), - track: self.track_all.load(Ordering::Relaxed) - || path.to_string_lossy().contains(".ltx"), - })) - } - fn create(&self, path: &Path) -> io::Result> { - self.check("create")?; - let inner = DirectFileSystem.create(path)?; - #[cfg(feature = "replica")] - if let Some(pause) = self.create_pause.get() { - pause.entered.wait(); - pause.release.wait(); - } - Ok(Box::new(File { - inner, - faults: self.clone(), - checksum: path.to_string_lossy().ends_with(".crab-ltx-checksums"), - track: self.track_all.load(Ordering::Relaxed) - || path.to_string_lossy().contains(".ltx"), - })) - } - filesystem_operation!(file_len(path: &Path) -> u64); - filesystem_operation!(canonicalize(path: &Path) -> PathBuf); - filesystem_operation!(exists(path: &Path) -> bool); - filesystem_operation!(create_dir(path: &Path) -> ()); - filesystem_operation!(create_dir_all(path: &Path) -> ()); - filesystem_operation!(remove_file(path: &Path) -> ()); - filesystem_operation!(rename(from: &Path, to: &Path) -> ()); - fn rename_uncommitted(&self, from: &Path, to: &Path) -> io::Result<()> { - self.check("rename_uncommitted")?; - DirectFileSystem.rename_uncommitted(from, to) - } - fn sync_parent(&self, path: &Path) -> io::Result<()> { - self.check("sync_parent")?; - self.parent_syncs.fetch_add(1, Ordering::Relaxed); - DirectFileSystem.sync_parent(path) - } - filesystem_operation!(persist_new(path: &Path, bytes: &[u8]) -> ()); - fn persist_file_new(&self, source: &Path, destination: &Path) -> io::Result<()> { - self.check("persist_file_new")?; - DirectFileSystem.persist_file_new(source, destination)?; - Ok(()) - } -} - -fn fixture() -> (tempfile::TempDir, Arc, Host, Db) { - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let host = Host::default() - .with_filesystem(faults.clone()) - .with_local_disk_budget(crab_ltx::DiskBudget::new(1 << 30)); - let mut writer = Db::open_with_host( - &directory.path().join("source.sqlite"), - Limits::default(), - host.clone(), - ) - .unwrap(); - writer - .transaction(|tx| { - tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(randomblob(20000))") - }) - .unwrap(); - (directory, faults, host, writer) -} - -fn injected(result: crab_ltx::Result) { - assert!( - matches!(result, Err(CrabError::Io(error)) if error.kind() == io::ErrorKind::StorageFull) - ); -} - -#[cfg(feature = "replica")] -mod activation; -mod capture; -mod compaction; -mod injection; -mod matrix; -mod prepare; -mod restore; -mod volatile; diff --git a/crates/crab-ltx/tests/host/hooks/activation.rs b/crates/crab-ltx/tests/host/hooks/activation.rs deleted file mode 100644 index 927412c37..000000000 --- a/crates/crab-ltx/tests/host/hooks/activation.rs +++ /dev/null @@ -1,595 +0,0 @@ -use super::*; - -fn seed_cache(root: &Path, entries: usize, entry_bytes: usize) { - std::fs::create_dir(root).unwrap(); - let mut index = std::collections::BTreeMap::new(); - for number in 0..entries { - let key = format!("restart-diagnostic-{number:05}"); - let path = root.join(blake3::hash(key.as_bytes()).to_hex().as_str()); - std::fs::write(path, vec![0_u8; entry_bytes]).unwrap(); - index.insert(key, entry_bytes as u64); - } - std::fs::write( - root.join("index-v1.json"), - serde_json::to_vec(&serde_json::json!({"version": 1, "entries": index})).unwrap(), - ) - .unwrap(); -} - -#[tokio::test] -async fn directory_cache_reopens_off_async_worker() { - let directory = tempfile::tempdir().unwrap(); - let root = directory.path().join("cache"); - seed_cache(&root, 2, 1024); - let faults = Arc::new(Faults::default()); - *faults.forbidden_thread.lock().unwrap() = Some(std::thread::current().id()); - let host = Host::default() - .with_filesystem(faults) - .with_local_disk_budget(crab_ltx::DiskBudget::new(1 << 20)) - .with_directory_cache(root) - .await - .unwrap(); - assert_eq!(host.directory_cache_stats().unwrap().entries(), 2); -} - -#[tokio::test] -async fn directory_cache_rejects_oversized_index_before_reading() { - let directory = tempfile::tempdir().unwrap(); - let root = directory.path().join("cache"); - std::fs::create_dir(&root).unwrap(); - std::fs::File::create(root.join("index-v1.json")) - .unwrap() - .set_len(32 << 20) - .unwrap(); - let faults = Arc::new(Faults::default()); - faults.track_all.store(true, Ordering::Relaxed); - let host = Host::default() - .with_filesystem(faults.clone()) - .with_directory_cache(root) - .await - .unwrap(); - assert_eq!(host.directory_cache_stats().unwrap().entries(), 0); - assert_eq!(faults.read_calls.load(Ordering::Relaxed), 0); -} - -struct PausedCacheAdmission(Arc); - -impl crab_ltx::DiskBudgetAdmission for PausedCacheAdmission { - fn reconcile(&self, bytes: u64) -> crab_ltx::Result<()> { - if bytes != 0 { - self.0.wait("reserve"); - } - Ok(()) - } -} - -#[tokio::test] -async fn canceled_cache_reopen_retains_admission_until_the_dispatched_job_finishes() { - let directory = tempfile::tempdir().unwrap(); - let root = directory.path().join("cache"); - seed_cache(&root, 2, 1); - let budget = crab_ltx::DiskBudget::new(16); - let pause = Arc::new(Pause { - operation: "reserve", - entered: tokio::sync::Notify::new(), - released: Mutex::new(false), - wake: std::sync::Condvar::new(), - }); - let release = Release(pause.clone()); - budget - .install_admission(Arc::new(PausedCacheAdmission(pause.clone()))) - .unwrap(); - let jobs = Arc::new(tokio::sync::Semaphore::new(1)); - let host = Host::default() - .with_local_disk_budget(budget.clone()) - .with_job_slots(jobs.clone()); - let task = tokio::spawn(host.with_directory_cache(root)); - // This current-thread runtime must progress while the blocking constructor - // owns a reservation. Canceling its waiter cannot release that job early. - tokio::time::timeout(Duration::from_secs(5), pause.entered.notified()) - .await - .unwrap(); - task.abort(); - assert!(task.await.err().unwrap().is_cancelled()); - assert_eq!(jobs.available_permits(), 0); - assert_eq!(budget.used(), 1); - drop(release); - tokio::time::timeout(Duration::from_secs(5), async { - while budget.used() != 0 || jobs.available_permits() != 1 { - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); -} - -#[tokio::test] -async fn concurrent_cache_reopens_share_the_remaining_disk_budget() { - let directory = tempfile::tempdir().unwrap(); - let first = directory.path().join("first"); - let second = directory.path().join("second"); - seed_cache(&first, 2, 1); - seed_cache(&second, 2, 1); - let budget = crab_ltx::DiskBudget::new(16); - let occupied = budget.try_reserve(13).unwrap(); - let host = Host::default().with_local_disk_budget(budget.clone()); - let (first, second) = tokio::join!( - host.clone().with_directory_cache(first), - host.with_directory_cache(second), - ); - let first = first.unwrap(); - let second = second.unwrap(); - assert_eq!( - first.directory_cache_stats().unwrap().entries() - + second.directory_cache_stats().unwrap().entries(), - 3 - ); - assert_eq!(budget.used(), 16); - drop((first, second, occupied)); - assert_eq!(budget.used(), 0); -} - -#[tokio::test] -#[ignore = "filesystem diagnostic; records cache reopen cost, not a service SLO"] -async fn directory_cache_restart_diagnostic() { - for entries in [1_024, 4_096, 16_384] { - let directory = tempfile::tempdir().unwrap(); - let root = directory.path().join("cache"); - // Seed membership directly so setup excludes the per-fill index rewrite. - // These are cache metadata fixtures, not authenticated LTX directory nodes. - seed_cache(&root, entries, 1024); - - for repetition in 0..3 { - let budget = crab_ltx::DiskBudget::new(256 << 20); - let host = Host::default().with_local_disk_budget(budget.clone()); - let started = std::time::Instant::now(); - let host = host.with_directory_cache(root.clone()).await.unwrap(); - let elapsed_us = started.elapsed().as_micros(); - let stats = host.directory_cache_stats().unwrap(); - assert_eq!(stats.entries(), entries); - assert_eq!(stats.bytes(), entries as u64 * 1024); - println!( - "{}", - serde_json::json!({ - "diagnostic": "directory_cache_restart", - "entries": entries, - "repetition": repetition, - "elapsed_us": elapsed_us, - "debug_assertions": cfg!(debug_assertions), - }) - ); - drop(host); - assert_eq!(budget.used(), 0); - } - } -} - -async fn prepared_root(host: Host, writer: &mut Db, store: Store) -> crab_ltx::CellPagedDatabase { - let replica = CellReplica::new( - CellStorageLayout::new(store, ObjectPath::from("activation-admission"), [81; 16]), - [82; 32], - [83; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let captured = writer.capture().unwrap(); - let root = replica.prepare(None, &captured, 1, 1).await.unwrap().root(); - replica.open_root(&root).await.unwrap().paged() -} - -fn checksum_path(database: &Path) -> PathBuf { - let mut path = database.as_os_str().to_owned(); - path.push(".crab-ltx-checksums"); - PathBuf::from(path) -} - -#[tokio::test] -async fn writable_activation_dispatches_filesystem_work_with_one_job_slot() { - let (directory, faults, host, mut writer) = fixture(); - let jobs = Arc::new(tokio::sync::Semaphore::new(1)); - let dirty = Arc::new(tokio::sync::Semaphore::new(1)); - let host = host - .with_job_slots(jobs.clone()) - .with_dirty_slots(dirty.clone()) - .with_directory_cache(directory.path().join("cache")) - .await - .unwrap(); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(zeroblob(40000000))")) - .unwrap(); - let paged = prepared_root(host, &mut writer, Store::new(Arc::new(InMemory::new()))).await; - faults.track_all.store(true, Ordering::Relaxed); - faults.largest_write.store(0, Ordering::Relaxed); - *faults.forbidden_thread.lock().unwrap() = Some(std::thread::current().id()); - - let prepared = tokio::time::timeout( - Duration::from_secs(5), - paged.prepare_writable(&directory.path().join("active.sqlite")), - ) - .await - .unwrap(); - - *faults.forbidden_thread.lock().unwrap() = None; - assert!(prepared.is_ok(), "{:?}", prepared.err()); - assert_eq!(jobs.available_permits(), 1); - assert_eq!(dirty.available_permits(), 1); - assert_eq!(faults.largest_write.load(Ordering::Relaxed), 64 << 10); - writer.close().unwrap(); -} - -#[tokio::test] -async fn canceled_activation_retains_admission_until_file_cleanup_finishes() { - for operation in ["create", "write_all", "file_len"] { - let (directory, faults, host, mut writer) = fixture(); - let jobs = Arc::new(tokio::sync::Semaphore::new(1)); - let dirty = Arc::new(tokio::sync::Semaphore::new(1)); - let host = host - .with_job_slots(jobs.clone()) - .with_dirty_slots(dirty.clone()); - let paged = prepared_root(host, &mut writer, Store::new(Arc::new(InMemory::new()))).await; - let destination = directory.path().join("active.sqlite"); - let task_destination = destination.clone(); - let task_paged = paged.clone(); - let pause = Arc::new(Pause { - operation, - entered: tokio::sync::Notify::new(), - released: Mutex::new(false), - wake: std::sync::Condvar::new(), - }); - let release = Release(pause.clone()); - *faults.pause.lock().unwrap() = Some(pause.clone()); - *faults.forbidden_thread.lock().unwrap() = Some(std::thread::current().id()); - let task = - tokio::spawn(async move { task_paged.prepare_writable(&task_destination).await }); - tokio::time::timeout(Duration::from_secs(5), pause.entered.notified()) - .await - .unwrap(); - // This task runs on the same single-thread runtime as activation. Reaching - // here while the filesystem is paused proves unrelated async progress. - task.abort(); - assert!(task.await.err().unwrap().is_cancelled()); - assert_eq!(dirty.available_permits(), 0, "{operation}"); - assert_eq!(jobs.available_permits(), 0, "{operation}"); - let cleanup_pause = Arc::new(Pause { - operation: "remove_file", - entered: tokio::sync::Notify::new(), - released: Mutex::new(false), - wake: std::sync::Condvar::new(), - }); - let cleanup_release = Release(cleanup_pause.clone()); - *faults.pause.lock().unwrap() = Some(cleanup_pause.clone()); - drop(release); - tokio::time::timeout(Duration::from_secs(5), cleanup_pause.entered.notified()) - .await - .unwrap(); - assert_eq!(dirty.available_permits(), 0, "cleanup after {operation}"); - assert_eq!(jobs.available_permits(), 0, "cleanup after {operation}"); - drop(cleanup_release); - let permit = tokio::time::timeout(Duration::from_secs(5), dirty.acquire()) - .await - .unwrap() - .unwrap(); - drop(permit); - assert!(!checksum_path(&destination).exists(), "{operation}"); - assert_eq!(jobs.available_permits(), 1, "{operation}"); - *faults.pause.lock().unwrap() = None; - *faults.forbidden_thread.lock().unwrap() = None; - let prepared = paged.prepare_writable(&destination).await.unwrap(); - let mut restored = prepared.open_writable(&destination).unwrap(); - let count: u64 = restored - .query_with(|db| db.query_row("SELECT COUNT(*) FROM t", [], |row| row.get(0))) - .unwrap(); - assert_eq!(count, 1, "{operation}"); - restored.close().unwrap(); - writer.close().unwrap(); - } -} - -#[tokio::test] -async fn activation_file_failures_cleanup_before_retry_and_preserve_existing_destinations() { - let (directory, faults, host, mut writer) = fixture(); - let paged = prepared_root(host, &mut writer, Store::new(Arc::new(InMemory::new()))).await; - let destination = directory.path().join("active.sqlite"); - for operation in ["create", "write_all", "file_len"] { - faults.plan([operation]); - injected(paged.clone().prepare_writable(&destination).await); - assert!(!checksum_path(&destination).exists(), "{operation}"); - } - for existing in [&destination, &checksum_path(&destination)] { - std::fs::write(existing, b"existing").unwrap(); - assert!(matches!( - paged.clone().prepare_writable(&destination).await, - Err(CrabError::InvalidState(_)) - )); - assert_eq!(std::fs::read(existing).unwrap(), b"existing"); - std::fs::remove_file(existing).unwrap(); - } - paged.prepare_writable(&destination).await.unwrap(); - writer.close().unwrap(); -} - -#[tokio::test(start_paused = true)] -async fn cold_activation_overlaps_leaf_reads_within_shared_io_admission() { - let (directory, _faults, _host, mut writer) = fixture(); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(zeroblob(10000000))")) - .unwrap(); - let backend = InMemory::new(); - let layout = - |store| CellStorageLayout::new(store, ObjectPath::from("activation-leaf-reads"), [84; 16]); - let replica = CellReplica::new( - layout(Store::new(Arc::new(backend.clone()))), - [85; 32], - [86; 16], - Limits::default(), - ) - .unwrap(); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - let delay = Duration::from_millis(100); - for slots in [1, 4, 16] { - let io = Arc::new(tokio::sync::Semaphore::new(slots)); - let replica = CellReplica::new( - layout(Store::new(Arc::new(ThrottledStore::new( - backend.clone(), - ThrottleConfig { - wait_get_per_call: delay, - ..ThrottleConfig::default() - }, - )))), - [85; 32], - [86; 16], - Limits::default(), - ) - .unwrap() - .with_host( - Host::default() - .with_io_slots(io.clone()) - .with_job_slots(Arc::new(tokio::sync::Semaphore::new(1))), - ); - let verified = replica.open_root(&root).await.unwrap(); - assert_eq!(verified.directory_height(), 1); - let paged = verified.paged(); - let leaves = paged.page_count().div_ceil(256); - let destination = directory.path().join(format!("parallel-{slots}.sqlite")); - let started = tokio::time::Instant::now(); - let prepared = paged.prepare_writable(&destination).await.unwrap(); - let elapsed = started.elapsed(); - // open_root already loaded the authenticated parent. Four shared slots - // overlap waits, while excess slots cannot exceed the eight-read ceiling. - assert_eq!( - elapsed, - delay * leaves.div_ceil(slots.min(8) as u32), - "slots={slots}, leaves={leaves}" - ); - assert_eq!(io.available_permits(), slots); - let mut restored = prepared.open_writable(&destination).unwrap(); - let count: u64 = restored - .query_with(|db| db.query_row("SELECT COUNT(*) FROM t", [], |row| row.get(0))) - .unwrap(); - assert_eq!(count, 2); - restored.close().unwrap(); - } - writer.close().unwrap(); -} - -struct ActivationExecutor(Arc); - -impl Executor for ActivationExecutor { - fn dispatch(&self, job: Box) -> io::Result<()> { - tokio::task::spawn_blocking(job); - Ok(()) - } - - fn start_worker(&self, job: Box) -> io::Result> { - self.0.check("start_worker")?; - Ok(Box::new(TestWorker(std::thread::spawn(job)))) - } -} - -async fn prepared_activation( - store: Store, - value: i64, -) -> ( - tempfile::TempDir, - Arc, - crab_ltx::CellWritableDatabase, - PathBuf, -) { - let (directory, faults, host, mut writer) = fixture(); - writer - .transaction(|tx| tx.execute("UPDATE t SET v = ?1", [value])) - .unwrap(); - let host = host.with_executor(Arc::new(ActivationExecutor(faults.clone()))); - let paged = prepared_root(host, &mut writer, store).await; - writer.close().unwrap(); - let destination = directory.path().join("active.sqlite"); - let prepared = paged.prepare_writable(&destination).await.unwrap(); - (directory, faults, prepared, destination) -} - -#[tokio::test(flavor = "multi_thread")] -async fn sparse_activation_io_does_not_block_other_cells() { - verify_registry_isolation(Store::new(Arc::new(InMemory::new()))).await; -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires the RustFS environment documented in examples/README.md"] -async fn rustfs_sparse_activation_io_does_not_block_other_cells() { - let endpoint = std::env::var("CRAB_LTX_TEST_ENDPOINT").unwrap(); - let store = crab_storage::build_explicit_store( - &std::env::var("CRAB_LTX_TEST_BUCKET").unwrap(), - crab_storage::ObjectStoreCredentials::Aws { - access_key_id: std::env::var("AWS_ACCESS_KEY_ID").unwrap(), - secret_access_key: std::env::var("AWS_SECRET_ACCESS_KEY").unwrap(), - session_token: None, - region: "us-east-1".into(), - }, - Some(&endpoint), - endpoint.starts_with("http://"), - ) - .unwrap(); - let run = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_nanos(); - let prefix = format!( - "crab-ltx-tests/activation-registry/{run}-{}", - std::process::id() - ); - let store = Store::new(Arc::new(object_store::prefix::PrefixStore::new( - store.inner().clone(), - ObjectPath::from(prefix.clone()), - ))); - verify_registry_isolation(store).await; - eprintln!("RustFS activation registry isolation passed: {prefix}"); -} - -async fn verify_registry_isolation(store: Store) { - for operation in ["create", "set_len", "start_worker"] { - let (_slow_directory, faults, slow, slow_path) = - prepared_activation(store.clone(), 1).await; - let (_opening_directory, _, opening, opening_path) = - prepared_activation(store.clone(), 2).await; - let (_closing_directory, _, closing, closing_path) = - prepared_activation(store.clone(), 3).await; - let closing = closing.open_writable(&closing_path).unwrap(); - let pause = Arc::new(Pause { - operation, - entered: tokio::sync::Notify::new(), - released: Mutex::new(false), - wake: std::sync::Condvar::new(), - }); - let release = Release(pause.clone()); - *faults.pause.lock().unwrap() = Some(pause.clone()); - let slow = tokio::task::spawn_blocking(move || { - read_and_close(slow.open_writable(&slow_path).unwrap()) - }); - tokio::time::timeout(Duration::from_secs(5), pause.entered.notified()) - .await - .unwrap(); - let mut opening = tokio::task::spawn_blocking(move || { - read_and_close(opening.open_writable(&opening_path).unwrap()) - }); - let mut closing = tokio::task::spawn_blocking(move || read_and_close(closing)); - let (opened, closed) = tokio::join!( - tokio::time::timeout(Duration::from_secs(1), &mut opening), - tokio::time::timeout(Duration::from_secs(1), &mut closing), - ); - let progressed = (opened.is_ok(), closed.is_ok()); - // Always drain blocked jobs before asserting the isolation property. - // The regression must fail cleanly against the original global lock. - drop(release); - *faults.pause.lock().unwrap() = None; - assert_eq!(slow.await.unwrap(), 1); - let count = match opened { - Ok(result) => result.unwrap(), - Err(_) => opening.await.unwrap(), - }; - let closed_value = match closed { - Ok(result) => result.unwrap(), - Err(_) => closing.await.unwrap(), - }; - assert_eq!((count, closed_value), (2, 3)); - assert_eq!(progressed, (true, true), "paused {operation}"); - } -} - -#[tokio::test(flavor = "multi_thread")] -async fn concurrent_sparse_open_preserves_the_in_progress_path_claim() { - let (_directory, faults, prepared, destination) = - prepared_activation(Store::new(Arc::new(InMemory::new())), 1).await; - let pause = Arc::new(Pause { - operation: "create", - entered: tokio::sync::Notify::new(), - released: Mutex::new(false), - wake: std::sync::Condvar::new(), - }); - let release = Release(pause.clone()); - *faults.pause.lock().unwrap() = Some(pause.clone()); - let first = { - let prepared = prepared.clone(); - let destination = destination.clone(); - tokio::task::spawn_blocking(move || prepared.open_writable(&destination)) - }; - tokio::time::timeout(Duration::from_secs(5), pause.entered.notified()) - .await - .unwrap(); - assert!(!destination.exists()); - // Rejected attempts must neither enter file creation nor remove the first - // attempt's claim. Exercise another attempt after the first refusal. - let mut conflicts = { - let prepared = prepared.clone(); - let destination = destination.clone(); - tokio::task::spawn_blocking(move || { - for _ in 0..2 { - assert!(matches!( - prepared.clone().open_writable(&destination), - Err(CrabError::InvalidState(_)) - )); - } - }) - }; - let refused = tokio::time::timeout(Duration::from_secs(1), &mut conflicts).await; - let prompt = refused.is_ok(); - drop(release); - *faults.pause.lock().unwrap() = None; - let db = first.await.unwrap().unwrap(); - match refused { - Ok(result) => result.unwrap(), - Err(_) => conflicts.await.unwrap(), - } - assert_eq!((prompt, read_and_close(db)), (true, 1)); -} - -#[tokio::test(flavor = "multi_thread")] -async fn failed_sparse_open_releases_its_claim_without_deleting_files() { - for operation in [ - "create", - "set_len", - "start_worker", - "create_dir", - "existing", - ] { - let (_directory, faults, prepared, destination) = - prepared_activation(Store::new(Arc::new(InMemory::new())), 1).await; - if operation == "existing" { - std::fs::write(&destination, b"existing").unwrap(); - assert!(matches!( - prepared.clone().open_writable(&destination), - Err(CrabError::Io(error)) if error.kind() == io::ErrorKind::AlreadyExists - )); - assert_eq!(std::fs::read(&destination).unwrap(), b"existing"); - } else { - faults.plan([operation]); - injected(prepared.clone().open_writable(&destination)); - } - if operation != "create" { - assert!( - destination.exists(), - "failed {operation} must leave its file quarantined" - ); - assert!(matches!( - prepared.clone().open_writable(&destination), - Err(CrabError::Io(error)) if error.kind() == io::ErrorKind::AlreadyExists - )); - // Only the test's owner discards this failed local artifact. The - // library must not turn an interrupted sparse file into a fresh one. - std::fs::remove_file(&destination).unwrap(); - } - let db = prepared.open_writable(&destination).unwrap(); - assert_eq!(read_and_close(db), 1, "retry after {operation}"); - } -} - -fn read_and_close(mut db: Db) -> i64 { - let value = db - .query_with(|db| db.query_row("SELECT v FROM t", [], |row| row.get(0))) - .unwrap(); - db.close().unwrap(); - value -} diff --git a/crates/crab-ltx/tests/host/hooks/capture.rs b/crates/crab-ltx/tests/host/hooks/capture.rs deleted file mode 100644 index f5ce35c08..000000000 --- a/crates/crab-ltx/tests/host/hooks/capture.rs +++ /dev/null @@ -1,396 +0,0 @@ -//! Capture transfer bounds, deferred barriers, fencing, and pruning. - -use super::*; - -#[test] -fn capture_and_inspection_bound_each_filesystem_transfer() { - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let host = Host::default().with_filesystem(faults.clone()); - let mut writer = Db::open_with_host( - &directory.path().join("streamed.sqlite"), - Limits::default(), - host, - ) - .unwrap(); - writer - .transaction(|tx| { - tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(randomblob(2000000))") - }) - .unwrap(); - - let batch = writer.capture().unwrap(); - - assert!(batch.segments[0].info().size_bytes > 1_000_000); - assert!(faults.largest_write.load(Ordering::Relaxed) < 128 * 1024); - assert!(faults.largest_read.load(Ordering::Relaxed) < 128 * 1024); - - faults.largest_read.store(0, Ordering::Relaxed); - faults.read_calls.store(0, Ordering::Relaxed); - faults.largest_write.store(0, Ordering::Relaxed); - let (snapshot, _) = writer - .snapshot(&directory.path().join("streamed-snapshot.ltx")) - .unwrap(); - assert!(snapshot.info().size_bytes > 1_000_000); - assert!(faults.largest_write.load(Ordering::Relaxed) < 128 * 1024); - assert!(faults.largest_read.load(Ordering::Relaxed) < 128 * 1024); -} -#[test] -fn deferred_captures_share_one_directory_barrier() { - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let host = Host::default().with_filesystem(faults.clone()); - let mut writer = Db::open_with_host( - &directory.path().join("deferred.sqlite"), - Limits::default(), - host, - ) - .unwrap(); - - writer - .transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - let first = writer.capture_deferred().unwrap(); - writer - .transaction(|tx| tx.execute("INSERT INTO t VALUES(2)", [])) - .unwrap(); - let second = writer.capture_deferred().unwrap(); - - assert_eq!(faults.file_syncs.load(Ordering::Relaxed), 0); - assert_eq!(faults.parent_syncs.load(Ordering::Relaxed), 0); - assert!(!first.segments.is_empty()); - assert!(!second.segments.is_empty()); - - writer.durability_barrier().unwrap(); - assert_eq!( - faults.file_syncs.load(Ordering::Relaxed), - first.segments.len() + second.segments.len() - ); - // The first local barrier also seals the new ltx/0, ltx, and session names. - assert_eq!(faults.parent_syncs.load(Ordering::Relaxed), 4); - writer - .transaction(|tx| tx.execute("INSERT INTO t VALUES(3)", [])) - .unwrap(); - let third = writer.capture_deferred().unwrap(); - writer.durability_barrier().unwrap(); - assert_eq!( - faults.file_syncs.load(Ordering::Relaxed), - first.segments.len() + second.segments.len() + third.segments.len() - ); - assert_eq!(faults.parent_syncs.load(Ordering::Relaxed), 5); - writer.close().unwrap(); -} -#[test] -fn failed_deferred_barrier_fences_before_acknowledgement() { - for operation in ["sync_all", "sync_parent"] { - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let host = Host::default().with_filesystem(faults.clone()); - let mut writer = Db::open_with_host( - &directory.path().join("deferred-failure.sqlite"), - Limits::default(), - host, - ) - .unwrap(); - writer - .transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - let captured = writer.capture_deferred().unwrap(); - - faults.arm(Some(operation)); - assert!(matches!( - writer.durability_barrier(), - Err(CrabError::Io(error)) if error.kind() == io::ErrorKind::StorageFull - )); - assert!(matches!(writer.capture(), Err(CrabError::Fenced))); - assert!(!captured.segments.is_empty()); - } -} -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cell_checksum_write_failure_fences_after_sealing_the_cut() { - let (directory, faults, host, mut source) = fixture(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("cell-checksums"), - [4; 16], - ), - [5; 32], - [6; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let root = replica - .prepare(None, &source.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - source.close().unwrap(); - - let destination = directory.path().join("cell-active.sqlite"); - let writable = replica - .open_root(&root) - .await - .unwrap() - .paged() - .prepare_writable(&destination) - .await - .unwrap(); - let mut writer = writable.open_writable(&destination).unwrap(); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(2)")) - .unwrap(); - let synced_before_cut = faults.file_syncs.load(Ordering::Relaxed); - writer.capture_deferred().unwrap(); - assert_eq!(faults.file_syncs.load(Ordering::Relaxed), synced_before_cut); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(3)")) - .unwrap(); - faults.arm(Some("write_all_at")); - injected(writer.capture_deferred()); - faults.arm(None); - assert!(matches!(writer.capture(), Err(CrabError::Fenced))); -} -#[test] -fn capture_partial_write_sync_and_rename_failures_fence_the_session() { - for operation in ["write_all", "sync_all", "rename"] { - let (_directory, faults, _host, mut writer) = fixture(); - faults.arm(Some(operation)); - injected(writer.capture()); - faults.arm(None); - assert!( - matches!(writer.transaction(|_| Ok(())), Err(CrabError::Fenced)), - "{operation}" - ); - } -} -#[cfg(feature = "replica")] -#[test] -fn published_cut_pruning_bounds_each_filesystem_transfer() { - let (_directory, faults, host, mut writer) = fixture(); - writer - .transaction(|tx| tx.execute("INSERT INTO t VALUES(randomblob(2000000))", [])) - .unwrap(); - let batch = writer.capture_deferred().unwrap(); - assert!(batch.segments[0].info().size_bytes > 1_000_000); - let retained = host.local_disk_used(); - faults.largest_read.store(0, Ordering::Relaxed); - - assert_eq!(writer.prune_captured(&batch).unwrap(), batch.segments.len()); - - assert!(faults.largest_read.load(Ordering::Relaxed) < 128 * 1024); - assert!(host.local_disk_used() < retained); - assert!( - batch - .segments - .iter() - .all(|segment| !segment.path().exists()) - ); - writer.close().unwrap(); -} - -#[cfg(feature = "replica")] -#[test] -fn published_cut_pruning_rejects_changed_bytes_without_releasing_accounting() { - for damage in ["truncated", "extended", "corrupted", "replaced"] { - let (_directory, _faults, host, mut writer) = fixture(); - let batch = writer.capture_deferred().unwrap(); - let path = batch.segments[0].path(); - let original = std::fs::read(path).unwrap(); - let retained = host.local_disk_used(); - let mut changed = original.clone(); - match damage { - "truncated" => changed.truncate(changed.len() - 1), - "extended" => changed.push(0), - "corrupted" => { - let middle = changed.len() / 2; - changed[middle] ^= 1; - } - "replaced" => { - // A separately valid cut must not satisfy this batch's identity. - let (_other_directory, _faults, _host, mut other) = fixture(); - let other_batch = other.capture().unwrap(); - changed = std::fs::read(other_batch.segments[0].path()).unwrap(); - } - _ => unreachable!(), - } - std::fs::write(path, &changed).unwrap(); - - assert!(writer.prune_captured(&batch).is_err(), "{damage}"); - assert!(path.exists(), "{damage}"); - assert_eq!(host.local_disk_used(), retained, "{damage}"); - - std::fs::write(path, &original).unwrap(); - assert_eq!(writer.prune_captured(&batch).unwrap(), batch.segments.len()); - assert_eq!(writer.prune_captured(&batch).unwrap(), 0); - writer.close().unwrap(); - } -} - -#[cfg(feature = "replica")] -#[test] -fn captured_pruning_retains_accounting_after_io_failure() { - for operation in ["read_exact_at", "remove_file"] { - let (_directory, faults, _host, mut writer) = fixture(); - let batch = writer.capture().unwrap(); - faults.arm(Some(operation)); - injected(writer.prune_captured(&batch)); - assert!(batch.segments[0].path().exists()); - faults.arm(None); - assert_eq!(writer.prune_captured(&batch).unwrap(), batch.segments.len()); - assert_eq!(writer.prune_captured(&batch).unwrap(), 0); - } -} -#[cfg(feature = "replica")] -#[test] -fn published_deferred_capture_is_pruned_without_a_local_durability_barrier() { - let (_directory, faults, _host, mut writer) = fixture(); - let batch = writer.capture_deferred().unwrap(); - - assert_eq!(faults.file_syncs.load(Ordering::Relaxed), 0); - assert_eq!(faults.parent_syncs.load(Ordering::Relaxed), 0); - faults.arm(Some("sync_parent")); - assert_eq!(writer.prune_captured(&batch).unwrap(), batch.segments.len()); - faults.arm(None); - assert_eq!(faults.file_syncs.load(Ordering::Relaxed), 0); - assert_eq!(faults.parent_syncs.load(Ordering::Relaxed), 0); - - writer.close().unwrap(); - assert_eq!(faults.file_syncs.load(Ordering::Relaxed), 0); - assert_eq!(faults.parent_syncs.load(Ordering::Relaxed), 0); -} - -fn resumed_checksum_fixture() -> (tempfile::TempDir, Arc, Db) { - let (directory, faults, host, mut source) = fixture(); - source - .transaction(|tx| tx.execute("INSERT INTO t VALUES(zeroblob(33554432))", [])) - .unwrap(); - source.capture().unwrap(); - source.persist_continuation().unwrap(); - source.close().unwrap(); - faults.track_all.store(true, Ordering::Relaxed); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("checksum-handoff"), - [4; 16], - ), - [5; 32], - [6; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let resumed = replica - .open_resumed( - &directory.path().join("source.sqlite"), - &directory.path().join("resumed.sqlite"), - ) - .unwrap(); - (directory, faults, resumed) -} - -#[test] -fn checksum_handoff_batches_reads_and_preserves_the_dense_sidecar() { - let (directory, faults, resumed) = resumed_checksum_fixture(); - let sidecar = directory.path().join("resumed.sqlite.crab-ltx-checksums"); - let before = std::fs::read(&sidecar).unwrap(); - faults.read_calls.store(0, Ordering::Relaxed); - faults.largest_read.store(0, Ordering::Relaxed); - - resumed.persist_continuation().unwrap(); - - assert!( - faults.read_calls.load(Ordering::Relaxed) <= before.len().div_ceil(64 * 1024), - "checksum handoff made {} reads for {} bytes", - faults.read_calls.load(Ordering::Relaxed), - before.len() - ); - assert!(faults.largest_read.load(Ordering::Relaxed) <= 64 * 1024); - assert_eq!(std::fs::read(sidecar).unwrap(), before); - resumed.close().unwrap(); -} - -#[test] -fn checksum_capture_batches_adjacent_updates() { - let (_directory, faults, mut resumed) = resumed_checksum_fixture(); - resumed - .transaction(|tx| tx.execute("UPDATE t SET v=randomblob(length(v))", [])) - .unwrap(); - faults.write_calls.store(0, Ordering::Relaxed); - faults.largest_write.store(0, Ordering::Relaxed); - faults.checksum_reads.store(0, Ordering::Relaxed); - faults.checksum_read_bytes.store(0, Ordering::Relaxed); - faults.largest_checksum_read.store(0, Ordering::Relaxed); - - let captured = resumed.capture_deferred().unwrap(); - - // Checkpoint maintenance can seal another cut. Each cut must read the - // preceding cut's merged sidecar through a fresh bounded window. - let checksum_bytes = captured - .segments - .iter() - .map(|cut| cut.info().database_pages as usize * 8) - .sum::(); - let checksum_blocks = captured - .segments - .iter() - .map(|cut| (cut.info().database_pages as usize * 8).div_ceil(4096)) - .sum::(); - assert!( - faults.checksum_reads.load(Ordering::Relaxed) <= checksum_blocks, - "capture made {} checksum reads for {checksum_bytes} sidecar bytes", - faults.checksum_reads.load(Ordering::Relaxed), - ); - assert!(faults.checksum_read_bytes.load(Ordering::Relaxed) <= checksum_bytes); - assert!(faults.largest_checksum_read.load(Ordering::Relaxed) <= 4096); - eprintln!( - "checksum capture: {} reads, {} bytes, {} maximum transfer, {} cuts", - faults.checksum_reads.load(Ordering::Relaxed), - faults.checksum_read_bytes.load(Ordering::Relaxed), - faults.largest_checksum_read.load(Ordering::Relaxed), - captured.segments.len(), - ); - - assert!( - faults.write_calls.load(Ordering::Relaxed) < 1024, - "capture made {} writes for {} cut bytes", - faults.write_calls.load(Ordering::Relaxed), - captured - .segments - .iter() - .map(|cut| cut.info().size_bytes) - .sum::() - ); - assert!(faults.largest_write.load(Ordering::Relaxed) <= 64 * 1024); - // Updating the same small row in successive cuts must reread its block; - // retaining a window across the sidecar merge would use stale checksums. - for value in [17, 23] { - resumed - .transaction(|tx| tx.execute("UPDATE t SET v=? WHERE rowid=1", [value])) - .unwrap(); - faults.checksum_reads.store(0, Ordering::Relaxed); - faults.checksum_read_bytes.store(0, Ordering::Relaxed); - let small = resumed.capture_deferred().unwrap(); - let changed = small - .segments - .iter() - .map(|cut| { - crab_ltx::internal::inspect_ltx(&std::fs::read(cut.path()).unwrap()) - .unwrap() - .pages as usize - }) - .sum::(); - assert!( - changed < 8, - "a point edit unexpectedly captured {changed} pages" - ); - assert!(faults.checksum_reads.load(Ordering::Relaxed) <= changed); - assert!(faults.checksum_read_bytes.load(Ordering::Relaxed) <= changed * 4096); - } - // Re-read and fold the persisted blocks before allowing a clean handoff. - resumed.persist_continuation().unwrap(); - resumed.close().unwrap(); -} diff --git a/crates/crab-ltx/tests/host/hooks/compaction.rs b/crates/crab-ltx/tests/host/hooks/compaction.rs deleted file mode 100644 index 1bf65ea66..000000000 --- a/crates/crab-ltx/tests/host/hooks/compaction.rs +++ /dev/null @@ -1,394 +0,0 @@ -//! Compaction transfer overlap, local write coalescing, and scratch cleanup. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn compaction_reservation_covers_all_coexisting_scratch_files() { - for page_size in [512, 65_536] { - let source = tempfile::TempDir::new().unwrap(); - let path = source.path().join("source.sqlite"); - let initial = crab_ltx::rusqlite::Connection::open(&path).unwrap(); - initial - .execute_batch(&format!("PRAGMA page_size={page_size}; CREATE TABLE t(v)")) - .unwrap(); - drop(initial); - let mut writer = Db::open(&path, Limits::default()).unwrap(); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(randomblob(1048576))")) - .unwrap(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("scratch-bound"), - [111; 16], - ), - [112; 32], - [113; 16], - Limits::default(), - ) - .unwrap(); - let prepared = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap(); - let root = prepared.root(); - let count = prepared.verified().segment_count(); - writer.close().unwrap(); - let scratch = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let slots = Arc::new(tokio::sync::Semaphore::new(8)); - let replica = replica.with_host( - Host::default() - .with_filesystem(faults.clone()) - .with_scratch_slots(slots.clone()), - ); - let pause = Arc::new(Pause { - operation: "remove_file", - entered: tokio::sync::Notify::new(), - released: Mutex::new(false), - wake: std::sync::Condvar::new(), - }); - let release = Release(pause.clone()); - *faults.pause.lock().unwrap() = Some(pause.clone()); - let destination = scratch.path().to_owned(); - let mut task = tokio::spawn(async move { - replica - .prepare_compaction(&root, 0..count, 9, &destination) - .await - }); - tokio::select! { - result = &mut task => panic!("compaction finished before cleanup: {:?}", result.unwrap().map(|prepared| prepared.root())), - result = tokio::time::timeout(Duration::from_secs(10), pause.entered.notified()) => result.unwrap(), - } - // Every spool/output grows monotonically and cleanup has not started: - // these five final file lengths are the operation's peak logical bytes. - let sizes = std::fs::read_dir(scratch.path()) - .unwrap() - .map(|entry| entry.unwrap().metadata().unwrap().len()) - .collect::>(); - assert_eq!(sizes.len(), 5); - assert!( - sizes.iter().sum::() <= ((8 - slots.available_permits()) << 20) as u64, - "{page_size}-byte pages exceed reserved scratch" - ); - drop(release); - assert_eq!(task.await.unwrap().unwrap().root().position, root.position); - assert_eq!(slots.available_permits(), 8); - assert_eq!(std::fs::read_dir(scratch.path()).unwrap().count(), 0); - } -} - -#[cfg(feature = "replica")] -#[tokio::test(start_paused = true)] -async fn compaction_overlaps_independent_remote_transfers() { - let (_directory, _faults, _host, mut writer) = fixture(); - let mut captured = writer.capture().unwrap(); - for value in [2, 3, 4] { - writer - .transaction(|tx| tx.execute("INSERT INTO t VALUES(?1)", [value])) - .unwrap(); - let next = writer.capture().unwrap(); - captured.segments.extend(next.segments); - captured.position = next.position; - } - assert_eq!(captured.segments.len(), 4); - let delay = Duration::from_millis(100); - let backend = InMemory::new(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(ThrottledStore::new( - backend, - ThrottleConfig { - wait_get_per_call: delay, - wait_put_per_call: delay, - ..ThrottleConfig::default() - }, - ))), - ObjectPath::from("parallel-compaction-downloads"), - [84; 16], - ), - [85; 32], - [86; 16], - Limits::default(), - ) - .unwrap(); - let root = replica.prepare(None, &captured, 1, 1).await.unwrap().root(); - let scratch = tempfile::TempDir::new().unwrap(); - let started = tokio::time::Instant::now(); - - replica - .prepare_compaction(&root, 0..4, 9, scratch.path()) - .await - .unwrap(); - - // Cached root metadata is presence-checked in one parallel HEAD wave. - // Streaming directory construction still precedes the final root upload. - assert_eq!(started.elapsed(), delay * 5); - writer.close().unwrap(); -} -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cell_compaction_coalesces_local_output_writes() { - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let host = Host::default().with_filesystem(faults.clone()); - let mut writer = Db::open_with_host( - &directory.path().join("source.sqlite"), - Limits::default(), - host.clone(), - ) - .unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(3000000))", - ) - }) - .unwrap(); - let captured = writer.capture().unwrap(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("buffered-compaction-output"), - [66; 16], - ), - [67; 32], - [68; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let root = replica.prepare(None, &captured, 1, 1).await.unwrap().root(); - faults.track_all.store(true, Ordering::Relaxed); - faults.read_calls.store(0, Ordering::Relaxed); - faults.write_calls.store(0, Ordering::Relaxed); - let scratch = tempfile::TempDir::new().unwrap(); - - let compacted = replica - .prepare_compaction(&root, 0..1, 9, scratch.path()) - .await - .unwrap(); - - assert_eq!(compacted.root().position, root.position); - assert!( - faults.write_calls.load(Ordering::Relaxed) <= 1_000, - "compaction used {} local writes", - faults.write_calls.load(Ordering::Relaxed) - ); - assert!( - faults.read_calls.load(Ordering::Relaxed) < 128, - "compaction used {} local reads", - faults.read_calls.load(Ordering::Relaxed) - ); - writer.close().unwrap(); -} - -#[tokio::test] -async fn cell_compaction_dispatches_local_io_with_one_job_slot() { - let (directory, faults, host, mut writer) = fixture(); - let jobs = Arc::new(tokio::sync::Semaphore::new(1)); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("compaction-job-admission"), - [91; 16], - ), - [92; 32], - [93; 16], - Limits::default(), - ) - .unwrap() - .with_host(host.with_job_slots(jobs.clone())); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - *faults.forbidden_thread.lock().unwrap() = Some(std::thread::current().id()); - - let result = tokio::time::timeout( - Duration::from_secs(5), - replica.prepare_compaction(&root, 0..1, 9, directory.path()), - ) - .await; - - *faults.forbidden_thread.lock().unwrap() = None; - let compacted = result.unwrap().unwrap(); - assert_eq!(compacted.root().position, root.position); - assert_eq!(jobs.available_permits(), 1); - assert!(!std::fs::read_dir(directory.path()).unwrap().any(|entry| { - entry - .unwrap() - .file_name() - .to_string_lossy() - .contains(".crab-compaction-") - })); -} -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cell_compaction_uses_injected_filesystem_and_cleans_failed_scratch() { - let (directory, faults, host, mut writer) = fixture(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("cell-compaction"), - [11; 16], - ), - [12; 32], - [13; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - - for operation in [ - "create", - "open_rw", - "read_exact_at", - "write_all_at", - "write_all", - "sync_all", - ] { - faults.arm(Some(operation)); - injected( - replica - .prepare_compaction(&root, 0..1, 9, directory.path()) - .await, - ); - assert!(!std::fs::read_dir(directory.path()).unwrap().any(|entry| { - entry - .unwrap() - .file_name() - .to_string_lossy() - .contains(".crab-compaction-") - })); - } - - faults.arm(None); - let compacted = replica - .prepare_compaction(&root, 0..1, 9, directory.path()) - .await - .unwrap(); - assert_eq!(compacted.root().position, root.position); - assert!(faults.calls.lock().unwrap().contains("open_rw")); -} - -#[tokio::test] -async fn canceled_compaction_retains_files_and_admission_through_cleanup() { - for operation in [ - "create", - "open_rw", - "read_exact_at", - "write_all_at", - "write_all", - "sync_all", - ] { - let (directory, faults, host, mut writer) = fixture(); - let jobs = Arc::new(tokio::sync::Semaphore::new(1)); - let dirty = Arc::new(tokio::sync::Semaphore::new(1)); - let recovery = Arc::new(tokio::sync::Semaphore::new(1)); - let scratch = Arc::new(tokio::sync::Semaphore::new(128)); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("canceled-compaction"), - [94; 16], - ), - [95; 32], - [96; 16], - Limits::default(), - ) - .unwrap() - .with_host( - host.with_job_slots(jobs.clone()) - .with_dirty_slots(dirty.clone()) - .with_recovery_slots(recovery.clone()) - .with_scratch_slots(scratch.clone()), - ); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - let pause = Arc::new(Pause { - operation, - entered: tokio::sync::Notify::new(), - released: Mutex::new(false), - wake: std::sync::Condvar::new(), - }); - let release = Release(pause.clone()); - *faults.pause.lock().unwrap() = Some(pause.clone()); - *faults.forbidden_thread.lock().unwrap() = Some(std::thread::current().id()); - let task_replica = replica.clone(); - let destination = directory.path().to_owned(); - let task = tokio::spawn(async move { - task_replica - .prepare_compaction(&root, 0..1, 9, &destination) - .await - }); - tokio::time::timeout(Duration::from_secs(2), pause.entered.notified()) - .await - .unwrap(); - task.abort(); - assert!(matches!(task.await, Err(error) if error.is_cancelled())); - assert_eq!(jobs.available_permits(), 0, "{operation}"); - assert_eq!(dirty.available_permits(), 0, "{operation}"); - assert_eq!(recovery.available_permits(), 0, "{operation}"); - assert!(scratch.available_permits() < 128, "{operation}"); - - let cleanup_pause = Arc::new(Pause { - operation: "remove_file", - entered: tokio::sync::Notify::new(), - released: Mutex::new(false), - wake: std::sync::Condvar::new(), - }); - let cleanup_release = Release(cleanup_pause.clone()); - *faults.pause.lock().unwrap() = Some(cleanup_pause.clone()); - drop(release); - tokio::time::timeout(Duration::from_secs(2), cleanup_pause.entered.notified()) - .await - .unwrap(); - assert_eq!(jobs.available_permits(), 0, "cleanup after {operation}"); - assert_eq!(dirty.available_permits(), 0, "cleanup after {operation}"); - assert_eq!(recovery.available_permits(), 0, "cleanup after {operation}"); - assert!( - scratch.available_permits() < 128, - "cleanup after {operation}" - ); - *faults.pause.lock().unwrap() = None; - drop(cleanup_release); - tokio::time::timeout(Duration::from_secs(2), async { - let _job = jobs.acquire().await.unwrap(); - let _dirty = dirty.acquire().await.unwrap(); - let _recovery = recovery.acquire().await.unwrap(); - let _scratch = scratch.acquire_many(128).await.unwrap(); - }) - .await - .unwrap(); - *faults.forbidden_thread.lock().unwrap() = None; - assert!( - !std::fs::read_dir(directory.path()).unwrap().any(|entry| { - entry - .unwrap() - .file_name() - .to_string_lossy() - .contains(".crab-compaction-") - }), - "{operation}" - ); - let compacted = replica - .prepare_compaction(&root, 0..1, 9, directory.path()) - .await - .unwrap(); - assert_eq!(compacted.root().position, root.position); - } -} diff --git a/crates/crab-ltx/tests/host/hooks/injection.rs b/crates/crab-ltx/tests/host/hooks/injection.rs deleted file mode 100644 index a6d801abe..000000000 --- a/crates/crab-ltx/tests/host/hooks/injection.rs +++ /dev/null @@ -1,290 +0,0 @@ -//! Injected host hooks, plans, and VFS selection. - -use super::*; - -#[tokio::test(flavor = "multi_thread")] -async fn directory_cache_fill_releases_origin_admission_before_local_sync() { - let backend = Arc::new(InMemory::new()); - for jobs in [1, 2] { - verify_cache_fill_admission(|| Store::new(backend.clone()), "cache-admission", jobs).await; - } -} - -#[tokio::test(flavor = "multi_thread")] -#[ignore = "requires the RustFS environment documented in examples/README.md"] -async fn rustfs_directory_cache_fill_releases_origin_admission_before_local_sync() { - let bucket = std::env::var("CRAB_LTX_TEST_BUCKET").unwrap(); - let endpoint = std::env::var("CRAB_LTX_TEST_ENDPOINT").unwrap(); - let access_key_id = std::env::var("AWS_ACCESS_KEY_ID").unwrap(); - let secret_access_key = std::env::var("AWS_SECRET_ACCESS_KEY").unwrap(); - let run = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_nanos(); - let prefix = format!( - "crab-ltx-tests/cache-admission/{run}-{}", - std::process::id() - ); - let store = || { - crab_storage::build_explicit_store( - &bucket, - crab_storage::ObjectStoreCredentials::Aws { - access_key_id: access_key_id.clone(), - secret_access_key: secret_access_key.clone(), - session_token: None, - region: "us-east-1".into(), - }, - Some(&endpoint), - endpoint.starts_with("http://"), - ) - .unwrap() - }; - for jobs in [1, 2] { - verify_cache_fill_admission(&store, &prefix, jobs).await; - } - eprintln!("RustFS cache admission and exact SQLite restore passed: {prefix}"); -} - -async fn verify_cache_fill_admission(store: impl Fn() -> Store, prefix: &str, job_count: usize) { - let (directory, faults, host, mut writer) = fixture(); - let replica = |cell| { - CellReplica::new( - CellStorageLayout::new(store(), ObjectPath::from(prefix), [231; 16]), - [cell; 32], - [233; 16], - Limits::default(), - ) - .unwrap() - }; - let captured = writer.capture().unwrap(); - let root = replica(232) - .prepare(None, &captured, 1, 1) - .await - .unwrap() - .root(); - let other_root = replica(234) - .prepare(None, &captured, 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - let io = Arc::new(tokio::sync::Semaphore::new(1)); - let jobs = Arc::new(tokio::sync::Semaphore::new(job_count)); - let cache_root = directory.path().join("cache"); - let cache_host = host - .with_io_slots(io.clone()) - .with_job_slots(jobs.clone()) - .with_directory_cache(cache_root.clone()) - .await - .unwrap(); - let reader = replica(232).with_host(cache_host.clone()); - let pause = Arc::new(Pause { - operation: "sync_all", - entered: tokio::sync::Notify::new(), - released: Mutex::new(false), - wake: std::sync::Condvar::new(), - }); - *faults.pause.lock().unwrap() = Some(pause.clone()); - let release = Release(pause.clone()); - let mut read = tokio::spawn(async move { reader.open_root(&root).await }); - tokio::time::timeout(Duration::from_secs(5), pause.entered.notified()) - .await - .unwrap(); - let reopened = tokio::time::timeout(Duration::from_secs(1), &mut read) - .await - .expect("verified root must not wait for optional cache persistence") - .unwrap() - .unwrap(); - assert_eq!(reopened.root(), root); - - let same = replica(232).with_host(cache_host.clone()); - tokio::time::timeout(Duration::from_secs(1), same.open_root(&root)) - .await - .expect("a concurrent lookup must skip the paused fill lock") - .unwrap(); - - // A fresh store identity bypasses the process byte cache. Both readers - // share the paused disk cache and its blocking job slots. - let other = replica(234).with_host(cache_host.clone()); - let verified = tokio::time::timeout(Duration::from_secs(1), other.open_root(&other_root)).await; - assert_eq!(jobs.available_permits(), 0); - assert!( - tokio::time::timeout(Duration::from_millis(50), cache_host.drain_cache_fills()) - .await - .is_err() - ); - let reopen = cache_host.clone().with_directory_cache(cache_root); - tokio::pin!(reopen); - assert!( - tokio::time::timeout(Duration::from_millis(50), &mut reopen) - .await - .is_err() - ); - drop(release); - let reopened_cache = tokio::time::timeout(Duration::from_secs(5), &mut reopen) - .await - .unwrap() - .unwrap(); - cache_host.drain_cache_fills().await; - assert!(reopened_cache.directory_cache_stats().unwrap().entries() > 0); - let finished = tokio::time::timeout(Duration::from_secs(5), jobs.acquire()) - .await - .unwrap() - .unwrap(); - drop(finished); - let verified = verified - .expect("cache persistence must not hold origin admission") - .unwrap(); - let destination = directory.path().join("restored.sqlite"); - verified.restore(&destination).await.unwrap(); - let db = crab_ltx::rusqlite::Connection::open(destination).unwrap(); - let count: u64 = db - .query_row("SELECT count(*) FROM t", [], |row| row.get(0)) - .unwrap(); - assert_eq!(count, 1); -} - -#[test] -fn committed_wal_observation_uses_the_injected_reader() { - let (_directory, faults, _host, mut writer) = fixture(); - faults.arm(Some("read_exact_at")); - injected(writer.transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(2)"))); - faults.arm(None); - assert!(matches!(writer.capture(), Err(CrabError::Fenced))); -} -#[test] -fn fresh_session_claims_do_not_bypass_the_host() { - for operation in ["canonicalize", "file_len", "create_dir"] { - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - faults.arm(Some(operation)); - injected(Db::open_with_host( - &directory.path().join("db"), - Limits::default(), - Host::default().with_filesystem(faults), - )); - assert!(!directory.path().join("db").exists(), "{operation}"); - } -} -#[test] -fn exact_local_resume_and_all_checkpoints_preserve_the_injected_plan() { - let (directory, faults, host, mut writer) = fixture(); - let first = writer.capture().unwrap(); - faults.arm(Some("open")); - injected(host.verify(&first.segments, first.position, Limits::default())); - faults.arm(None); - let plan = host - .verify(&first.segments, first.position, Limits::default()) - .unwrap(); - writer.close().unwrap(); - let destination = directory.path().join("resumed"); - for operation in ["exists", "persist_new"] { - faults.arm(Some(operation)); - injected(Db::resume_with_host( - &plan, - &destination, - Limits::default(), - host.clone(), - )); - assert!(!destination.exists(), "{operation}"); - } - faults.arm(None); - let mut writer = - Db::resume_with_host(&plan, &destination, Limits::default(), host.clone()).unwrap(); - assert_eq!(writer.position(), first.position); - let mut segments = first.segments; - for (index, mode) in [ - CheckpointMode::Passive, - CheckpointMode::Full, - CheckpointMode::Restart, - CheckpointMode::Truncate, - ] - .into_iter() - .enumerate() - { - writer - .transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(1)")) - .unwrap(); - let batch = writer.checkpoint(mode).unwrap(); - assert_eq!(batch.timing.checkpoint_runs, 1); - assert!(batch.timing.checkpoint_frames >= batch.timing.checkpoint_backfilled); - assert_eq!(batch.timing.checkpoint_busy_errors, 0); - segments.extend(batch.segments); - let plan = host - .verify(&segments, batch.position, Limits::default()) - .unwrap(); - let restored = directory.path().join(format!("check-{index}")); - host.restore(&plan, &restored).unwrap(); - let db = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - let count: usize = db - .query_row("SELECT count(*) FROM t", [], |r| r.get(0)) - .unwrap(); - assert_eq!(count, index + 2); - } - let calls = faults.calls.lock().unwrap(); - for operation in [ - "canonicalize", - "create_dir", - "read_exact_at", - "exists", - "persist_new", - ] { - assert!(calls.contains(operation), "{operation}"); - } -} -#[test] -fn snapshot_and_compaction_installation_are_injectable_and_never_clobber() { - let (directory, faults, host, mut writer) = fixture(); - let batch = writer.capture().unwrap(); - let plan = host - .verify(&batch.segments, batch.position, Limits::default()) - .unwrap(); - let destination = directory.path().join("snapshot.ltx"); - for operation in [ - "create", - "write_all", - "sync_all", - "open", - "persist_file_new", - ] { - faults.arm(Some(operation)); - injected(host.compact(&plan, &destination)); - assert!(!destination.exists(), "{operation}"); - assert!( - std::fs::read_dir(directory.path()) - .unwrap() - .flatten() - .all(|entry| !entry - .file_name() - .to_string_lossy() - .contains(".tmp-crab-ltx-compaction-")), - "{operation} left compaction scratch" - ); - } - faults.arm(Some("persist_file_new")); - injected(writer.snapshot(&destination)); - assert!(!destination.exists()); - faults.arm(None); - faults.track_all.store(true, Ordering::Relaxed); - faults.largest_write.store(0, Ordering::Relaxed); - host.compact(&plan, &destination).unwrap(); - assert!(faults.largest_write.load(Ordering::Relaxed) < 128 * 1024); - assert!(!faults.calls.lock().unwrap().contains("persist_new")); - let before = std::fs::read(&destination).unwrap(); - assert!(host.compact(&plan, &destination).is_err()); - assert_eq!(std::fs::read(destination).unwrap(), before); -} -#[test] -fn unknown_sqlite_vfs_does_not_fall_back_to_the_platform_vfs() { - let directory = tempfile::TempDir::new().unwrap(); - let path = directory.path().join("db"); - assert!( - Db::open_with_host( - &path, - Limits::default(), - Host::default().with_sqlite_vfs("absent-test-vfs") - ) - .is_err() - ); - assert!(!path.exists()); -} diff --git a/crates/crab-ltx/tests/host/hooks/matrix.rs b/crates/crab-ltx/tests/host/hooks/matrix.rs deleted file mode 100644 index 87c953362..000000000 --- a/crates/crab-ltx/tests/host/hooks/matrix.rs +++ /dev/null @@ -1,241 +0,0 @@ -//! Ordered fault matrix across the durability seams. -//! -//! Each case injects a failure at one seam and asserts three things: the call -//! fails with the class callers branch on, the durable outcome is safe (no -//! published cut, no installed destination, or a fenced session), and a clean -//! retry after the fault clears produces the exact artifact. - -use super::*; -use crab_ltx::FailureClass; - -/// Lists the published LTX file names under one managed database. -fn published_cuts(database: &Path) -> Vec { - let name = database - .file_name() - .expect("database file name") - .to_string_lossy() - .into_owned(); - let directory = database - .parent() - .expect("database parent") - .join(format!(".{name}-crab-ltx/ltx/0")); - let Ok(entries) = std::fs::read_dir(directory) else { - return Vec::new(); - }; - let mut names: Vec = entries - .filter_map(|entry| entry.ok()) - .map(|entry| entry.file_name().to_string_lossy().into_owned()) - .filter(|name| name.ends_with(".ltx")) - .collect(); - names.sort(); - names -} - -/// Asserts the session refuses further work once a capture failure fenced it. -fn assert_fenced(writer: &mut Db) { - assert!(matches!( - writer.transaction(|_| Ok(())), - Err(CrabError::Fenced) - )); -} - -#[test] -fn exhausted_local_disk_budget_refuses_before_the_commit() { - let directory = tempfile::TempDir::new().unwrap(); - // Each transaction reserves its worst-case capture up front, so the bound - // that matters here is the per-capture bound, not the database size. - let limits = Limits { - max_capture_bytes: 64 * 1024, - max_file_bytes: 1024 * 1024, - ..Limits::default() - }; - let budget = crab_ltx::DiskBudget::new(600 * 1024); - let host = Host::default().with_local_disk_budget(budget.clone()); - let mut writer = - Db::open_with_host(&directory.path().join("bounded.sqlite"), limits, host).unwrap(); - writer - .transaction(|tx| tx.execute_batch("CREATE TABLE t(v)")) - .unwrap(); - writer.capture().unwrap(); - - let mut refusals = 0; - let mut committed = 0_i64; - for _ in 0..128 { - let result = - writer.transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(randomblob(16384))")); - match result { - Ok(()) => { - writer.capture().unwrap(); - committed += 1; - } - Err(CrabError::Limit(crab_ltx::LimitKind::LocalDiskBytes)) => { - refusals += 1; - break; - } - Err(error) => panic!("unexpected refusal: {error}"), - } - } - assert_eq!(refusals, 1, "the budget must refuse the next commit"); - assert!(budget.used() <= budget.capacity()); - // The refusal happened before SQLite published anything: the refused row is - // absent, no cut is pending, and the session is not fenced. - assert!(!writer.has_pending_capture()); - let rows = writer - .query_with(|connection| { - connection.query_row("SELECT count(*) FROM t", [], |row| row.get::<_, i64>(0)) - }) - .unwrap(); - assert_eq!(rows, committed); -} - -#[test] -fn capture_write_and_sync_faults_leave_no_published_cut() { - for operation in ["write_all", "sync_all", "rename"] { - let (directory, faults, _host, mut writer) = fixture(); - let database = directory.path().join("source.sqlite"); - assert!(published_cuts(&database).is_empty()); - - faults.plan([operation]); - let error = writer.capture().unwrap_err(); - assert_eq!( - error.classify(), - FailureClass::Capacity, - "{operation}: an injected disk-full failure is a capacity class" - ); - assert!( - published_cuts(&database).is_empty(), - "{operation}: a failed capture must not publish a cut" - ); - assert_fenced(&mut writer); - } -} - -#[test] -fn deferred_capture_barrier_fault_fences_before_acknowledgement() { - let (directory, faults, _host, mut writer) = fixture(); - let database = directory.path().join("source.sqlite"); - let batch = writer.capture_deferred().unwrap(); - assert_eq!(batch.segments.len(), 1); - - faults.plan(["sync_parent"]); - let error = writer.durability_barrier().unwrap_err(); - assert_eq!(error.classify(), FailureClass::Capacity); - assert_fenced(&mut writer); - - // The deferred cut exists but was never durable, so no caller may - // acknowledge it, and no later barrier call can release it. - assert_eq!(published_cuts(&database).len(), 1); - assert!(matches!( - writer.durability_barrier(), - Err(CrabError::Fenced) - )); - // A fresh session claims a new directory instead of adopting the residue. - writer.close().unwrap(); - assert!(Db::open(&database, Limits::default()).is_err()); -} - -#[test] -fn checkpoint_boundary_fault_fences_without_publishing() { - let (directory, faults, _host, mut writer) = fixture(); - let database = directory.path().join("source.sqlite"); - let published = writer.capture().unwrap(); - assert_eq!(published.segments.len(), 1); - let cuts_before = published_cuts(&database); - - faults.plan(["rename"]); - let error = writer.checkpoint(CheckpointMode::Truncate).unwrap_err(); - assert_eq!(error.classify(), FailureClass::Capacity); - assert_eq!( - published_cuts(&database), - cuts_before, - "a failed checkpoint must not publish a boundary cut" - ); - assert_fenced(&mut writer); -} - -#[test] -fn restore_install_fault_leaves_no_destination_and_retries_exactly() { - let source = tempfile::TempDir::new().unwrap(); - let mut clean = Db::open(&source.path().join("clean.sqlite"), Limits::default()).unwrap(); - clean - .transaction(|tx| { - tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(randomblob(4096))") - }) - .unwrap(); - let batch = clean.capture().unwrap(); - clean.close().unwrap(); - - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let host = Host::default().with_filesystem(faults.clone()); - let plan = host - .verify(&batch.segments, batch.position, Limits::default()) - .unwrap(); - let destination = directory.path().join("restored.sqlite"); - - faults.plan(["persist_new"]); - let error = host.restore(&plan, &destination).unwrap_err(); - assert_eq!(error.classify(), FailureClass::Capacity); - assert!( - !destination.exists(), - "a failed install must not leave a destination" - ); - - // The identical call succeeds once the fault clears, and matches the plan. - let restored = host.restore(&plan, &destination).unwrap(); - assert_eq!(restored, batch.position); - assert!(destination.exists()); -} - -#[test] -fn compaction_install_fault_leaves_no_destination_and_retries_exactly() { - let source = tempfile::TempDir::new().unwrap(); - let mut clean = Db::open(&source.path().join("clean.sqlite"), Limits::default()).unwrap(); - clean - .transaction(|tx| { - tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(randomblob(4096))") - }) - .unwrap(); - let batch = clean.capture().unwrap(); - clean.close().unwrap(); - - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let host = Host::default().with_filesystem(faults.clone()); - let plan = host - .verify(&batch.segments, batch.position, Limits::default()) - .unwrap(); - let destination = directory.path().join("compacted.ltx"); - - faults.plan(["persist_file_new"]); - let error = host.compact(&plan, &destination).unwrap_err(); - assert_eq!(error.classify(), FailureClass::Capacity); - assert!(!destination.exists(), "a failed install must not publish"); - - let compacted = host.compact(&plan, &destination).unwrap(); - assert_eq!(compacted.info().max_txid, batch.position.txid); - let compact_plan = crab_ltx::VerifiedPlan::new( - std::slice::from_ref(&compacted), - batch.position, - Limits::default(), - ) - .unwrap(); - let restored = directory.path().join("restored.sqlite"); - assert_eq!( - crab_ltx::restore_exact(&compact_plan, &restored).unwrap(), - batch.position - ); -} - -#[test] -fn ordered_plan_injects_each_seam_once() { - let (_directory, faults, _host, mut writer) = fixture(); - // The capture writes the cut, then renames it into place: a plan that fails - // both operations must inject exactly once per seam. - faults.plan(["write_all", "rename"]); - assert!(writer.capture().is_err()); - assert!(matches!( - writer.transaction(|_| Ok(())), - Err(CrabError::Fenced) - )); -} diff --git a/crates/crab-ltx/tests/host/hooks/prepare.rs b/crates/crab-ltx/tests/host/hooks/prepare.rs deleted file mode 100644 index 435103d8b..000000000 --- a/crates/crab-ltx/tests/host/hooks/prepare.rs +++ /dev/null @@ -1,418 +0,0 @@ -//! Prepare transfer bounds, upload overlap, and cancellation cleanup. - -use super::*; - -async fn verified_fixture( - extra_bytes: i64, -) -> (tempfile::TempDir, Arc, crab_ltx::VerifiedRoot) { - let (directory, faults, host, mut source) = fixture(); - source - .transaction(|tx| { - tx.execute("INSERT INTO t VALUES(zeroblob(?1))", [extra_bytes])?; - Ok(()) - }) - .unwrap(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("async-checksums"), - [4; 16], - ), - [5; 32], - [6; 16], - Limits::default(), - ) - .unwrap() - .with_host(host.with_job_slots(Arc::new(tokio::sync::Semaphore::new(1)))); - let root = replica - .prepare(None, &source.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - source.close().unwrap(); - let verified = replica.open_root(&root).await.unwrap(); - (directory, faults, verified) -} - -#[tokio::test] -async fn writable_activation_avoids_syncing_derived_files() { - let (directory, faults, verified) = verified_fixture(0).await; - for barrier in ["sync_all", "sync_parent"] { - let destination = directory.path().join(format!("{barrier}.sqlite")); - faults.arm(Some(barrier)); - let prepared = verified.paged().prepare_writable(&destination).await; - let opened = prepared.and_then(|prepared| prepared.open_writable(&destination)); - faults.arm(None); - let mut writer = opened.unwrap(); - let bytes: i64 = writer - .query_with(|db| db.query_row("SELECT sum(length(v)) FROM t", [], |row| row.get(0))) - .unwrap(); - assert_eq!(bytes, 20_000); - writer.close().unwrap(); - } -} - -#[tokio::test] -async fn immutable_activation_avoids_syncing_derived_files() { - let (directory, faults, verified) = verified_fixture(0).await; - for barrier in ["sync_all", "sync_parent"] { - let destination = directory.path().join(format!("{barrier}.sqlite")); - faults.arm(Some(barrier)); - let opened = verified.open_read_only(&destination); - faults.arm(None); - let view = opened.unwrap(); - let bytes: i64 = view - .connection() - .unwrap() - .query_row("SELECT sum(length(v)) FROM t", [], |row| row.get(0)) - .unwrap(); - assert_eq!(bytes, 20_000); - } -} - -#[tokio::test] -async fn writable_preparation_dispatches_filesystem_work_off_the_async_thread() { - // More than 8,192 pages exercises streamed checksum writes as well as the - // final metadata check, with only one available host executor slot. - let (directory, faults, verified) = verified_fixture(35_000_000).await; - let destination = directory.path().join("active.sqlite"); - *faults.forbidden_thread.lock().unwrap() = Some(std::thread::current().id()); - let prepared = verified.paged().prepare_writable(&destination).await; - *faults.forbidden_thread.lock().unwrap() = None; - let mut writer = prepared.unwrap().open_writable(&destination).unwrap(); - let size: i64 = writer - .transaction(|connection| { - connection.query_row("SELECT sum(length(v)) FROM t", [], |row| row.get(0)) - }) - .unwrap(); - assert_eq!(size, 35_020_000); - writer.close().unwrap(); -} - -#[tokio::test] -async fn failed_checksum_preparation_cleans_up_without_blocking_the_async_thread() { - let (directory, faults, verified) = verified_fixture(0).await; - for operation in ["write_all", "file_len"] { - let destination = directory.path().join(format!("{operation}.sqlite")); - faults.arm(Some(operation)); - *faults.forbidden_thread.lock().unwrap() = Some(std::thread::current().id()); - let result = verified.paged().prepare_writable(&destination).await; - *faults.forbidden_thread.lock().unwrap() = None; - faults.arm(None); - assert!( - matches!(result, Err(CrabError::Io(error)) if error.kind() == io::ErrorKind::StorageFull) - ); - assert!( - !directory - .path() - .join(format!("{operation}.sqlite.crab-ltx-checksums")) - .exists() - ); - } -} - -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cell_prepare_bounds_source_transfers_without_local_writes() { - let (directory, faults, host, mut writer) = fixture(); - writer - .transaction(|tx| { - tx.execute("INSERT INTO t VALUES(randomblob(10000000))", [])?; - Ok(()) - }) - .unwrap(); - let captured = writer.capture().unwrap(); - faults.track_all.store(true, Ordering::Relaxed); - faults.largest_read.store(0, Ordering::Relaxed); - faults.read_calls.store(0, Ordering::Relaxed); - faults.largest_write.store(0, Ordering::Relaxed); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("cell-streaming-bound"), - [31; 16], - ), - [32; 32], - [33; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - - replica.prepare(None, &captured, 1, 1).await.unwrap(); - - assert!( - faults.largest_read.load(Ordering::Relaxed) <= 8 * 1024 * 1024, - "largest source read was {} bytes", - faults.largest_read.load(Ordering::Relaxed) - ); - assert_eq!(faults.largest_write.load(Ordering::Relaxed), 0); - let expected_reads: usize = captured - .segments - .iter() - .map(|segment| segment.info().size_bytes.div_ceil(8 << 20) as usize) - .sum(); - assert_eq!(faults.read_calls.load(Ordering::Relaxed), expected_reads); - writer.close().unwrap(); - drop(directory); -} -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn caller_constructed_segment_uses_full_inspection_fallback() { - let (_directory, faults, host, mut writer) = fixture(); - let mut captured = writer.capture().unwrap(); - captured.segments[0] = LocalSegment::new( - captured.segments[0].path().to_owned(), - captured.segments[0].info().clone(), - ); - faults.track_all.store(true, Ordering::Relaxed); - faults.read_calls.store(0, Ordering::Relaxed); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("cell-external-segment"), - [34; 16], - ), - [35; 32], - [36; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - - replica.prepare(None, &captured, 1, 1).await.unwrap(); - - assert!(faults.read_calls.load(Ordering::Relaxed) > 1); - writer.close().unwrap(); -} -#[cfg(feature = "replica")] -#[tokio::test(start_paused = true)] -async fn prepare_overlaps_independent_immutable_uploads() { - let (_directory, _faults, _host, mut writer) = fixture(); - let first = writer.capture_deferred().unwrap(); - let backend = InMemory::new(); - let layout = CellStorageLayout::new( - Store::new(Arc::new(backend.clone())), - ObjectPath::from("parallel-preparation"), - [71; 16], - ); - let initial = CellReplica::new(layout, [72; 32], [73; 16], Limits::default()).unwrap(); - let root = initial.prepare(None, &first, 1, 1).await.unwrap().root(); - writer.prune_captured(&first).unwrap(); - writer - .transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(randomblob(4096))")) - .unwrap(); - let captured = writer.capture_deferred().unwrap(); - - let delay = Duration::from_millis(100); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(ThrottledStore::new( - backend, - ThrottleConfig { - wait_put_per_call: delay, - ..ThrottleConfig::default() - }, - ))), - ObjectPath::from("parallel-preparation"), - [71; 16], - ), - [72; 32], - [73; 16], - Limits::default(), - ) - .unwrap(); - let started = tokio::time::Instant::now(); - - replica.prepare(Some(&root), &captured, 2, 1).await.unwrap(); - - assert_eq!(started.elapsed(), delay); - writer.close().unwrap(); -} -#[cfg(feature = "replica")] -#[tokio::test(start_paused = true)] -async fn warm_append_reuses_its_authenticated_root_metadata() { - let (_directory, _faults, _host, mut writer) = fixture(); - let first = writer.capture_deferred().unwrap(); - let delay = Duration::from_millis(100); - let backend = InMemory::new(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(ThrottledStore::new( - backend.clone(), - ThrottleConfig { - wait_get_per_call: delay, - wait_put_per_call: delay, - ..ThrottleConfig::default() - }, - ))), - ObjectPath::from("cached-root-metadata"), - [74; 16], - ), - [75; 32], - [76; 16], - Limits::default(), - ) - .unwrap(); - let root = replica.prepare(None, &first, 1, 1).await.unwrap().root(); - writer.prune_captured(&first).unwrap(); - writer - .transaction(|transaction| transaction.execute_batch("INSERT INTO t VALUES(2)")) - .unwrap(); - let second = writer.capture_deferred().unwrap(); - let started = tokio::time::Instant::now(); - - let prepared = replica.prepare(Some(&root), &second, 2, 1).await.unwrap(); - - // The store throttles HEAD with the PUT delay: one parallel presence - // wave plus one overlapping immutable-upload wave, with no serial - // metadata GETs or directory-before-root upload dependency. - assert_eq!(started.elapsed(), delay * 2); - assert_eq!(prepared.root().position, second.position); - let independent = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(ThrottledStore::new( - backend, - ThrottleConfig { - wait_get_per_call: delay, - ..ThrottleConfig::default() - }, - ))), - ObjectPath::from("cached-root-metadata"), - [74; 16], - ), - [75; 32], - [76; 16], - Limits::default(), - ) - .unwrap(); - let cold_started = tokio::time::Instant::now(); - independent.open_root(&prepared.root()).await.unwrap(); - assert!(cold_started.elapsed() >= delay * 2); - writer.close().unwrap(); -} -#[cfg(feature = "replica")] -#[tokio::test(start_paused = true)] -async fn prepare_opens_captured_segments_concurrently() { - let (_directory, _faults, _host, mut writer) = fixture(); - let mut captured = writer.capture_deferred().unwrap(); - for value in [2, 3] { - writer - .transaction(|tx| tx.execute("INSERT INTO t VALUES(?1)", [value])) - .unwrap(); - let next = writer.capture_deferred().unwrap(); - captured.segments.extend(next.segments); - captured.position = next.position; - } - assert_eq!(captured.segments.len(), 3); - - let delay = Duration::from_millis(100); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("parallel-capture-inputs"), - [81; 16], - ), - [82; 32], - [83; 16], - Limits::default(), - ) - .unwrap() - .with_host(Host::default().with_executor(Arc::new(DelayedExecutor { delay }))); - let started = tokio::time::Instant::now(); - - replica.prepare(None, &captured, 1, 1).await.unwrap(); - - // One open wave plus size/read/size upload jobs. Serial opens would add - // two more delay intervals before the already-concurrent uploads begin. - assert_eq!(started.elapsed(), delay * 4); - writer.close().unwrap(); -} -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cell_prepare_needs_no_scratch_or_local_durability_barrier() { - let (_directory, faults, host, mut writer) = fixture(); - let captured = writer.capture_deferred().unwrap(); - let scratch_slots = Arc::new(tokio::sync::Semaphore::new(0)); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("cell-ephemeral-scratch"), - [41; 16], - ), - [42; 32], - [43; 16], - Limits::default(), - ) - .unwrap() - .with_host( - host.with_scratch_slots(scratch_slots) - .with_local_disk_budget(crab_ltx::DiskBudget::new(0)), - ); - faults.calls.lock().unwrap().clear(); - faults.file_syncs.store(0, Ordering::Relaxed); - faults.parent_syncs.store(0, Ordering::Relaxed); - - replica.prepare(None, &captured, 1, 1).await.unwrap(); - - let calls = faults.calls.lock().unwrap(); - assert!(!calls.contains("create")); - assert!(!calls.contains("open_rw")); - assert!(!calls.contains("write_all")); - assert!(!calls.contains("sync_all")); - assert!(!calls.contains("sync_parent")); - assert_eq!(faults.file_syncs.load(Ordering::Relaxed), 0); - assert_eq!(faults.parent_syncs.load(Ordering::Relaxed), 0); - drop(calls); - writer.close().unwrap(); -} -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cancelled_cell_prepare_releases_pinned_source_without_publishing_a_root() { - let directory = tempfile::TempDir::new().unwrap(); - let mut writer = Db::open(&directory.path().join("source.sqlite"), Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction - .execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(randomblob(2000000))") - }) - .unwrap(); - let captures = writer.capture().unwrap(); - let backend = Arc::new(ThrottledStore::new( - InMemory::new(), - ThrottleConfig { - wait_put_per_call: Duration::from_secs(5), - ..ThrottleConfig::default() - }, - )); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(backend), - ObjectPath::from("cell-cancelled-prepare"), - [21; 16], - ), - [22; 32], - [23; 16], - Limits::default(), - ) - .unwrap() - .with_host(Host::default()); - - let result = tokio::time::timeout( - Duration::from_millis(250), - replica.prepare(None, &captures, 1, 1), - ) - .await; - assert!( - result.is_err(), - "the throttled immutable upload must be cancelled" - ); - writer.prune_captured(&captures).unwrap(); - assert!( - captures - .segments - .iter() - .all(|segment| !segment.path().exists()) - ); - writer.close().unwrap(); -} diff --git a/crates/crab-ltx/tests/host/hooks/restore.rs b/crates/crab-ltx/tests/host/hooks/restore.rs deleted file mode 100644 index 5ae3c2754..000000000 --- a/crates/crab-ltx/tests/host/hooks/restore.rs +++ /dev/null @@ -1,243 +0,0 @@ -//! Restore write coalescing and scratch cleanup. - -use super::*; - -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn read_view_storage_full_preserves_existing_files_and_can_retry() { - let (directory, faults, host, mut writer) = fixture(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("read-view-storage-full"), - [4; 16], - ), - [5; 32], - [6; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - let verified = replica.open_root(&root).await.unwrap(); - let stale = directory.path().join("stale-reader.sqlite"); - std::fs::write(&stale, b"untrusted previous process bytes").unwrap(); - assert!(verified.open_read_only(&stale).is_err()); - assert_eq!( - std::fs::read(&stale).unwrap(), - b"untrusted previous process bytes" - ); - - let destination = directory.path().join("fresh-reader.sqlite"); - // Snapshot placeholders are disposable and have no durability barrier. - // Only creation can fail here; publication and warm reuse own their fsyncs. - faults.arm(Some("create")); - injected(verified.open_read_only(&destination)); - assert!( - !destination.exists(), - "failed create retained a placeholder" - ); - faults.arm(None); - let view = verified.open_read_only(&destination).unwrap(); - assert_eq!(view.root(), root); - let count: u64 = view - .connection() - .unwrap() - .query_row("SELECT count(*) FROM t", [], |row| row.get(0)) - .unwrap(); - assert_eq!(count, 1); - drop(view); - assert!(!destination.exists()); -} - -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cancelled_read_view_install_removes_its_unclaimed_destination() { - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let host = Host::default().with_filesystem(faults.clone()); - let mut writer = Db::open_with_host( - &directory.path().join("source.sqlite"), - Limits::default(), - host.clone(), - ) - .unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE counter(value INTEGER NOT NULL); INSERT INTO counter VALUES (1)", - ) - }) - .unwrap(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("cancelled-read-view"), - [1; 16], - ), - [2; 32], - [3; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - let verified = replica.open_root(&root).await.unwrap(); - let pause = Arc::new(InstallPause::new()); - assert!(faults.create_pause.set(Arc::clone(&pause)).is_ok()); - let destination = directory.path().join("reader.sqlite"); - let open_destination = destination.clone(); - let open = tokio::spawn(async move { - tokio::task::spawn_blocking(move || verified.open_read_only(&open_destination)).await - }); - let entered = Arc::clone(&pause); - tokio::time::timeout( - Duration::from_secs(3), - tokio::task::spawn_blocking(move || entered.entered.wait()), - ) - .await - .unwrap() - .unwrap(); - assert!(destination.exists()); - open.abort(); - assert!(open.await.is_err()); - tokio::task::spawn_blocking(move || pause.release.wait()) - .await - .unwrap(); - tokio::time::timeout(Duration::from_secs(3), async { - while destination.exists() { - tokio::time::sleep(Duration::from_millis(10)).await; - } - }) - .await - .unwrap(); - assert!(!std::fs::read_dir(directory.path()).unwrap().any(|entry| { - entry - .unwrap() - .file_name() - .to_string_lossy() - .contains(".crab-restore-") - })); -} - -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cell_restore_write_and_install_failures_clean_owned_scratch() { - let (directory, faults, host, mut writer) = fixture(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("cell-restore"), - [1; 16], - ), - [2; 32], - [3; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - let verified = replica.open_root(&root).await.unwrap(); - let destination = directory.path().join("cell-restored.sqlite"); - - faults.arm(Some("write_all")); - injected(verified.restore(&destination).await); - assert!(!destination.exists()); - assert!(!std::fs::read_dir(directory.path()).unwrap().any(|entry| { - entry - .unwrap() - .file_name() - .to_string_lossy() - .contains(".crab-restore-") - })); - - faults.arm(Some("persist_file_new")); - injected(verified.restore(&destination).await); - assert!(!destination.exists()); - assert!(!std::fs::read_dir(directory.path()).unwrap().any(|entry| { - entry - .unwrap() - .file_name() - .to_string_lossy() - .contains(".crab-restore-") - })); - - faults.arm(None); - assert_eq!(verified.restore(&destination).await.unwrap(), root.position); -} -#[cfg(feature = "replica")] -#[tokio::test(flavor = "multi_thread")] -async fn cell_restore_coalesces_page_writes_within_bounded_windows() { - let directory = tempfile::TempDir::new().unwrap(); - let faults = Arc::new(Faults::default()); - let host = Host::default().with_filesystem(faults.clone()); - let mut writer = Db::open_with_host( - &directory.path().join("source.sqlite"), - Limits::default(), - host.clone(), - ) - .unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(value BLOB NOT NULL);\ - INSERT INTO payload VALUES(randomblob(3000000))", - ) - }) - .unwrap(); - let expected: Vec = writer - .query_with(|connection| { - connection.query_row("SELECT value FROM payload", [], |row| row.get(0)) - }) - .unwrap(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - ObjectPath::from("coalesced-restore-output"), - [69; 16], - ), - [70; 32], - [71; 16], - Limits::default(), - ) - .unwrap() - .with_host(host); - let root = replica - .prepare(None, &writer.capture().unwrap(), 1, 1) - .await - .unwrap() - .root(); - writer.close().unwrap(); - let verified = replica.open_root(&root).await.unwrap(); - faults.track_all.store(true, Ordering::Relaxed); - faults.write_calls.store(0, Ordering::Relaxed); - let destination = directory.path().join("restored.sqlite"); - - assert_eq!(verified.restore(&destination).await.unwrap(), root.position); - let actual: Vec = crab_ltx::rusqlite::Connection::open(&destination) - .unwrap() - .query_row("SELECT value FROM payload", [], |row| row.get(0)) - .unwrap(); - assert_eq!(actual, expected); - assert!( - faults.write_calls.load(Ordering::Relaxed) <= 16, - "restore used {} local writes", - faults.write_calls.load(Ordering::Relaxed) - ); - assert!(faults.largest_write.load(Ordering::Relaxed) <= 1 << 20); -} diff --git a/crates/crab-ltx/tests/host/hooks/volatile.rs b/crates/crab-ltx/tests/host/hooks/volatile.rs deleted file mode 100644 index ea65b7ae3..000000000 --- a/crates/crab-ltx/tests/host/hooks/volatile.rs +++ /dev/null @@ -1,457 +0,0 @@ -//! Models which host-file bytes, names, and parent directories survive a crash. -//! -//! SQLite's VFS is deliberately outside this model; the tests below only -//! qualify LTX artifacts and checksum sidecars supplied by `FileSystem`. - -use super::*; -use std::collections::{HashMap, HashSet}; - -#[derive(Default)] -struct State { - next: u64, - live: HashMap, - durable: HashMap, - live_dirs: HashSet, - durable_dirs: HashSet, - synced: HashMap>, -} - -#[derive(Default)] -struct VolatileFs { - state: Arc>, - fail_parent: AtomicBool, - fail_parent_path: Mutex>, -} - -struct VolatileFile { - inner: Box, - id: Option, - path: PathBuf, - state: Arc>, -} - -impl VolatileFs { - fn tracked(path: &Path) -> bool { - let name = path.as_os_str().to_string_lossy(); - name.contains("-crab-ltx") || name.contains(".crab-ltx-") - } - - fn id(&self, path: &Path) -> Option { - if !Self::tracked(path) { - return None; - } - let mut state = self.state.lock().unwrap(); - if let Some(id) = state.live.get(path) { - return Some(*id); - } - state.next += 1; - let id = state.next; - state.live.insert(path.to_owned(), id); - Some(id) - } - - fn wrap(&self, path: &Path, inner: Box) -> Box { - Box::new(VolatileFile { - inner, - id: self.id(path), - path: path.to_owned(), - state: self.state.clone(), - }) - } - - fn moved(&self, from: &Path, to: &Path) { - let mut state = self.state.lock().unwrap(); - if let Some(id) = state.live.remove(from) { - state.live.insert(to.to_owned(), id); - } - } - - fn directory_barrier(&self, path: &Path) -> io::Result<()> { - if self.fail_parent.load(Ordering::Relaxed) - || self.fail_parent_path.lock().unwrap().as_deref() == Some(path) - { - return Err(io::Error::other("modeled parent sync failure")); - } - DirectFileSystem.sync_parent(path)?; - let parent = path.parent(); - let mut state = self.state.lock().unwrap(); - state.durable.retain(|name, _| name.parent() != parent); - let entries: Vec<_> = state - .live - .iter() - .filter(|(name, _)| name.parent() == parent) - .map(|(name, id)| (name.clone(), *id)) - .collect(); - state.durable.extend(entries); - state.durable_dirs.retain(|name| name.parent() != parent); - let directories: Vec<_> = state - .live_dirs - .iter() - .filter(|name| name.parent() == parent) - .cloned() - .collect(); - state.durable_dirs.extend(directories); - Ok(()) - } - - fn crash(&self) { - let mut state = self.state.lock().unwrap(); - let directory_survives = |directory: &Path| { - directory.ancestors().all(|ancestor| { - !state.live_dirs.contains(ancestor) || state.durable_dirs.contains(ancestor) - }) - }; - let surviving_dirs: Vec<_> = state - .durable_dirs - .iter() - .filter(|directory| directory_survives(directory)) - .cloned() - .collect(); - let surviving_files: Vec<_> = state - .durable - .iter() - .filter(|(path, _)| path.parent().is_some_and(directory_survives)) - .map(|(path, id)| (path.clone(), *id)) - .collect(); - let paths: HashSet<_> = state - .live - .keys() - .chain(state.durable.keys()) - .cloned() - .collect(); - for path in paths { - let _ = std::fs::remove_file(path); - } - let mut directories: Vec<_> = state.live_dirs.iter().collect(); - directories.sort_by_key(|path| std::cmp::Reverse(path.components().count())); - for directory in directories { - let _ = std::fs::remove_dir_all(directory); - } - for directory in &surviving_dirs { - std::fs::create_dir_all(directory).unwrap(); - } - for (path, id) in &surviving_files { - if let Some(bytes) = state.synced.get(id) { - std::fs::write(path, bytes).unwrap(); - } - } - state.live = surviving_files.into_iter().collect(); - state.live_dirs = surviving_dirs.into_iter().collect(); - } -} - -impl FileIo for VolatileFile { - fn write_all(&mut self, bytes: &[u8]) -> io::Result<()> { - self.inner.write_all(bytes) - } - fn write_all_at(&mut self, offset: u64, bytes: &[u8]) -> io::Result<()> { - self.inner.write_all_at(offset, bytes) - } - fn read_exact_at(&mut self, offset: u64, len: usize) -> io::Result> { - self.inner.read_exact_at(offset, len) - } - fn sync_all(&mut self) -> io::Result<()> { - self.inner.sync_all()?; - if let Some(id) = self.id { - let bytes = std::fs::read(&self.path)?; - self.state.lock().unwrap().synced.insert(id, bytes); - } - Ok(()) - } - fn file_len(&self) -> io::Result { - self.inner.file_len() - } - fn set_len(&mut self, len: u64) -> io::Result<()> { - self.inner.set_len(len) - } -} - -impl FileSystem for VolatileFs { - fn open(&self, path: &Path) -> io::Result> { - Ok(self.wrap(path, DirectFileSystem.open(path)?)) - } - fn open_rw(&self, path: &Path) -> io::Result> { - Ok(self.wrap(path, DirectFileSystem.open_rw(path)?)) - } - fn create(&self, path: &Path) -> io::Result> { - Ok(self.wrap(path, DirectFileSystem.create(path)?)) - } - fn file_len(&self, path: &Path) -> io::Result { - DirectFileSystem.file_len(path) - } - fn create_dir_all(&self, path: &Path) -> io::Result<()> { - let mut created = Vec::new(); - let mut ancestor = path; - while !DirectFileSystem.exists(ancestor)? { - created.push(ancestor.to_owned()); - ancestor = ancestor - .parent() - .ok_or_else(|| io::Error::other("missing directory parent"))?; - } - DirectFileSystem.create_dir_all(path)?; - self.state.lock().unwrap().live_dirs.extend(created); - Ok(()) - } - fn rename(&self, from: &Path, to: &Path) -> io::Result<()> { - DirectFileSystem.rename_uncommitted(from, to)?; - self.moved(from, to); - self.directory_barrier(to) - } - fn rename_uncommitted(&self, from: &Path, to: &Path) -> io::Result<()> { - DirectFileSystem.rename_uncommitted(from, to)?; - self.moved(from, to); - Ok(()) - } - fn remove_file(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.remove_file(path)?; - self.state.lock().unwrap().live.remove(path); - Ok(()) - } - fn canonicalize(&self, path: &Path) -> io::Result { - DirectFileSystem.canonicalize(path) - } - fn exists(&self, path: &Path) -> io::Result { - DirectFileSystem.exists(path) - } - fn create_dir(&self, path: &Path) -> io::Result<()> { - DirectFileSystem.create_dir(path)?; - self.state.lock().unwrap().live_dirs.insert(path.to_owned()); - Ok(()) - } - fn sync_parent(&self, path: &Path) -> io::Result<()> { - self.directory_barrier(path) - } - fn persist_new(&self, path: &Path, bytes: &[u8]) -> io::Result<()> { - DirectFileSystem.persist_new(path, bytes)?; - if let Some(id) = self.id(path) { - self.state.lock().unwrap().synced.insert(id, bytes.to_vec()); - } - self.directory_barrier(path) - } - fn persist_file_new(&self, source: &Path, destination: &Path) -> io::Result<()> { - DirectFileSystem.persist_file_new(source, destination)?; - self.moved(source, destination); - self.directory_barrier(destination) - } -} - -#[test] -fn only_synced_ltx_bytes_and_parent_synced_names_survive_modeled_crash() { - for barrier in ["none", "file", "file_parent", "ancestors"] { - let directory = tempfile::TempDir::new().unwrap(); - let fs = Arc::new(VolatileFs::default()); - let host = Host::default().with_filesystem(fs.clone()); - let path = directory.path().join("source.sqlite"); - let mut db = Db::open_with_host(&path, Limits::default(), host).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) - .unwrap(); - let cut = db.capture_deferred().unwrap(); - let segment = cut.segments[0].path().to_owned(); - if barrier != "none" { - fs.open_rw(&segment).unwrap().sync_all().unwrap(); - } - if barrier == "file_parent" || barrier == "ancestors" { - fs.sync_parent(&segment).unwrap(); - } - if barrier == "ancestors" { - let l0 = segment.parent().unwrap(); - let ltx = l0.parent().unwrap(); - let session = ltx.parent().unwrap(); - for directory in [l0, ltx, session] { - fs.sync_parent(directory).unwrap(); - } - } - drop(db); - fs.crash(); - if barrier == "ancestors" { - let plan = crab_ltx::VerifiedPlan::new(&cut.segments, cut.position, Limits::default()) - .unwrap(); - let restored = directory.path().join("restored.sqlite"); - crab_ltx::restore_exact(&plan, &restored).unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - let value: i64 = connection - .query_row("SELECT v FROM t", [], |row| row.get(0)) - .unwrap(); - assert_eq!(value, 7); - } else { - assert!( - !segment.exists(), - "{barrier} must not preserve the complete cut path" - ); - } - } -} - -#[test] -fn deferred_barrier_parent_failure_does_not_preserve_a_named_cut() { - let directory = tempfile::TempDir::new().unwrap(); - let fs = Arc::new(VolatileFs::default()); - let host = Host::default().with_filesystem(fs.clone()); - let mut db = Db::open_with_host( - &directory.path().join("source.sqlite"), - Limits::default(), - host, - ) - .unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) - .unwrap(); - let cut = db.capture_deferred().unwrap(); - let segment = cut.segments[0].path().to_owned(); - - fs.fail_parent.store(true, Ordering::Relaxed); - assert!(db.durability_barrier().is_err()); - assert!(matches!(db.capture(), Err(CrabError::Fenced))); - drop(db); - fs.fail_parent.store(false, Ordering::Relaxed); - fs.crash(); - assert!(!segment.exists()); -} - -#[test] -fn immediate_capture_survives_modeled_crash_at_its_return_boundary() { - let directory = tempfile::TempDir::new().unwrap(); - let fs = Arc::new(VolatileFs::default()); - let host = Host::default().with_filesystem(fs.clone()); - let mut db = Db::open_with_host( - &directory.path().join("source.sqlite"), - Limits::default(), - host, - ) - .unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) - .unwrap(); - let cut = db.capture().unwrap(); - drop(db); - fs.crash(); - - let plan = crab_ltx::VerifiedPlan::new(&cut.segments, cut.position, Limits::default()).unwrap(); - let restored = directory.path().join("restored.sqlite"); - crab_ltx::restore_exact(&plan, &restored).unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - let value: i64 = connection - .query_row("SELECT v FROM t", [], |row| row.get(0)) - .unwrap(); - assert_eq!(value, 7); -} - -#[test] -fn missing_session_directory_barrier_fences_local_acknowledgement() { - for deferred in [false, true] { - let directory = tempfile::TempDir::new().unwrap(); - let fs = Arc::new(VolatileFs::default()); - let host = Host::default().with_filesystem(fs.clone()); - let source = directory.path().join("source.sqlite"); - let mut db = Db::open_with_host(&source, Limits::default(), host).unwrap(); - let session = fs - .state - .lock() - .unwrap() - .live_dirs - .iter() - .find(|path| path.to_string_lossy().ends_with("-crab-ltx")) - .cloned() - .unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(7)")) - .unwrap(); - *fs.fail_parent_path.lock().unwrap() = Some(session.clone()); - if deferred { - db.capture_deferred().unwrap(); - assert!(db.durability_barrier().is_err()); - } else { - assert!(db.capture().is_err()); - } - assert!(matches!(db.capture(), Err(CrabError::Fenced))); - *fs.fail_parent_path.lock().unwrap() = None; - drop(db); - fs.crash(); - assert!(!session.exists(), "deferred={deferred}"); - } -} - -#[test] -fn deferred_batch_survives_only_after_its_shared_barrier() { - let directory = tempfile::TempDir::new().unwrap(); - let fs = Arc::new(VolatileFs::default()); - let host = Host::default().with_filesystem(fs.clone()); - let mut db = Db::open_with_host( - &directory.path().join("source.sqlite"), - Limits::default(), - host, - ) - .unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - let first = db.capture_deferred().unwrap(); - db.transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(2)")) - .unwrap(); - let second = db.capture_deferred().unwrap(); - db.durability_barrier().unwrap(); - drop(db); - fs.crash(); - - let mut segments = first.segments; - segments.extend(second.segments); - let plan = crab_ltx::VerifiedPlan::new(&segments, second.position, Limits::default()).unwrap(); - let restored = directory.path().join("restored.sqlite"); - crab_ltx::restore_exact(&plan, &restored).unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - let count: i64 = connection - .query_row("SELECT count(*) FROM t", [], |row| row.get(0)) - .unwrap(); - assert_eq!(count, 2); -} - -#[test] -fn checkpoint_restart_preserves_the_acknowledged_cut_chain() { - let directory = tempfile::TempDir::new().unwrap(); - let fs = Arc::new(VolatileFs::default()); - let host = Host::default().with_filesystem(fs.clone()); - let mut db = Db::open_with_host( - &directory.path().join("source.sqlite"), - Limits::default(), - host, - ) - .unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - let first = db.capture().unwrap(); - db.transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(2)")) - .unwrap(); - let second = db.checkpoint(CheckpointMode::Truncate).unwrap(); - drop(db); - fs.crash(); - - let mut segments = first.segments; - segments.extend(second.segments); - let plan = crab_ltx::VerifiedPlan::new(&segments, second.position, Limits::default()).unwrap(); - let restored = directory.path().join("restored.sqlite"); - crab_ltx::restore_exact(&plan, &restored).unwrap(); - let connection = crab_ltx::rusqlite::Connection::open(restored).unwrap(); - let count: i64 = connection - .query_row("SELECT count(*) FROM t", [], |row| row.get(0)) - .unwrap(); - assert_eq!(count, 2); -} - -#[test] -fn mutable_sidecar_and_unsynced_prune_revert_at_modeled_crash() { - let directory = tempfile::TempDir::new().unwrap(); - let fs = VolatileFs::default(); - let sidecar = directory.path().join("source.sqlite.crab-ltx-checksums"); - let mut file = fs.create(&sidecar).unwrap(); - file.write_all(&[1]).unwrap(); - file.sync_all().unwrap(); - fs.sync_parent(&sidecar).unwrap(); - file.write_all_at(0, &[2]).unwrap(); - drop(file); - fs.crash(); - assert_eq!(std::fs::read(&sidecar).unwrap(), vec![1]); - - fs.remove_file(&sidecar).unwrap(); - fs.crash(); - assert_eq!(std::fs::read(&sidecar).unwrap(), vec![1]); - fs.remove_file(&sidecar).unwrap(); - fs.sync_parent(&sidecar).unwrap(); - fs.crash(); - assert!(!sidecar.exists()); -} diff --git a/crates/crab-ltx/tests/ltx.rs b/crates/crab-ltx/tests/ltx.rs deleted file mode 100644 index 3cb0b3188..000000000 --- a/crates/crab-ltx/tests/ltx.rs +++ /dev/null @@ -1,10 +0,0 @@ -//! LTX capture, crash recovery, and node-frame tests. - -mod ltx { - pub mod crash; - pub mod node_frame; - pub mod pinned; - pub mod properties; - pub mod transactions; - pub mod vectors; -} diff --git a/crates/crab-ltx/tests/ltx/crash.rs b/crates/crab-ltx/tests/ltx/crash.rs deleted file mode 100644 index 7865f2449..000000000 --- a/crates/crab-ltx/tests/ltx/crash.rs +++ /dev/null @@ -1,198 +0,0 @@ -use crab_ltx::{Db, Limits, LocalSegment, SegmentInfo, VerifiedPlan, restore_exact}; -use std::io::{BufRead, Read, Write}; -use std::process::{Command, Stdio}; - -#[cfg(feature = "replica")] -#[test] -fn clean_continuation_writer() { - let Some(path) = std::env::var_os("CRAB_LTX_CLEAN_RESUME_TEST_DIR") else { - return; - }; - let path = std::path::PathBuf::from(path); - let mut db = Db::open(&path.join("source.sqlite"), Limits::default()).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - let batch = db.capture().unwrap(); - db.persist_continuation().unwrap(); - db.close().unwrap(); - std::fs::write( - path.join("position"), - format!("{} {}", batch.position.txid, batch.position.checksum), - ) - .unwrap(); -} - -#[cfg(feature = "replica")] -#[test] -fn clean_continuation_resumes_exact_chain_across_process_exit() { - use std::sync::Arc; - - use crab_ltx::{CellReplica, CellStorageLayout}; - use crab_storage::Store; - use object_store::{memory::InMemory, path::Path}; - - let directory = tempfile::TempDir::new().unwrap(); - let status = Command::new(std::env::current_exe().unwrap()) - .args(["ltx::crash::clean_continuation_writer", "--exact"]) - .env("CRAB_LTX_CLEAN_RESUME_TEST_DIR", directory.path()) - .status() - .unwrap(); - assert!(status.success()); - let recorded = std::fs::read_to_string(directory.path().join("position")).unwrap(); - let mut fields = recorded.split_whitespace(); - let txid: u64 = fields.next().unwrap().parse().unwrap(); - let checksum: u64 = fields.next().unwrap().parse().unwrap(); - let replica = CellReplica::new( - CellStorageLayout::new( - Store::new(Arc::new(InMemory::new())), - Path::from("test"), - [1; 16], - ), - [2; 32], - [3; 16], - Limits::default(), - ) - .unwrap(); - let mut resumed = replica - .open_resumed( - &directory.path().join("source.sqlite"), - &directory.path().join("resumed.sqlite"), - ) - .unwrap(); - assert_eq!(resumed.position().txid, txid); - assert_eq!(resumed.position().checksum, checksum); - let value: i64 = resumed - .query_with(|connection| connection.query_row("SELECT v FROM t", [], |row| row.get(0))) - .unwrap(); - assert_eq!(value, 1); - resumed - .transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(2)")) - .unwrap(); - let next = resumed.capture().unwrap(); - assert_eq!(next.position.txid, txid + 1); - assert_eq!(next.segments[0].info().pre_checksum, checksum); - resumed.close().unwrap(); -} - -// Test-harness entry point in a separate process. The parent kills it while the -// SQLite writer/read-lock connections are live; no orderly close is simulated. -#[test] -fn crash_writer() { - let Some(path) = std::env::var_os("CRAB_LTX_CRASH_TEST_DIR") else { - return; - }; - let path = std::path::PathBuf::from(path); - let mut db = Db::open(&path.join("repo.sqlite"), Limits::default()).unwrap(); - db.transaction(|tx| { - tx.execute("CREATE TABLE witnesses (value TEXT)", [])?; - tx.execute("INSERT INTO witnesses VALUES ('survived process kill')", [])?; - Ok(()) - }) - .unwrap(); - let batch = db.capture().unwrap(); - assert_eq!(batch.segments.len(), 1); - let segment = &batch.segments[0]; - std::fs::copy(segment.path(), path.join("captured.ltx")).unwrap(); - let info = segment.info(); - let digest: String = info - .blake3 - .iter() - .map(|byte| format!("{byte:02x}")) - .collect(); - println!( - "LTX-CUT {} {} {} {} {} {} {} {}", - info.min_txid, - info.max_txid, - info.page_size, - info.database_pages, - info.pre_checksum, - info.post_checksum, - info.size_bytes, - digest - ); - std::io::stdout().flush().unwrap(); - let mut byte = [0]; - std::io::stdin().read_exact(&mut byte).unwrap(); - panic!("parent must kill this process, not release stdin"); -} - -#[test] -fn captured_sql_survives_process_kill_and_source_directory_loss() { - let source = tempfile::TempDir::new().unwrap(); - let replica = tempfile::TempDir::new().unwrap(); - let restored = tempfile::TempDir::new().unwrap(); - let child = Command::new(std::env::current_exe().unwrap()) - .args(["crash_writer", "--nocapture"]) - .env("CRAB_LTX_CRASH_TEST_DIR", source.path()) - .stdin(Stdio::piped()) - .stdout(Stdio::piped()) - .stderr(Stdio::inherit()) - .spawn() - .unwrap(); - let mut child = KillOnDrop(child); - // Child::wait closes its stored stdin before reaping. Keep the pipe alive - // independently so an EOF panic cannot race the intended process kill. - let _stdin = child.0.stdin.take().unwrap(); - let stdout = child.0.stdout.take().unwrap(); - let (send, receive) = std::sync::mpsc::channel(); - std::thread::spawn(move || { - for line in std::io::BufReader::new(stdout).lines() { - let line = line.unwrap(); - if let Some(cut) = line.strip_prefix("LTX-CUT ") { - let _ = send.send(cut.to_owned()); - break; - } - } - }); - let cut = receive - .recv_timeout(std::time::Duration::from_secs(15)) - .unwrap(); - let fields: Vec<_> = cut.split_whitespace().collect(); - let mut digest = [0u8; 32]; - for (i, byte) in digest.iter_mut().enumerate() { - *byte = u8::from_str_radix(&fields[7][i * 2..i * 2 + 2], 16).unwrap(); - } - let info = SegmentInfo { - min_txid: fields[0].parse().unwrap(), - max_txid: fields[1].parse().unwrap(), - page_size: fields[2].parse().unwrap(), - database_pages: fields[3].parse().unwrap(), - pre_checksum: fields[4].parse().unwrap(), - post_checksum: fields[5].parse().unwrap(), - size_bytes: fields[6].parse().unwrap(), - blake3: digest, - }; - child.0.kill().unwrap(); - let exit = child.0.wait().unwrap(); - assert!(!exit.success()); - #[cfg(unix)] - { - use std::os::unix::process::ExitStatusExt; - assert_eq!(exit.signal(), Some(9)); - } - let path = replica.path().join("captured.ltx"); - std::fs::copy(source.path().join("captured.ltx"), &path).unwrap(); - source.close().unwrap(); - let position = info.position(); - let plan = VerifiedPlan::new( - &[LocalSegment::new(path, info)], - position, - Limits::default(), - ) - .unwrap(); - let destination = restored.path().join("repo.sqlite"); - restore_exact(&plan, &destination).unwrap(); - let conn = crab_ltx::rusqlite::Connection::open(destination).unwrap(); - let value: String = conn - .query_row("SELECT value FROM witnesses", [], |row| row.get(0)) - .unwrap(); - assert_eq!(value, "survived process kill"); -} - -struct KillOnDrop(std::process::Child); -impl Drop for KillOnDrop { - fn drop(&mut self) { - let _ = self.0.kill(); - let _ = self.0.wait(); - } -} diff --git a/crates/crab-ltx/tests/ltx/node_frame.rs b/crates/crab-ltx/tests/ltx/node_frame.rs deleted file mode 100644 index 5279ef7eb..000000000 --- a/crates/crab-ltx/tests/ltx/node_frame.rs +++ /dev/null @@ -1,90 +0,0 @@ -#![cfg(feature = "replica")] - -use bytes::Bytes; -use crab_ltx::{Db, Limits, NodeFrameScope, encode_node_frame, inspect_node_frame}; - -fn scope() -> NodeFrameScope { - NodeFrameScope { - leader_session: [1; 16], - log_epoch: 2, - node_sequence: 3, - application: [4; 16], - cell: [5; 32], - incarnation: [6; 16], - cell_epoch: 7, - commit_sequence: 8, - } -} - -#[test] -fn canonical_frame_roundtrips_verified_ltx() { - let directory = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&directory.path().join("frame.sqlite"), Limits::default()).unwrap(); - database - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE events(id INTEGER PRIMARY KEY, body TEXT NOT NULL);\ - INSERT INTO events(body) VALUES ('durable')", - ) - }) - .unwrap(); - let capture = database.capture().unwrap(); - let segment = capture.segments.first().unwrap(); - let body = Bytes::from(std::fs::read(segment.path()).unwrap()); - - let frame = encode_node_frame( - scope(), - segment.info().clone(), - body.clone(), - Limits::default(), - ) - .unwrap(); - let decoded = inspect_node_frame(frame.encoded().clone(), Limits::default()).unwrap(); - - assert_eq!(decoded.scope(), scope()); - assert_eq!(decoded.segment(), segment.info()); - assert_eq!(decoded.body(), &body); - assert_eq!(decoded.digest(), frame.digest()); - database.close().unwrap(); -} - -#[test] -fn frame_rejects_corruption_trailing_bytes_and_zero_scope() { - let directory = tempfile::TempDir::new().unwrap(); - let mut database = Db::open(&directory.path().join("frame.sqlite"), Limits::default()).unwrap(); - database - .transaction(|transaction| transaction.execute_batch("CREATE TABLE values_(v)")) - .unwrap(); - let capture = database.capture().unwrap(); - let segment = capture.segments.first().unwrap(); - let body = Bytes::from(std::fs::read(segment.path()).unwrap()); - let frame = encode_node_frame( - scope(), - segment.info().clone(), - body.clone(), - Limits::default(), - ) - .unwrap(); - - let mut corrupted = frame.encoded().to_vec(); - let last = corrupted.len() - 1; - corrupted[last] ^= 1; - assert!(inspect_node_frame(Bytes::from(corrupted), Limits::default()).is_err()); - - let mut trailing = frame.encoded().to_vec(); - trailing.push(0); - assert!(inspect_node_frame(Bytes::from(trailing), Limits::default()).is_err()); - - let mut invalid_scope = scope(); - invalid_scope.leader_session = [0; 16]; - assert!( - encode_node_frame( - invalid_scope, - segment.info().clone(), - body, - Limits::default(), - ) - .is_err() - ); - database.close().unwrap(); -} diff --git a/crates/crab-ltx/tests/ltx/pinned.rs b/crates/crab-ltx/tests/ltx/pinned.rs deleted file mode 100644 index b67249942..000000000 --- a/crates/crab-ltx/tests/ltx/pinned.rs +++ /dev/null @@ -1,100 +0,0 @@ -//! A reader that pins the WAL must not stall capture, fence the session, or -//! leave a chain that cannot be restored exactly. - -use crab_ltx::rusqlite::Connection; -use crab_ltx::{CheckpointMode, Db, Limits, VerifiedPlan, restore_exact}; - -fn wal_bytes(database: &std::path::Path) -> u64 { - let mut path = database.as_os_str().to_owned(); - path.push("-wal"); - std::fs::metadata(std::path::Path::new(&path)) - .map(|metadata| metadata.len()) - .unwrap_or(0) -} - -#[test] -fn a_pinned_wal_reader_keeps_capture_progress_and_restores_exactly() { - let temp = tempfile::TempDir::new().unwrap(); - let database = temp.path().join("pinned.sqlite"); - let limits = Limits::default(); - let mut db = Db::open(&database, limits).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v)")) - .unwrap(); - let mut segments = db.capture().unwrap().segments; - let pinned_wal = wal_bytes(&database); - - // A second connection holds a read mark, so SQLite cannot backfill the WAL - // past it and every checkpoint stays busy. - let reader = Connection::open(&database).unwrap(); - reader - .execute_batch("BEGIN; SELECT count(*) FROM t;") - .unwrap(); - - for _ in 0..64 { - db.transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(randomblob(4096))")) - .unwrap(); - segments.extend(db.capture().unwrap().segments); - } - assert!( - !db.has_pending_capture(), - "capture must keep sealing cuts while a reader pins the WAL" - ); - assert!( - wal_bytes(&database) > pinned_wal, - "a pinned checkpoint is expected to leave the WAL growing" - ); - - // A passive checkpoint reports its busy outcome instead of failing, so the - // session stays usable and the caller keeps every cut. - let batch = db.checkpoint(CheckpointMode::Passive).unwrap(); - segments.extend(batch.segments); - assert_eq!(batch.position, db.position()); - - // Releasing the reader lets a truncate checkpoint backfill the whole WAL. - reader.execute_batch("COMMIT").unwrap(); - drop(reader); - let batch = db.checkpoint(CheckpointMode::Truncate).unwrap(); - segments.extend(batch.segments); - let position = batch.position; - assert_eq!(position, db.position()); - db.close().unwrap(); - - // The complete chain still restores the exact database. - let plan = VerifiedPlan::new(&segments, position, limits).unwrap(); - let restored = temp.path().join("restored.sqlite"); - assert_eq!(restore_exact(&plan, &restored).unwrap(), position); - let connection = Connection::open(&restored).unwrap(); - let rows: i64 = connection - .query_row("SELECT count(*) FROM t", [], |row| row.get(0)) - .unwrap(); - assert_eq!(rows, 64); -} - -#[test] -fn checkpoint_returns_every_sealed_cut_without_reopening_it_during_collection() { - let temp = tempfile::TempDir::new().unwrap(); - let database = temp.path().join("checkpoint.sqlite"); - let limits = Limits::default(); - let mut db = Db::open(&database, limits).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v BLOB)")) - .unwrap(); - let first = db.capture().unwrap(); - db.transaction(|tx| tx.execute_batch("INSERT INTO t VALUES(randomblob(4096))")) - .unwrap(); - - let batch = db.checkpoint(CheckpointMode::Truncate).unwrap(); - assert_eq!(batch.segments.len(), 2); - assert_eq!(batch.timing.verification_nanos, 0); - let mut segments = first.segments; - segments.extend(batch.segments); - let plan = VerifiedPlan::new(&segments, batch.position, limits).unwrap(); - db.close().unwrap(); - - let restored = temp.path().join("checkpoint-restored.sqlite"); - restore_exact(&plan, &restored).unwrap(); - let connection = Connection::open(&restored).unwrap(); - let length: i64 = connection - .query_row("SELECT length(v) FROM t", [], |row| row.get(0)) - .unwrap(); - assert_eq!(length, 4096); -} diff --git a/crates/crab-ltx/tests/ltx/properties.proptest-regressions b/crates/crab-ltx/tests/ltx/properties.proptest-regressions deleted file mode 100644 index 2eaf8e156..000000000 --- a/crates/crab-ltx/tests/ltx/properties.proptest-regressions +++ /dev/null @@ -1,7 +0,0 @@ -# Seeds for failure cases proptest has generated in the past. It is -# automatically read and these particular cases re-run before any -# novel cases are generated. -# -# It is recommended to check this file in to source control so that -# everyone who runs the test benefits from these saved cases. -cc 7f306f0e286cf527cc4ac0797ffce5e2a69ce2be789086c53ecb73a98a07a9da # shrinks to steps = [Step { rows: 1, blob_bytes: 0, delete_rows: false }] diff --git a/crates/crab-ltx/tests/ltx/properties.rs b/crates/crab-ltx/tests/ltx/properties.rs deleted file mode 100644 index 7f8a1d24b..000000000 --- a/crates/crab-ltx/tests/ltx/properties.rs +++ /dev/null @@ -1,194 +0,0 @@ -//! Randomized capture, restore, and compaction properties. -//! -//! The restoration contract is that a verified plan reproduces exactly the -//! captured database image, and the repository invariant states it as -//! "reconstruction is byte-identical to original or returns an error". These -//! properties hold that over generated write shapes, so page sets, freelist -//! pages, overflow chains, and WAL cuts vary instead of matching one -//! hand-written fixture. Damage properties hold the other half: a torn segment -//! may never be accepted, however small the tear. - -use std::fs; - -use crab_ltx::rusqlite::{Result as SqlResult, Transaction}; -use crab_ltx::{Db, Limits, LocalSegment, Position, VerifiedPlan, compact_exact, restore_exact}; -use proptest::prelude::*; - -#[derive(Clone, Copy, Debug)] -struct Step { - rows: u8, - blob_bytes: u32, - delete_rows: bool, -} - -fn steps() -> impl Strategy> { - prop::collection::vec( - (1_u8..=4, 0_u32..=32 * 1024, any::()).prop_map(|(rows, blob_bytes, delete_rows)| { - Step { - rows, - blob_bytes, - delete_rows, - } - }), - 1..4, - ) -} - -#[derive(Clone, Copy, Debug)] -enum Damage { - /// Keeps only a prefix of the encoded segment, as a torn write would. - Truncate { keep: u32 }, - /// Flips one byte while keeping the length intact. - Flip { at: u32, mask: u8 }, -} - -impl Damage { - fn apply(self, bytes: &[u8]) -> Vec { - let length = bytes.len().max(1); - match self { - Damage::Truncate { keep } => bytes[..(keep as usize) % length].to_vec(), - Damage::Flip { at, mask } => { - let mut damaged = bytes.to_vec(); - if let Some(byte) = damaged.get_mut((at as usize) % length) { - *byte ^= mask | 1; - } - damaged - } - } - } -} - -fn damage_strategy() -> impl Strategy { - prop_oneof![ - any::().prop_map(|keep| Damage::Truncate { keep }), - (any::(), any::()).prop_map(|(at, mask)| Damage::Flip { at, mask }), - ] -} - -fn apply(transaction: &Transaction<'_>, step: Step) -> SqlResult<()> { - for _ in 0..step.rows { - transaction.execute( - "INSERT INTO payload(data) VALUES(randomblob(?1))", - [i64::from(step.blob_bytes)], - )?; - } - if step.delete_rows { - transaction.execute("DELETE FROM payload WHERE id % 3 = 0", [])?; - } - Ok(()) -} - -proptest! { - #![proptest_config(ProptestConfig { cases: 12, ..ProptestConfig::default() })] - - #[test] - fn capture_restore_and_compaction_reproduce_the_captured_image(steps in steps()) { - let directory = tempfile::TempDir::new().unwrap(); - let source = directory.path().join("source.sqlite"); - let mut writer = Db::open(&source, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(id INTEGER PRIMARY KEY, data BLOB NOT NULL)", - ) - }) - .unwrap(); - - // Keep the exact chain behind each cut, the way a published root carries - // the segments up to its captured position. - let mut cuts: Vec<(Vec, Position)> = Vec::new(); - let mut chain: Vec = Vec::new(); - for step in steps { - writer - .transaction(|transaction| apply(transaction, step)) - .unwrap(); - let batch = writer.capture().unwrap(); - chain.extend(batch.segments.iter().cloned()); - cuts.push((chain.clone(), batch.position)); - } - writer.close().unwrap(); - - let last_position = cuts.last().expect("at least one cut").1; - for (segments, position) in cuts { - let plan = VerifiedPlan::new(&segments, position, Limits::default()).unwrap(); - let direct = directory - .path() - .join(format!("direct-{}.sqlite", position.txid)); - let repeat = directory - .path() - .join(format!("repeat-{}.sqlite", position.txid)); - restore_exact(&plan, &direct).unwrap(); - restore_exact(&plan, &repeat).unwrap(); - prop_assert_eq!( - fs::read(&direct).unwrap(), - fs::read(&repeat).unwrap(), - "two restores of one plan differ" - ); - - let compacted = compact_exact( - &plan, - &directory.path().join(format!("snapshot-{}.ltx", position.txid)), - ) - .unwrap(); - let compact_plan = - VerifiedPlan::new(&[compacted], position, Limits::default()).unwrap(); - let from_compaction = directory - .path() - .join(format!("compact-{}.sqlite", position.txid)); - restore_exact(&compact_plan, &from_compaction).unwrap(); - prop_assert_eq!( - fs::read(&direct).unwrap(), - fs::read(&from_compaction).unwrap(), - "compaction changed the restored image" - ); - - // The last cut is the source database as it stands after close, so - // its restored image must be the original file byte for byte. - if position == last_position { - prop_assert_eq!( - fs::read(&direct).unwrap(), - fs::read(&source).unwrap(), - "restored image differs from the source database" - ); - } - } - } - - #[test] - fn torn_segment_bytes_are_never_accepted(damage in damage_strategy()) { - let directory = tempfile::TempDir::new().unwrap(); - let source = directory.path().join("source.sqlite"); - let mut writer = Db::open(&source, Limits::default()).unwrap(); - writer - .transaction(|transaction| { - transaction.execute_batch( - "CREATE TABLE payload(id INTEGER PRIMARY KEY, data BLOB NOT NULL)", - )?; - transaction.execute("INSERT INTO payload(data) VALUES(randomblob(4096))", [])?; - Ok(()) - }) - .unwrap(); - let batch = writer.capture().unwrap(); - writer.close().unwrap(); - let limits = Limits::default(); - // The pristine chain is the control: the harness must accept it. - prop_assert!(VerifiedPlan::new(&batch.segments, batch.position, limits).is_ok()); - - let victim = batch.segments.last().expect("capture produced a segment"); - let info = victim.info().clone(); - let bytes = fs::read(victim.path()).unwrap(); - let damaged = damage.apply(&bytes); - let damaged_path = directory.path().join("damaged.ltx"); - fs::write(&damaged_path, &damaged).unwrap(); - let mut chain = batch.segments.clone(); - let last = chain.len() - 1; - chain[last] = LocalSegment::new(damaged_path, info); - - prop_assert!( - VerifiedPlan::new(&chain, batch.position, limits).is_err(), - "a torn segment was accepted: {} of {} bytes", - damaged.len(), - bytes.len() - ); - } -} diff --git a/crates/crab-ltx/tests/ltx/transactions.rs b/crates/crab-ltx/tests/ltx/transactions.rs deleted file mode 100644 index 843cbbc5f..000000000 --- a/crates/crab-ltx/tests/ltx/transactions.rs +++ /dev/null @@ -1,117 +0,0 @@ -use crab_ltx::rusqlite::{Connection, ErrorCode}; -use crab_ltx::{Db, DiskBudget, Host, Limits, TransactionError, VerifiedPlan, restore_exact}; - -#[test] -fn automatic_rollback_preserves_committed_cut_and_writer() { - for (name, sql, code) in [ - ( - "full", - "INSERT INTO payloads VALUES(2, zeroblob(131072))", - ErrorCode::DiskFull, - ), - ( - "constraint", - "INSERT OR ROLLBACK INTO payloads VALUES(1, X'00')", - ErrorCode::ConstraintViolation, - ), - ] { - let directory = tempfile::tempdir().unwrap(); - let limits = Limits { - max_database_bytes: 64 * 1024, - max_capture_bytes: 128 * 1024, - ..Limits::default() - }; - let budget = DiskBudget::new(2 * 1024 * 1024); - let mut db = Db::open_with_host( - &directory.path().join("source.sqlite"), - limits, - Host::default().with_local_disk_budget(budget.clone()), - ) - .unwrap(); - db.transaction(|tx| { - tx.execute_batch( - "CREATE TABLE payloads(id INTEGER PRIMARY KEY, value BLOB); \ - INSERT INTO payloads VALUES(1, X'01')", - ) - }) - .unwrap(); - // Leave the prior commit uncaptured: automatic rollback must preserve - // that pending cut while discarding every write in the failed command. - let reserved = budget.used(); - let failed = db.transaction_with(|tx| { - tx.execute("UPDATE payloads SET value = X'02' WHERE id = 1", [])?; - let failure = tx.execute(sql, []).unwrap_err(); - assert!( - tx.is_autocommit(), - "{name} must roll back the whole transaction" - ); - Err::<(), _>(failure) - }); - assert!( - matches!(failed, Err(TransactionError::Operation(ref error)) - if error.sqlite_error_code() == Some(code)), - "{name}: {failed:?}" - ); - assert_eq!( - budget.used(), - reserved, - "{name}: rollback must refund write admission" - ); - let first = db.capture().unwrap(); - let plan = VerifiedPlan::new(&first.segments, first.position, limits).unwrap(); - let restored = directory.path().join("before.sqlite"); - restore_exact(&plan, &restored).unwrap(); - let connection = Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT hex(value) FROM payloads", [], |row| row - .get::<_, String>(0)) - .unwrap(), - "01", - "{name}" - ); - connection.close().unwrap(); - db.transaction(|tx| tx.execute("UPDATE payloads SET value = X'03' WHERE id = 1", [])) - .unwrap(); - let second = db.capture().unwrap(); - let segments = first - .segments - .into_iter() - .chain(second.segments) - .collect::>(); - let plan = VerifiedPlan::new(&segments, second.position, limits).unwrap(); - let restored = directory.path().join("after.sqlite"); - restore_exact(&plan, &restored).unwrap(); - let connection = Connection::open(restored).unwrap(); - assert_eq!( - connection - .query_row("SELECT hex(value) FROM payloads", [], |row| row - .get::<_, String>(0)) - .unwrap(), - "03", - "{name}" - ); - connection.close().unwrap(); - db.close().unwrap(); - assert_eq!(budget.used(), 0); - } -} - -#[test] -fn callback_commit_cannot_be_mistaken_for_automatic_rollback() { - let directory = tempfile::tempdir().unwrap(); - let mut db = Db::open(&directory.path().join("source.sqlite"), Limits::default()).unwrap(); - db.transaction(|tx| tx.execute_batch("CREATE TABLE t(v); INSERT INTO t VALUES(1)")) - .unwrap(); - db.capture().unwrap(); - // Transaction control violates the callback contract. Autocommit alone - // cannot prove rollback: the WAL observer must still fence an escaped commit. - let failed = db.transaction_with(|tx| { - tx.execute_batch("UPDATE t SET v = 2; COMMIT")?; - assert!(tx.is_autocommit()); - Err::<(), _>(crab_ltx::rusqlite::Error::InvalidQuery) - }); - assert!(matches!(failed, Err(TransactionError::Sqlite(_)))); - assert!(matches!(db.capture(), Err(crab_ltx::CrabError::Fenced))); - db.close().unwrap(); -} diff --git a/crates/crab-ltx/tests/ltx/vectors.rs b/crates/crab-ltx/tests/ltx/vectors.rs deleted file mode 100644 index 268c6f2bf..000000000 --- a/crates/crab-ltx/tests/ltx/vectors.rs +++ /dev/null @@ -1,82 +0,0 @@ -//! Deterministic decoder replay over the external vectors. -//! -//! This is the CI-runnable half of the fuzzing contract: every entry point -//! `fuzz/fuzz_targets` drives is exercised over truncations, deterministic bit -//! flips, and length prefixes of the upstream vectors. A panic fails the test, -//! so the decoders stay total on malformed input; deep random search stays in -//! the fuzz targets, which need a nightly toolchain. - -#[cfg(feature = "replica")] -use crab_ltx::Limits; - -/// External vectors written by the upstream celld encoder. -const VECTORS: &[&str] = &[ - "celld-10cb130-delta-2-2-512.ltx", - "celld-10cb130-snapshot-block-512.ltx", - "celld-10cb130-snapshot-frame-512.ltx", - "celld-10cb130-snapshot-block-4096.ltx", - "superfly-ltx-v0.5.2-delta-2-2-512.ltx", - "superfly-ltx-v0.5.2-snapshot-block-512.ltx", -]; - -fn vector_bytes(name: &str) -> Vec { - let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) - .join("tests/vectors") - .join(name); - std::fs::read(path).unwrap_or_else(|error| panic!("{name}: {error}")) -} - -/// Runs every inspection entry point over one input. -/// -/// None of them may panic, whatever the bytes are. -fn inspect_everything(bytes: &[u8]) { - #[cfg(feature = "replica")] - let limits = Limits::default(); - let _ = crab_ltx::internal::inspect_ltx(bytes); - #[cfg(feature = "replica")] - { - let _ = crab_ltx::internal::inspect_root(bytes); - let _ = crab_ltx::internal::inspect_segment_page(bytes); - let _ = crab_ltx::internal::inspect_directory_node(bytes); - let _ = crab_ltx::internal::inspect_bundle(bytes, limits); - let _ = crab_ltx::internal::inspect_node_frame(bytes, limits); - } -} - -#[test] -fn external_vectors_are_accepted() { - for name in VECTORS { - let bytes = vector_bytes(name); - assert!( - crab_ltx::internal::inspect_ltx(&bytes).is_ok(), - "{name} must decode" - ); - inspect_everything(&bytes); - } -} - -#[test] -fn truncated_and_mutated_vectors_never_panic() { - for name in VECTORS { - let bytes = vector_bytes(name); - for end in 0..bytes.len() { - inspect_everything(&bytes[..end]); - } - // Deterministic single-bit and single-byte mutations across the file, - // denser near the header and trailer where the parsers branch. - let mut offsets: Vec = (0..bytes.len().min(160)).collect(); - offsets.extend((160..bytes.len()).step_by(37)); - offsets.extend(bytes.len().saturating_sub(80)..bytes.len()); - for offset in offsets { - for delta in [1_u8, 0x80, 0xff] { - let mut mutated = bytes.clone(); - mutated[offset] ^= delta; - inspect_everything(&mutated); - } - } - // A length prefix must not be interpreted as a body. - let mut prefixed = (bytes.len() as u64).to_be_bytes().to_vec(); - prefixed.extend_from_slice(&bytes); - inspect_everything(&prefixed); - } -} diff --git a/crates/crab-ltx/tests/vectors/README.md b/crates/crab-ltx/tests/vectors/README.md deleted file mode 100644 index accee6576..000000000 --- a/crates/crab-ltx/tests/vectors/README.md +++ /dev/null @@ -1,64 +0,0 @@ -# External LTX vectors - -These files were written by upstream implementations, not by `crab-ltx`. They -are the independent half of the format qualification: `crab-ltx` must decode -them, restore the exact database image they encode, and (for the sized-block -representation) reproduce them byte for byte. - -## Provenance - -| Field | Value | -| --- | --- | -| Producers | `celld-ltx` at `10cb1303dac710dcb3b557e318e08c855261f68b` (the revision `UPSTREAM.md` pins), and `superfly/ltx` v0.5.2 — the Go reference library Litestream v0.5.17 uses | -| Encoders | `ltx::encode_file_v0_5_2` (sized block), `ltx::encode_file` (legacy LZ4 frame), and the Go `ltx.Encoder` | -| Generators | `generate/` (Rust, celld) and `generate/go/` (Go, superfly/ltx) | -| Pattern | page `p` byte `i` is `(p * 37 + i) % 251` | -| Header | version 3, `min_txid = max_txid = 1`, timestamp `1_700_000_000_000`, zero WAL fields | - -| File | page size | pages | encoding | -| --- | --- | --- | --- | -| `celld-10cb130-snapshot-block-512.ltx` | 512 | 3 | sized block | -| `celld-10cb130-snapshot-frame-512.ltx` | 512 | 3 | legacy LZ4 frame | -| `celld-10cb130-snapshot-block-4096.ltx` | 4096 | 4 | sized block | -| `superfly-ltx-v0.5.2-snapshot-block-512.ltx` | 512 | 3 | sized block, Go reference writer | -| `celld-10cb130-delta-2-2-512.ltx` | 512 | 1 page (2) | sized block successor, continues the 512 snapshot | -| `superfly-ltx-v0.5.2-delta-2-2-512.ltx` | 512 | 1 page (2) | sized block successor, Go reference writer | - -The reference writer and the Celld port produce byte-identical 512-byte files -(`sha256 f43c185e114e793fcf427f6f98002bfeb24dd8bfec8ca5b2c75c248f2598664c`), -which `src/format_tests.rs::independent_producers_agree_byte_for_byte` pins. The -byte-equality assertion against this crate's writer is therefore also the -reverse-direction proof: bytes every reader accepts are bytes this crate emits. -The two successor files are identical as well (`sha256 52144b14…`), and -`external_delta_chain_restores_and_matches_our_writer` restores the snapshot plus -successor chain to the exact image with the replaced page. - -The snapshot post-apply checksum is `CHECKSUM_FLAG | (checksum ^ page_checksum)` -folded over the pages, which is the recurrence both implementations decode. - -## Regenerate - -The generator is its own workspace so `cargo test -p crab-ltx` never fetches or -builds upstream sources: - -```sh -CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-ltx-vectors" \ - cargo run --release --manifest-path \ - crates/crab-ltx/tests/vectors/generate/Cargo.toml -- \ - crates/crab-ltx/tests/vectors -``` - -`src/format_tests.rs::external_snapshots_decode_restore_and_match_our_writer` -consumes these files, and `tests/ltx/vectors.rs` reuses them as the seed corpus -for the decoder-panic replay that mirrors `fuzz/`. - -The Go vector regenerates with: - -```sh -cd crates/crab-ltx/tests/vectors/generate/go -go run . ../.. -``` - -Adding a vector: extend the generator's case list, regenerate, and record the -new row above. Changing a vector's bytes without regenerating fails the -byte-equality assertion, which is the point. diff --git a/crates/crab-ltx/tests/vectors/celld-10cb130-delta-2-2-512.ltx b/crates/crab-ltx/tests/vectors/celld-10cb130-delta-2-2-512.ltx deleted file mode 100644 index f31c8be59..000000000 Binary files a/crates/crab-ltx/tests/vectors/celld-10cb130-delta-2-2-512.ltx and /dev/null differ diff --git a/crates/crab-ltx/tests/vectors/celld-10cb130-snapshot-block-4096.ltx b/crates/crab-ltx/tests/vectors/celld-10cb130-snapshot-block-4096.ltx deleted file mode 100644 index 3c612ddab..000000000 Binary files a/crates/crab-ltx/tests/vectors/celld-10cb130-snapshot-block-4096.ltx and /dev/null differ diff --git a/crates/crab-ltx/tests/vectors/celld-10cb130-snapshot-block-512.ltx b/crates/crab-ltx/tests/vectors/celld-10cb130-snapshot-block-512.ltx deleted file mode 100644 index ff1404cd1..000000000 Binary files a/crates/crab-ltx/tests/vectors/celld-10cb130-snapshot-block-512.ltx and /dev/null differ diff --git a/crates/crab-ltx/tests/vectors/celld-10cb130-snapshot-frame-512.ltx b/crates/crab-ltx/tests/vectors/celld-10cb130-snapshot-frame-512.ltx deleted file mode 100644 index 6036242d4..000000000 Binary files a/crates/crab-ltx/tests/vectors/celld-10cb130-snapshot-frame-512.ltx and /dev/null differ diff --git a/crates/crab-ltx/tests/vectors/generate/Cargo.lock b/crates/crab-ltx/tests/vectors/generate/Cargo.lock deleted file mode 100644 index 90fa0fb63..000000000 --- a/crates/crab-ltx/tests/vectors/generate/Cargo.lock +++ /dev/null @@ -1,1384 +0,0 @@ -# This file is automatically @generated by Cargo. -# It is not intended for manual editing. -version = 4 - -[[package]] -name = "adler2" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" - -[[package]] -name = "ahash" -version = "0.8.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a15f179cd60c4584b8a8c596927aadc462e27f2ca70c04e0071964a73ba7a75" -dependencies = [ - "cfg-if", - "once_cell", - "version_check", - "zerocopy", -] - -[[package]] -name = "android_system_properties" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" -dependencies = [ - "libc", -] - -[[package]] -name = "async-trait" -version = "0.1.92" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "autocfg" -version = "1.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" - -[[package]] -name = "base64" -version = "0.23.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" - -[[package]] -name = "bitflags" -version = "2.13.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" - -[[package]] -name = "bumpalo" -version = "3.20.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" - -[[package]] -name = "bytes" -version = "1.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" - -[[package]] -name = "cc" -version = "1.4.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "54413ede23c2daf518f35156dfde027feb2374004d63bd497f983c8db9c0e313" -dependencies = [ - "find-msvc-tools", - "shlex", -] - -[[package]] -name = "celld-ltx" -version = "0.0.0" -source = "git+https://github.com/denoland/celld.git?rev=10cb1303dac710dcb3b557e318e08c855261f68b#10cb1303dac710dcb3b557e318e08c855261f68b" -dependencies = [ - "async-trait", - "bytes", - "chrono", - "crc-fast", - "futures-util", - "http", - "lz4_flex", - "object_store", - "percent-encoding", - "rusqlite", - "thiserror", - "tokio", - "tracing", - "ureq", -] - -[[package]] -name = "cfg-if" -version = "1.0.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" - -[[package]] -name = "chrono" -version = "0.4.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" -dependencies = [ - "iana-time-zone", - "num-traits", - "windows-link", -] - -[[package]] -name = "core-foundation-sys" -version = "0.8.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" - -[[package]] -name = "crab-ltx-vector-generator" -version = "0.1.0" -dependencies = [ - "celld-ltx", -] - -[[package]] -name = "crc-fast" -version = "1.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e75b2483e97a5a7da73ac68a05b629f9c53cff58d8ed1c77866079e18b00dba5" -dependencies = [ - "digest", - "spin", -] - -[[package]] -name = "crc32fast" -version = "1.5.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78" -dependencies = [ - "cfg-if", -] - -[[package]] -name = "crypto-common" -version = "0.1.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" -dependencies = [ - "generic-array", - "typenum", -] - -[[package]] -name = "digest" -version = "0.10.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" -dependencies = [ - "crypto-common", -] - -[[package]] -name = "displaydoc" -version = "0.2.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "either" -version = "1.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" - -[[package]] -name = "fallible-iterator" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2acce4a10f12dc2fb14a218589d4f1f62ef011b2d0cc4b3cb1bba8e94da14649" - -[[package]] -name = "fallible-streaming-iterator" -version = "0.1.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a" - -[[package]] -name = "find-msvc-tools" -version = "0.1.13" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef25905e51abafe4dcea6c15fec58c57b601cdbd0ee53d22ea1d3016c587d39b" - -[[package]] -name = "flate2" -version = "1.1.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" -dependencies = [ - "crc32fast", - "miniz_oxide", - "zlib-rs", -] - -[[package]] -name = "form_urlencoded" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb4cb245038516f5f85277875cdaa4f7d2c9a0fa0468de06ed190163b1581fcf" -dependencies = [ - "percent-encoding", -] - -[[package]] -name = "futures" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a31d2a3fbaaeb2af2368bbdd904aa8e812d3c04a1ee10d3171f52d556e5d0a3" -dependencies = [ - "futures-channel", - "futures-core", - "futures-executor", - "futures-io", - "futures-sink", - "futures-task", - "futures-util", -] - -[[package]] -name = "futures-channel" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" -dependencies = [ - "futures-core", - "futures-sink", -] - -[[package]] -name = "futures-core" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" - -[[package]] -name = "futures-executor" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "031b47cf1a3c6cc8bc2fc76cd437f521619387907d469316e7c0bc278f1f5432" -dependencies = [ - "futures-core", - "futures-task", - "futures-util", -] - -[[package]] -name = "futures-io" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" - -[[package]] -name = "futures-macro" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "futures-sink" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" - -[[package]] -name = "futures-task" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" - -[[package]] -name = "futures-util" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" -dependencies = [ - "futures-channel", - "futures-core", - "futures-io", - "futures-macro", - "futures-sink", - "futures-task", - "memchr", - "pin-project-lite", - "slab", -] - -[[package]] -name = "generic-array" -version = "0.14.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" -dependencies = [ - "typenum", - "version_check", -] - -[[package]] -name = "getrandom" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0" -dependencies = [ - "cfg-if", - "libc", - "wasi", -] - -[[package]] -name = "hashbrown" -version = "0.14.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e5274423e17b7c9fc20b6e7e208532f9b19825d82dfd615708b70edd83df41f1" -dependencies = [ - "ahash", -] - -[[package]] -name = "hashlink" -version = "0.9.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ba4ff7128dee98c7dc9794b6a411377e1404dba1c97deb8d1a55297bd25d8af" -dependencies = [ - "hashbrown", -] - -[[package]] -name = "http" -version = "1.5.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" -dependencies = [ - "bytes", - "itoa", -] - -[[package]] -name = "httparse" -version = "1.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" - -[[package]] -name = "humantime" -version = "2.4.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" - -[[package]] -name = "iana-time-zone" -version = "0.1.65" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e31bc9ad994ba00e440a8aa5c9ef0ec67d5cb5e5cb0cc7f8b744a35b389cc470" -dependencies = [ - "android_system_properties", - "core-foundation-sys", - "iana-time-zone-haiku", - "js-sys", - "log", - "wasm-bindgen", - "windows-core", -] - -[[package]] -name = "iana-time-zone-haiku" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f31827a206f56af32e590ba56d5d2d085f558508192593743f16b2306495269f" -dependencies = [ - "cc", -] - -[[package]] -name = "icu_collections" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" -dependencies = [ - "displaydoc", - "potential_utf", - "utf8_iter", - "yoke", - "zerofrom", - "zerovec", -] - -[[package]] -name = "icu_locale_core" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" -dependencies = [ - "displaydoc", - "litemap", - "tinystr", - "writeable", - "zerovec", -] - -[[package]] -name = "icu_normalizer" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" -dependencies = [ - "icu_collections", - "icu_normalizer_data", - "icu_properties", - "icu_provider", - "smallvec", - "zerovec", -] - -[[package]] -name = "icu_normalizer_data" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" - -[[package]] -name = "icu_properties" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" -dependencies = [ - "displaydoc", - "icu_collections", - "icu_locale_core", - "icu_properties_data", - "icu_provider", - "zerotrie", - "zerovec", -] - -[[package]] -name = "icu_properties_data" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" - -[[package]] -name = "icu_provider" -version = "2.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" -dependencies = [ - "displaydoc", - "icu_locale_core", - "writeable", - "yoke", - "zerofrom", - "zerotrie", - "zerovec", -] - -[[package]] -name = "idna" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3b0875f23caa03898994f6ddc501886a45c7d3d62d04d2d90788d47be1b1e4de" -dependencies = [ - "idna_adapter", - "smallvec", - "utf8_iter", -] - -[[package]] -name = "idna_adapter" -version = "1.2.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb68373c0d6620ef8105e855e7745e18b0d00d3bdb07fb532e434244cdb9a714" -dependencies = [ - "icu_normalizer", - "icu_properties", -] - -[[package]] -name = "itertools" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2b192c782037fadd9cfa75548310488aabdbf3d2da73885b31bd0abd03351285" -dependencies = [ - "either", -] - -[[package]] -name = "itoa" -version = "1.0.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" - -[[package]] -name = "js-sys" -version = "0.3.105" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce57d20d1ea864ce2ac172ab472d409214f4fd359f0b2a2775abdf522e2af99e" -dependencies = [ - "cfg-if", - "futures-util", - "wasm-bindgen", -] - -[[package]] -name = "libc" -version = "0.2.189" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" - -[[package]] -name = "libsqlite3-sys" -version = "0.28.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c10584274047cb335c23d3e61bcef8e323adae7c5c8c760540f73610177fc3f" -dependencies = [ - "cc", - "pkg-config", - "vcpkg", -] - -[[package]] -name = "litemap" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" - -[[package]] -name = "lock_api" -version = "0.4.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "224399e74b87b5f3557511d98dff8b14089b3dadafcab6bb93eab67d3aace965" -dependencies = [ - "scopeguard", -] - -[[package]] -name = "log" -version = "0.4.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" - -[[package]] -name = "lz4_flex" -version = "0.11.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a" -dependencies = [ - "twox-hash", -] - -[[package]] -name = "memchr" -version = "2.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" - -[[package]] -name = "miniz_oxide" -version = "0.9.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" -dependencies = [ - "adler2", - "simd-adler32", -] - -[[package]] -name = "num-traits" -version = "0.2.19" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841" -dependencies = [ - "autocfg", -] - -[[package]] -name = "object_store" -version = "0.12.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fbfbfff40aeccab00ec8a910b57ca8ecf4319b335c542f2edcd19dd25a1e2a00" -dependencies = [ - "async-trait", - "bytes", - "chrono", - "futures", - "http", - "humantime", - "itertools", - "parking_lot", - "percent-encoding", - "thiserror", - "tokio", - "tracing", - "url", - "wasm-bindgen-futures", - "web-time", -] - -[[package]] -name = "once_cell" -version = "1.21.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" - -[[package]] -name = "parking_lot" -version = "0.12.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "93857453250e3077bd71ff98b6a65ea6621a19bb0f559a85248955ac12c45a1a" -dependencies = [ - "lock_api", - "parking_lot_core", -] - -[[package]] -name = "parking_lot_core" -version = "0.9.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2621685985a2ebf1c516881c026032ac7deafcda1a2c9b7850dc81e3dfcb64c1" -dependencies = [ - "cfg-if", - "libc", - "redox_syscall", - "smallvec", - "windows-link", -] - -[[package]] -name = "percent-encoding" -version = "2.3.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" - -[[package]] -name = "pin-project-lite" -version = "0.2.17" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" - -[[package]] -name = "pkg-config" -version = "0.3.34" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" - -[[package]] -name = "potential_utf" -version = "0.1.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" -dependencies = [ - "zerovec", -] - -[[package]] -name = "proc-macro2" -version = "1.0.107" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "quote" -version = "1.0.47" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" -dependencies = [ - "proc-macro2", -] - -[[package]] -name = "redox_syscall" -version = "0.5.18" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" -dependencies = [ - "bitflags", -] - -[[package]] -name = "ring" -version = "0.17.14" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7" -dependencies = [ - "cc", - "cfg-if", - "getrandom", - "libc", - "untrusted", - "windows-sys", -] - -[[package]] -name = "rusqlite" -version = "0.31.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b838eba278d213a8beaf485bd313fd580ca4505a00d5871caeb1457c55322cae" -dependencies = [ - "bitflags", - "fallible-iterator", - "fallible-streaming-iterator", - "hashlink", - "libsqlite3-sys", - "smallvec", -] - -[[package]] -name = "rustls" -version = "0.23.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d41d731c7d2f962d1ccc364cec258de3c0e93b38c2fb3ba97ac74513048d634" -dependencies = [ - "log", - "once_cell", - "ring", - "rustls-pki-types", - "rustls-webpki", - "subtle", - "zeroize", -] - -[[package]] -name = "rustls-pki-types" -version = "1.15.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" -dependencies = [ - "zeroize", -] - -[[package]] -name = "rustls-webpki" -version = "0.103.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" -dependencies = [ - "ring", - "rustls-pki-types", - "untrusted", -] - -[[package]] -name = "rustversion" -version = "1.0.23" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" - -[[package]] -name = "scopeguard" -version = "1.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" - -[[package]] -name = "serde" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" -dependencies = [ - "serde_core", -] - -[[package]] -name = "serde_core" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" -dependencies = [ - "serde_derive", -] - -[[package]] -name = "serde_derive" -version = "1.0.229" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "shlex" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" - -[[package]] -name = "simd-adler32" -version = "0.3.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" - -[[package]] -name = "slab" -version = "0.4.12" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" - -[[package]] -name = "smallvec" -version = "1.16.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba467056f1b547ed52077911161fc86985becbc60e8e1857c8a144dab0def891" - -[[package]] -name = "spin" -version = "0.10.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" - -[[package]] -name = "stable_deref_trait" -version = "1.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" - -[[package]] -name = "subtle" -version = "2.6.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" - -[[package]] -name = "syn" -version = "2.0.119" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "syn" -version = "3.0.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" -dependencies = [ - "proc-macro2", - "quote", - "unicode-ident", -] - -[[package]] -name = "synstructure" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "901704edd0dfe137f1987838ee4f259e4e063c31371bdb423f7ae38ec6f77f02" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "thiserror" -version = "2.0.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09e52cb86a36cede5cb101bf8908837b3e4c6e5e59fe7fd85c23fb56200d189e" -dependencies = [ - "thiserror-impl", -] - -[[package]] -name = "thiserror-impl" -version = "2.0.21" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fe5197923287db20a58125f0bc85c062f7f2c892de97b18c356f9efb14b28524" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tinystr" -version = "0.8.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" -dependencies = [ - "displaydoc", - "zerovec", -] - -[[package]] -name = "tokio" -version = "1.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" -dependencies = [ - "bytes", - "pin-project-lite", - "tokio-macros", -] - -[[package]] -name = "tokio-macros" -version = "2.7.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "tracing" -version = "0.1.44" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "63e71662fa4b2a2c3a26f570f037eb95bb1f85397f3cd8076caed2f026a6d100" -dependencies = [ - "pin-project-lite", - "tracing-attributes", - "tracing-core", -] - -[[package]] -name = "tracing-attributes" -version = "0.1.31" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "tracing-core" -version = "0.1.36" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db97caf9d906fbde555dd62fa95ddba9eecfd14cb388e4f491a66d74cd5fb79a" -dependencies = [ - "once_cell", -] - -[[package]] -name = "twox-hash" -version = "2.1.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" - -[[package]] -name = "typenum" -version = "1.20.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" - -[[package]] -name = "unicode-ident" -version = "1.0.26" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" - -[[package]] -name = "untrusted" -version = "0.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" - -[[package]] -name = "ureq" -version = "3.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a7ac20be9b7726e0bbdbf974c059676d9acb1cd414961f570a4e8231cacd7fc" -dependencies = [ - "base64", - "flate2", - "log", - "percent-encoding", - "rustls", - "rustls-pki-types", - "ureq-proto", - "utf8-zero", - "webpki-roots", -] - -[[package]] -name = "ureq-proto" -version = "0.6.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f86fd172ccca569e458f61b6bdd6220965a9ef36e672a6852953b51a0e1583be" -dependencies = [ - "base64", - "http", - "httparse", - "log", -] - -[[package]] -name = "url" -version = "2.5.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff67a8a4397373c3ef660812acab3268222035010ab8680ec4215f38ba3d0eed" -dependencies = [ - "form_urlencoded", - "idna", - "percent-encoding", - "serde", -] - -[[package]] -name = "utf8-zero" -version = "0.8.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8c0a043c9540bae7c578c88f91dda8bd82e59ae27c21baca69c8b191aaf5a6e" - -[[package]] -name = "utf8_iter" -version = "1.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" - -[[package]] -name = "vcpkg" -version = "0.2.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426" - -[[package]] -name = "version_check" -version = "0.9.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" - -[[package]] -name = "wasi" -version = "0.11.1+wasi-snapshot-preview1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" - -[[package]] -name = "wasm-bindgen" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "aecb87a33d3b0c5e3b7aa46336eaf486cffafbd281b195e4c8b80d50df2351bf" -dependencies = [ - "cfg-if", - "once_cell", - "rustversion", - "wasm-bindgen-macro", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-futures" -version = "0.4.78" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ef4c5d3d2cdf5c54f4231181768f5510842e350db025faf1f7163b1030ed928" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "wasm-bindgen-macro" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a690d511e3c1a8b3a55e33511e3c2c00c78415cd23650f32b808627f5696b9ed" -dependencies = [ - "quote", - "wasm-bindgen-macro-support", -] - -[[package]] -name = "wasm-bindgen-macro-support" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "411e4887f0071ef2d2164a9d5fdf2d20efbef78fccd3a78b0c10a1dc5295e48a" -dependencies = [ - "bumpalo", - "proc-macro2", - "quote", - "syn 3.0.6", - "wasm-bindgen-shared", -] - -[[package]] -name = "wasm-bindgen-shared" -version = "0.2.128" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "81941cd78d0c92026c33e5e01312845a4cb1e9af3407f9134b100dd03144103e" -dependencies = [ - "unicode-ident", -] - -[[package]] -name = "web-time" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a6580f308b1fad9207618087a65c04e7a10bc77e02c8e84e9b00dd4b12fa0bb" -dependencies = [ - "js-sys", - "wasm-bindgen", -] - -[[package]] -name = "webpki-roots" -version = "1.0.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7dcd9d09a39985f5344844e66b0c530a33843579125f23e21e9f0f220850f22a" -dependencies = [ - "rustls-pki-types", -] - -[[package]] -name = "windows-core" -version = "0.62.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" -dependencies = [ - "windows-implement", - "windows-interface", - "windows-link", - "windows-result", - "windows-strings", -] - -[[package]] -name = "windows-implement" -version = "0.60.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "windows-interface" -version = "0.59.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "windows-link" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" - -[[package]] -name = "windows-result" -version = "0.4.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-strings" -version = "0.5.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" -dependencies = [ - "windows-link", -] - -[[package]] -name = "windows-sys" -version = "0.52.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" -dependencies = [ - "windows-targets", -] - -[[package]] -name = "windows-targets" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" -dependencies = [ - "windows_aarch64_gnullvm", - "windows_aarch64_msvc", - "windows_i686_gnu", - "windows_i686_gnullvm", - "windows_i686_msvc", - "windows_x86_64_gnu", - "windows_x86_64_gnullvm", - "windows_x86_64_msvc", -] - -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" - -[[package]] -name = "windows_aarch64_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" - -[[package]] -name = "windows_i686_gnu" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" - -[[package]] -name = "windows_i686_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" - -[[package]] -name = "windows_i686_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" - -[[package]] -name = "windows_x86_64_gnu" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" - -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" - -[[package]] -name = "windows_x86_64_msvc" -version = "0.52.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" - -[[package]] -name = "writeable" -version = "0.6.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" - -[[package]] -name = "yoke" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" -dependencies = [ - "stable_deref_trait", - "yoke-derive", - "zerofrom", -] - -[[package]] -name = "yoke-derive" -version = "0.8.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "33811428bee40dbceb6d545e95754741d17a6aef9a4849f0fd62e2ba4f412a78" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", - "synstructure", -] - -[[package]] -name = "zerocopy" -version = "0.8.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c17e8fafad82b542ff3717217ecdc736231b59e387768c9630123b4ce4d2db44" -dependencies = [ - "zerocopy-derive", -] - -[[package]] -name = "zerocopy-derive" -version = "0.8.58" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "595f56e044df4f46a0c9a626f65c3d99eb8488f7e8a8baa12dd76326d9710bf2" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - -[[package]] -name = "zerofrom" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" -dependencies = [ - "zerofrom-derive", -] - -[[package]] -name = "zerofrom-derive" -version = "0.1.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f75b4683f6c7f45248d4d64056a24298c6281e0993356d7d1b4a1a962ef10d4a" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", - "synstructure", -] - -[[package]] -name = "zeroize" -version = "1.9.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" - -[[package]] -name = "zerotrie" -version = "0.2.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" -dependencies = [ - "displaydoc", - "yoke", - "zerofrom", -] - -[[package]] -name = "zerovec" -version = "0.11.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" -dependencies = [ - "yoke", - "zerofrom", - "zerovec-derive", -] - -[[package]] -name = "zerovec-derive" -version = "0.11.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" -dependencies = [ - "proc-macro2", - "quote", - "syn 3.0.6", -] - -[[package]] -name = "zlib-rs" -version = "0.6.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112" diff --git a/crates/crab-ltx/tests/vectors/generate/Cargo.toml b/crates/crab-ltx/tests/vectors/generate/Cargo.toml deleted file mode 100644 index 9685bf773..000000000 --- a/crates/crab-ltx/tests/vectors/generate/Cargo.toml +++ /dev/null @@ -1,16 +0,0 @@ -# Regenerates the external LTX vectors under `tests/vectors/`. -# -# The generator depends on the upstream snapshot `crab-ltx` was imported from -# (see UPSTREAM.md), so the bytes come from an independent implementation. It is -# its own workspace on purpose: `cargo test -p crab-ltx` must not build a second -# SQLite or fetch upstream sources. -[package] -name = "crab-ltx-vector-generator" -version = "0.1.0" -edition = "2021" -publish = false - -[workspace] - -[dependencies] -celld-ltx = { git = "https://github.com/denoland/celld.git", rev = "10cb1303dac710dcb3b557e318e08c855261f68b", default-features = false } diff --git a/crates/crab-ltx/tests/vectors/generate/go/go.mod b/crates/crab-ltx/tests/vectors/generate/go/go.mod deleted file mode 100644 index 0e1212e35..000000000 --- a/crates/crab-ltx/tests/vectors/generate/go/go.mod +++ /dev/null @@ -1,7 +0,0 @@ -module crab-ltx-go-vector-generator - -go 1.24 - -require github.com/superfly/ltx v0.5.2 - -require github.com/pierrec/lz4/v4 v4.1.23 // indirect diff --git a/crates/crab-ltx/tests/vectors/generate/go/go.sum b/crates/crab-ltx/tests/vectors/generate/go/go.sum deleted file mode 100644 index 338cab443..000000000 --- a/crates/crab-ltx/tests/vectors/generate/go/go.sum +++ /dev/null @@ -1,4 +0,0 @@ -github.com/pierrec/lz4/v4 v4.1.23 h1:oJE7T90aYBGtFNrI8+KbETnPymobAhzRrR8Mu8n1yfU= -github.com/pierrec/lz4/v4 v4.1.23/go.mod h1:EoQMVJgeeEOMsCqCzqFm2O0cJvljX2nGZjcRIPL34O4= -github.com/superfly/ltx v0.5.2 h1:XVzytSsIlpUaOd2I7M40lDhUQIzGIdqkLBNg5M5Ax48= -github.com/superfly/ltx v0.5.2/go.mod h1:0OtLSLHHHPa3qDrAmwkLW5pUBfj365JS+bR9FEJDrM4= diff --git a/crates/crab-ltx/tests/vectors/generate/go/main.go b/crates/crab-ltx/tests/vectors/generate/go/main.go deleted file mode 100644 index 89344f63c..000000000 --- a/crates/crab-ltx/tests/vectors/generate/go/main.go +++ /dev/null @@ -1,157 +0,0 @@ -// Writes the superfly/ltx reference vector with the same library Litestream -// uses (v0.5.2). Run from this directory: -// -// go run . ../../ (relative output directory) -// -// See ../vectors/README.md for provenance and the regeneration command. -package main - -import ( - "bytes" - "crypto/sha256" - "fmt" - "os" - "path/filepath" - - "github.com/superfly/ltx" -) - -// page mirrors the byte pattern the Crab vector test expects. -func page(pageSize uint32, pgno uint32) []byte { - data := make([]byte, pageSize) - for index := range data { - data[index] = byte((pgno*37 + uint32(index)) % 251) - } - return data -} - -// deltaPage is the replacement body for page 2 in the delta vector. -func deltaPage(pageSize uint32, pgno uint32) []byte { - data := page(pageSize, pgno) - for index := range data { - data[index] = 255 - data[index] - } - return data -} - -func main() { - if len(os.Args) != 2 { - fmt.Fprintln(os.Stderr, "usage: go run . ") - os.Exit(2) - } - const pageSize = 512 - const commit = 3 - writeSnapshot(os.Args[1], pageSize, commit) - writeDelta(os.Args[1], pageSize, commit) -} - -func writeSnapshot(output string, pageSize uint32, commit uint32) { - path := filepath.Join(output, "superfly-ltx-v0.5.2-snapshot-block-512.ltx") - file, err := os.Create(path) - if err != nil { - panic(err) - } - encoder, err := ltx.NewEncoder(file) - if err != nil { - panic(err) - } - - var image bytes.Buffer - for pgno := uint32(1); pgno <= commit; pgno++ { - image.Write(page(pageSize, pgno)) - } - // The reference implementation computes the rolling database checksum; a - // caller hands it to the encoder instead of deriving it by hand. - checksum, err := ltx.ChecksumReader(bytes.NewReader(image.Bytes()), int(pageSize)) - if err != nil { - panic(err) - } - - header := ltx.Header{ - Version: 3, - Flags: 0, - PageSize: pageSize, - Commit: commit, - MinTXID: 1, - MaxTXID: 1, - Timestamp: 1700000000000, - PreApplyChecksum: 0, - } - if err := encoder.EncodeHeader(header); err != nil { - panic(err) - } - for pgno := uint32(1); pgno <= commit; pgno++ { - if err := encoder.EncodePage(ltx.PageHeader{Pgno: pgno}, page(pageSize, pgno)); err != nil { - panic(err) - } - } - encoder.SetPostApplyChecksum(checksum) - if err := encoder.Close(); err != nil { - panic(err) - } - if err := file.Close(); err != nil { - panic(err) - } - bytes, err := os.ReadFile(path) - if err != nil { - panic(err) - } - fmt.Printf("%s: %d bytes checksum=%#x sha256=%x\n", - path, len(bytes), uint64(checksum), sha256.Sum256(bytes)) -} - -// writeDelta emits one reference-encoded successor file: it replaces page 2 of -// the snapshot and publishes the resulting database checksum. -func writeDelta(output string, pageSize uint32, commit uint32) { - old := page(pageSize, 2) - next := deltaPage(pageSize, 2) - image := append(append(page(pageSize, 1), old...), page(pageSize, 3)...) - pre, err := ltx.ChecksumReader(bytes.NewReader(image), int(pageSize)) - if err != nil { - panic(err) - } - // The successor checksum removes the replaced page and adds its replacement. - post := ltx.Checksum( - uint64(ltx.ChecksumFlag) | - (uint64(pre) ^ uint64(ltx.ChecksumPage(2, old)) ^ uint64(ltx.ChecksumPage(2, next))), - ) - - path := filepath.Join(output, "superfly-ltx-v0.5.2-delta-2-2-512.ltx") - file, err := os.Create(path) - if err != nil { - panic(err) - } - encoder, err := ltx.NewEncoder(file) - if err != nil { - panic(err) - } - header := ltx.Header{ - Version: 3, - Flags: 0, - PageSize: pageSize, - Commit: commit, - MinTXID: 2, - MaxTXID: 2, - Timestamp: 1700000000000, - PreApplyChecksum: pre, - } - if err := encoder.EncodeHeader(header); err != nil { - panic(err) - } - if err := encoder.EncodePage(ltx.PageHeader{Pgno: 2}, next); err != nil { - panic(err) - } - encoder.SetPostApplyChecksum(post) - if err := encoder.Close(); err != nil { - panic(err) - } - if err := file.Close(); err != nil { - panic(err) - } - bytes, err := os.ReadFile(path) - if err != nil { - panic(err) - } - fmt.Printf("%s: %d bytes pre=%#x post=%#x sha256=%x\n", - path, len(bytes), uint64(pre), uint64(post), sha256.Sum256(bytes)) -} diff --git a/crates/crab-ltx/tests/vectors/generate/src/main.rs b/crates/crab-ltx/tests/vectors/generate/src/main.rs deleted file mode 100644 index 466f7fc71..000000000 --- a/crates/crab-ltx/tests/vectors/generate/src/main.rs +++ /dev/null @@ -1,132 +0,0 @@ -//! Writes the external LTX vectors `crates/crab-ltx/tests/vectors/` holds. -//! -//! Every byte comes from the upstream celld encoder pinned in `UPSTREAM.md`, so -//! the Crab reader is proven against an independent implementation rather than -//! against itself. The same inputs are reproducible: the page pattern and the -//! header are fixed here, and the snapshot checksum is computed the way both -//! implementations define it. -//! -//! Run with a target directory outside the repository: -//! -//! ```sh -//! CARGO_TARGET_DIR="$HOME/Workspace/crabbuild-target/crab-ltx-vectors" \ -//! cargo run --release --manifest-path \ -//! crates/crab-ltx/tests/vectors/generate/Cargo.toml -- \ -//! crates/crab-ltx/tests/vectors -//! ``` - -use celld_ltx::ltx::{self, Header, encode_file, encode_file_v0_5_2}; -use celld_ltx::{CHECKSUM_FLAG, TXID}; -use std::path::PathBuf; - -/// Deterministic page body shared with the Crab vector test. -fn page(page_size: u32, pgno: u32) -> Vec { - (0..page_size) - .map(|index| ((pgno * 37 + index) % 251) as u8) - .collect() -} - -fn main() -> Result<(), Box> { - let output = PathBuf::from( - std::env::args() - .nth(1) - .ok_or("usage: vector-generator ")?, - ); - std::fs::create_dir_all(&output)?; - - let cases = [ - ("celld-10cb130-snapshot-block-512.ltx", 512_u32, 3_u32, true), - ("celld-10cb130-snapshot-frame-512.ltx", 512, 3, false), - ("celld-10cb130-snapshot-block-4096.ltx", 4096, 4, true), - ]; - for (name, page_size, commit, block) in cases { - let pages: Vec<(u32, Vec)> = (1..=commit) - .map(|pgno| (pgno, page(page_size, pgno))) - .collect(); - // A snapshot's post-apply checksum is the checksum-flag seed folded with - // every page checksum, which is exactly what both decoders recompute. - let mut post_apply_checksum = CHECKSUM_FLAG; - for (pgno, data) in &pages { - post_apply_checksum = - CHECKSUM_FLAG | (post_apply_checksum ^ ltx::checksum_page(*pgno, data)); - } - let header = Header { - version: 3, - flags: 0, - page_size, - commit, - min_txid: TXID(1), - max_txid: TXID(1), - timestamp: 1_700_000_000_000, - pre_apply_checksum: 0, - wal_offset: 0, - wal_size: 0, - wal_salt1: 0, - wal_salt2: 0, - node_id: 0, - }; - let bytes = if block { - encode_file_v0_5_2(&header, &pages, post_apply_checksum) - } else { - encode_file(&header, &pages, post_apply_checksum) - } - .map_err(|error| format!("{name}: {error:?}"))?; - std::fs::write(output.join(name), &bytes)?; - println!( - "{name}: {} bytes, page_size={page_size}, commit={commit}, post={post_apply_checksum:#x}", - bytes.len() - ); - } - write_delta(&output)?; - Ok(()) -} - -/// Writes one reference-encoded successor file for the 512-byte snapshot. -/// -/// The delta replaces page 2 and publishes the resulting database checksum: -/// the replaced page leaves the rolling checksum and its replacement enters it. -fn write_delta(output: &PathBuf) -> Result<(), Box> { - const PAGE_SIZE: u32 = 512; - const COMMIT: u32 = 3; - let old = page(PAGE_SIZE, 2); - let next: Vec = page(PAGE_SIZE, 2) - .into_iter() - .map(|byte| 255 - byte) - .collect(); - let snapshot_pages: Vec<(u32, Vec)> = (1..=COMMIT) - .map(|pgno| (pgno, page(PAGE_SIZE, pgno))) - .collect(); - let mut pre_apply_checksum = CHECKSUM_FLAG; - for (pgno, data) in &snapshot_pages { - pre_apply_checksum = - CHECKSUM_FLAG | (pre_apply_checksum ^ ltx::checksum_page(*pgno, data)); - } - let post_apply_checksum = CHECKSUM_FLAG - | (pre_apply_checksum - ^ ltx::checksum_page(2, &old) - ^ ltx::checksum_page(2, &next)); - let header = Header { - version: 3, - flags: 0, - page_size: PAGE_SIZE, - commit: COMMIT, - min_txid: TXID(2), - max_txid: TXID(2), - timestamp: 1_700_000_000_000, - pre_apply_checksum, - wal_offset: 0, - wal_size: 0, - wal_salt1: 0, - wal_salt2: 0, - node_id: 0, - }; - let name = "celld-10cb130-delta-2-2-512.ltx"; - let bytes = encode_file_v0_5_2(&header, &[(2, next)], post_apply_checksum) - .map_err(|error| format!("{name}: {error:?}"))?; - std::fs::write(output.join(name), &bytes)?; - println!( - "{name}: {} bytes, pre={pre_apply_checksum:#x}, post={post_apply_checksum:#x}", - bytes.len() - ); - Ok(()) -} diff --git a/crates/crab-ltx/tests/vectors/superfly-ltx-v0.5.2-delta-2-2-512.ltx b/crates/crab-ltx/tests/vectors/superfly-ltx-v0.5.2-delta-2-2-512.ltx deleted file mode 100644 index f31c8be59..000000000 Binary files a/crates/crab-ltx/tests/vectors/superfly-ltx-v0.5.2-delta-2-2-512.ltx and /dev/null differ diff --git a/crates/crab-ltx/tests/vectors/superfly-ltx-v0.5.2-snapshot-block-512.ltx b/crates/crab-ltx/tests/vectors/superfly-ltx-v0.5.2-snapshot-block-512.ltx deleted file mode 100644 index ff1404cd1..000000000 Binary files a/crates/crab-ltx/tests/vectors/superfly-ltx-v0.5.2-snapshot-block-512.ltx and /dev/null differ