From 7a26e55a544a1aa162551da0e4e3d08dddfae48c Mon Sep 17 00:00:00 2001 From: Scott Robinson Date: Thu, 1 Oct 2026 22:42:29 +0000 Subject: [PATCH 1/4] ci(integration): move to the dedicated ExtendDB runner pool ubuntu-latest is an Amazon-wide pool of 1,000 concurrent jobs shared first-come first-served across every org; our integration jobs sat queued 10-90 minutes behind other teams' load on 2026-09-29/30, which is what broke the merge queue (its 30-minute check timeout expired before jobs even started). All seven jobs now run on extenddb_ubuntu-2404_4-core: scoped to this repo, 4 vCPU / 16 GB (matching ubuntu-latest), capped at 20 concurrent, pinned to Ubuntu 24.04 ahead of ubuntu-latest's Oct-Nov migration to 26.04. Also: timeout-minutes: 30 on every job (none was set; on a capped pool a hung job holds a slot for hours), and a merge-queue-safe concurrency block that supersedes an in-flight run of the same PR on a new push without touching merge-group runs. --- .github/workflows/integration.yml | 35 ++++++++++++++++++++++++------- 1 file changed, 28 insertions(+), 7 deletions(-) diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index da5d274aa..48e6dbf26 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -8,9 +8,24 @@ on: permissions: contents: read +# Supersede an in-flight run of the SAME pull request when a new push arrives, +# so two pushes do not each hold seven slots on a capped pool. Conditional on +# purpose: a bare `cancel-in-progress: true` would also cancel merge-group +# runs, which evicts the entry from the merge queue and blocks the merge. +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: run-integration: - runs-on: ubuntu-latest + # Dedicated ExtendDB runner group (4 vCPU / 16 GB, capped at 20 concurrent), + # pinned to Ubuntu 24.04. ubuntu-latest is an Amazon-wide pool of 1,000 + # concurrent jobs shared first-come first-served across every org; our jobs + # sat queued 10–90 minutes behind other teams' load (2026-09-29/30). Pinned + # rather than "latest" because ubuntu-latest migrates to 26.04 in Oct–Nov + # 2026 and this compiles Rust; move to 26 deliberately. + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 services: postgres: # pgvector's own image, which is postgres:16 plus the extension. The @@ -88,7 +103,8 @@ jobs: run: sudo journalctl -t extenddb --no-pager -n 500 || true run-integration-sqlite: - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 steps: - uses: actions/checkout@v6 @@ -149,7 +165,8 @@ jobs: # transaction authorization regression reached main: the build compiled, so # a feature-matrix check passed, while nothing ever issued a request against # a dev-mode server. This job starts one and exercises the data plane. - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 steps: - uses: actions/checkout@v6 @@ -197,7 +214,8 @@ jobs: run: python3 -m pytest tests/test_dev_mode_authorization.py -v run-rust-integration: - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 services: postgres: image: pgvector/pgvector:pg16 @@ -296,7 +314,8 @@ jobs: # EXTENDDB_EXPECT_VECTORS flips to "1". Without this job the refusal surface # would stop being tested at exactly that point. run-rust-integration-postgres-novector: - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 services: postgres: image: postgres:16 @@ -368,7 +387,8 @@ jobs: # image and the job above already runs them; repeating them here would buy # a second release build and no coverage. run-rust-integration-sqlite: - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 steps: - uses: actions/checkout@v6 @@ -434,7 +454,8 @@ jobs: run: sudo journalctl -t extenddb --no-pager -n 500 || true integration: - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 needs: [ run-integration, From b0beb81fccf03e203452fa0024d47a00c15a36f6 Mon Sep 17 00:00:00 2001 From: Scott Robinson Date: Thu, 1 Oct 2026 23:06:02 +0000 Subject: [PATCH 2/4] ci(integration): key non-PR concurrency groups by run_id MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Independent audit finding: merge-group runs reuse the same ref (gh-readonly-queue/main/pr-N- recurs across re-queues of one PR, observed on three runs with different SHAs, two overlapping), and GitHub concurrency replaces a PENDING run in a group even with cancel-in-progress false. Keying non-PR runs by ref therefore let one merge-group run cancel another and evict its queue entry — the exact failure the block claimed to prevent. Non-PR runs now get a unique group (run_id); PR runs keep the per-PR supersede. Also corrects the 'seven slots' comment (six upstream jobs run concurrently). --- .github/workflows/integration.yml | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 48e6dbf26..98781ae1b 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -9,11 +9,14 @@ permissions: contents: read # Supersede an in-flight run of the SAME pull request when a new push arrives, -# so two pushes do not each hold seven slots on a capped pool. Conditional on -# purpose: a bare `cancel-in-progress: true` would also cancel merge-group -# runs, which evicts the entry from the merge queue and blocks the merge. +# so two pushes do not each hold six slots on a capped pool. Everything that is +# not a pull request gets a group unique to the run (run_id): merge-group runs +# can share a ref (gh-readonly-queue/main/pr-N- is reused across +# re-queues of the same PR), and GitHub concurrency replaces a PENDING run in +# a group even with cancel-in-progress false — so keying non-PR runs by ref +# would let one merge-group run cancel another and evict its queue entry. concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} + group: ${{ github.workflow }}-${{ github.event_name == 'pull_request' && github.event.pull_request.number || github.run_id }} cancel-in-progress: ${{ github.event_name == 'pull_request' }} jobs: From f1195a7a7aad02f80f5ca6f14b5f14cf21d78021 Mon Sep 17 00:00:00 2001 From: Scott Robinson Date: Thu, 1 Oct 2026 23:10:35 +0000 Subject: [PATCH 3/4] ci: move the remaining CI workflows to the dedicated runner pool MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Same three changes #379 made to integration.yml, applied to integration-mongodb, test, clippy, fmt, and licenses: runs-on extenddb_ubuntu-2404_4-core (pinned Ubuntu 24.04), a per-job timeout (30 min for the MongoDB suite whose longest job is 18 min; 15 min for the sub-5-minute workflows; licenses keeps its existing 20), and the corrected concurrency block — per-PR supersede on pull_request, run_id-unique groups for everything else so merge-group runs can never cancel one another. --- .github/workflows/clippy.yml | 16 ++++++++++++++- .github/workflows/fmt.yml | 16 ++++++++++++++- .github/workflows/integration-mongodb.yml | 25 +++++++++++++++++++---- .github/workflows/licenses.yml | 15 +++++++++++++- .github/workflows/test.yml | 19 +++++++++++++++-- 5 files changed, 82 insertions(+), 9 deletions(-) diff --git a/.github/workflows/clippy.yml b/.github/workflows/clippy.yml index 0594d12c6..ca3ad0281 100644 --- a/.github/workflows/clippy.yml +++ b/.github/workflows/clippy.yml @@ -5,9 +5,23 @@ on: push: branches: [main] +# Supersede an in-flight run of the SAME pull request when a new push arrives, +# so repeated pushes do not stack up on a capped pool. Non-PR runs get a group +# unique to the run (run_id): merge-group runs reuse a ref across re-queues, +# and GitHub concurrency replaces a PENDING run in a group even with +# cancel-in-progress false, so keying them by ref would let one merge-group +# run cancel another and evict its queue entry. +concurrency: + group: ${{ github.workflow }}-${{ github.event_name == 'pull_request' && github.event.pull_request.number || github.run_id }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: clippy: - runs-on: ubuntu-latest + # Dedicated ExtendDB runner group (Ubuntu 24.04, 4 vCPU / 16 GB, capped at + # 20 concurrent). ubuntu-latest is an Amazon-wide shared pool where our jobs + # waited 10-90 minutes to start (2026-09-29/30); see integration.yml. + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 15 steps: - uses: actions/checkout@v6 - uses: dtolnay/rust-toolchain@1.97.0 diff --git a/.github/workflows/fmt.yml b/.github/workflows/fmt.yml index ac5451b83..b0a847efe 100644 --- a/.github/workflows/fmt.yml +++ b/.github/workflows/fmt.yml @@ -5,9 +5,23 @@ on: push: branches: [main] +# Supersede an in-flight run of the SAME pull request when a new push arrives, +# so repeated pushes do not stack up on a capped pool. Non-PR runs get a group +# unique to the run (run_id): merge-group runs reuse a ref across re-queues, +# and GitHub concurrency replaces a PENDING run in a group even with +# cancel-in-progress false, so keying them by ref would let one merge-group +# run cancel another and evict its queue entry. +concurrency: + group: ${{ github.workflow }}-${{ github.event_name == 'pull_request' && github.event.pull_request.number || github.run_id }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: fmt: - runs-on: ubuntu-latest + # Dedicated ExtendDB runner group (Ubuntu 24.04, 4 vCPU / 16 GB, capped at + # 20 concurrent). ubuntu-latest is an Amazon-wide shared pool where our jobs + # waited 10-90 minutes to start (2026-09-29/30); see integration.yml. + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 15 steps: - uses: actions/checkout@v6 - uses: dtolnay/rust-toolchain@1.97.0 diff --git a/.github/workflows/integration-mongodb.yml b/.github/workflows/integration-mongodb.yml index 912df8bf7..a830d0962 100644 --- a/.github/workflows/integration-mongodb.yml +++ b/.github/workflows/integration-mongodb.yml @@ -18,9 +18,23 @@ permissions: # `run-tests --backend mongodb`, tearing everything down on exit. Both jobs # reuse it so CI runs the exact path developers run locally (no drift), while # splitting pytest and rust into parallel jobs the way integration.yml does. +# Supersede an in-flight run of the SAME pull request when a new push arrives, +# so repeated pushes do not stack up on a capped pool. Non-PR runs get a group +# unique to the run (run_id): merge-group runs reuse a ref across re-queues, +# and GitHub concurrency replaces a PENDING run in a group even with +# cancel-in-progress false, so keying them by ref would let one merge-group +# run cancel another and evict its queue entry. +concurrency: + group: ${{ github.workflow }}-${{ github.event_name == 'pull_request' && github.event.pull_request.number || github.run_id }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: mongodb-pytest: - runs-on: ubuntu-latest + # Dedicated ExtendDB runner group (Ubuntu 24.04, 4 vCPU / 16 GB, capped at + # 20 concurrent). ubuntu-latest is an Amazon-wide shared pool where our jobs + # waited 10-90 minutes to start (2026-09-29/30); see integration.yml. + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 steps: - uses: actions/checkout@v6 - uses: dtolnay/rust-toolchain@stable @@ -38,7 +52,8 @@ jobs: run: devtools/run-mongodb-tests -- --pytest --comprehensive --parallel mongodb-rust: - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 steps: - uses: actions/checkout@v6 - uses: dtolnay/rust-toolchain@stable @@ -56,7 +71,8 @@ jobs: run: devtools/run-mongodb-tests -- --rust --rust-integration mongodb-production-build: - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 steps: - uses: actions/checkout@v6 - uses: dtolnay/rust-toolchain@stable @@ -67,7 +83,8 @@ jobs: run: cargo check --release --no-default-features --features mongodb mongodb-integration: - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 30 needs: [mongodb-pytest, mongodb-rust, mongodb-production-build] if: always() steps: diff --git a/.github/workflows/licenses.yml b/.github/workflows/licenses.yml index 75d1ff2a6..e466004d2 100644 --- a/.github/workflows/licenses.yml +++ b/.github/workflows/licenses.yml @@ -40,9 +40,22 @@ on: permissions: contents: read +# Supersede an in-flight run of the SAME pull request when a new push arrives, +# so repeated pushes do not stack up on a capped pool. Non-PR runs get a group +# unique to the run (run_id): merge-group runs reuse a ref across re-queues, +# and GitHub concurrency replaces a PENDING run in a group even with +# cancel-in-progress false, so keying them by ref would let one merge-group +# run cancel another and evict its queue entry. +concurrency: + group: ${{ github.workflow }}-${{ github.event_name == 'pull_request' && github.event.pull_request.number || github.run_id }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: notices-current: - runs-on: ubuntu-latest + # Dedicated ExtendDB runner group (Ubuntu 24.04, 4 vCPU / 16 GB, capped at + # 20 concurrent). ubuntu-latest is an Amazon-wide shared pool where our jobs + # waited 10-90 minutes to start (2026-09-29/30); see integration.yml. + runs-on: extenddb_ubuntu-2404_4-core timeout-minutes: 20 steps: - uses: actions/checkout@v4 diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 73bbdb510..70325655d 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -8,9 +8,23 @@ on: permissions: contents: read +# Supersede an in-flight run of the SAME pull request when a new push arrives, +# so repeated pushes do not stack up on a capped pool. Non-PR runs get a group +# unique to the run (run_id): merge-group runs reuse a ref across re-queues, +# and GitHub concurrency replaces a PENDING run in a group even with +# cancel-in-progress false, so keying them by ref would let one merge-group +# run cancel another and evict its queue entry. +concurrency: + group: ${{ github.workflow }}-${{ github.event_name == 'pull_request' && github.event.pull_request.number || github.run_id }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + jobs: run-tests: - runs-on: ubuntu-latest + # Dedicated ExtendDB runner group (Ubuntu 24.04, 4 vCPU / 16 GB, capped at + # 20 concurrent). ubuntu-latest is an Amazon-wide shared pool where our jobs + # waited 10-90 minutes to start (2026-09-29/30); see integration.yml. + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 15 strategy: matrix: toolchain: [stable, "1.88.0"] @@ -23,7 +37,8 @@ jobs: - run: cargo test --workspace test: - runs-on: ubuntu-latest + runs-on: extenddb_ubuntu-2404_4-core + timeout-minutes: 15 needs: run-tests if: always() steps: From 4f478620dbc1a4a7c6ba202cad1594ce1af74a15 Mon Sep 17 00:00:00 2001 From: Scott Robinson Date: Thu, 1 Oct 2026 23:29:45 +0000 Subject: [PATCH 4/4] ci: 40-minute timeouts for the integration suites Second independent audit: four integration jobs reached 18-20 min on recent main runs (run-integration-sqlite 20:03), i.e. above 60% of the 30-minute cap. A timeout exists to catch HUNG jobs; at 30 min it would also clip the slow tail of healthy ones on a bad runner day and turn a passing merge-group round into a blocked merge. 40 min is 2x the worst observed duration while still bounding a hang. --- .github/workflows/integration-mongodb.yml | 8 ++++---- .github/workflows/integration.yml | 14 +++++++------- 2 files changed, 11 insertions(+), 11 deletions(-) diff --git a/.github/workflows/integration-mongodb.yml b/.github/workflows/integration-mongodb.yml index a830d0962..3df405a75 100644 --- a/.github/workflows/integration-mongodb.yml +++ b/.github/workflows/integration-mongodb.yml @@ -34,7 +34,7 @@ jobs: # 20 concurrent). ubuntu-latest is an Amazon-wide shared pool where our jobs # waited 10-90 minutes to start (2026-09-29/30); see integration.yml. runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 steps: - uses: actions/checkout@v6 - uses: dtolnay/rust-toolchain@stable @@ -53,7 +53,7 @@ jobs: mongodb-rust: runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 steps: - uses: actions/checkout@v6 - uses: dtolnay/rust-toolchain@stable @@ -72,7 +72,7 @@ jobs: mongodb-production-build: runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 steps: - uses: actions/checkout@v6 - uses: dtolnay/rust-toolchain@stable @@ -84,7 +84,7 @@ jobs: mongodb-integration: runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 needs: [mongodb-pytest, mongodb-rust, mongodb-production-build] if: always() steps: diff --git a/.github/workflows/integration.yml b/.github/workflows/integration.yml index 98781ae1b..5d1ab88ad 100644 --- a/.github/workflows/integration.yml +++ b/.github/workflows/integration.yml @@ -28,7 +28,7 @@ jobs: # rather than "latest" because ubuntu-latest migrates to 26.04 in Oct–Nov # 2026 and this compiles Rust; move to 26 deliberately. runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 services: postgres: # pgvector's own image, which is postgres:16 plus the extension. The @@ -107,7 +107,7 @@ jobs: run-integration-sqlite: runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 steps: - uses: actions/checkout@v6 @@ -169,7 +169,7 @@ jobs: # a feature-matrix check passed, while nothing ever issued a request against # a dev-mode server. This job starts one and exercises the data plane. runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 steps: - uses: actions/checkout@v6 @@ -218,7 +218,7 @@ jobs: run-rust-integration: runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 services: postgres: image: pgvector/pgvector:pg16 @@ -318,7 +318,7 @@ jobs: # would stop being tested at exactly that point. run-rust-integration-postgres-novector: runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 services: postgres: image: postgres:16 @@ -391,7 +391,7 @@ jobs: # a second release build and no coverage. run-rust-integration-sqlite: runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 steps: - uses: actions/checkout@v6 @@ -458,7 +458,7 @@ jobs: integration: runs-on: extenddb_ubuntu-2404_4-core - timeout-minutes: 30 + timeout-minutes: 40 needs: [ run-integration,