diff --git a/.github/workflows/integration-emulated-heavy.yml b/.github/workflows/integration-emulated-heavy.yml index 66beee42..7d1c104f 100644 --- a/.github/workflows/integration-emulated-heavy.yml +++ b/.github/workflows/integration-emulated-heavy.yml @@ -6,6 +6,9 @@ # pubsub, via tests/iac/_gcp_emulator/docker-compose.yml. Keyless. # • AWS — LocalStack (S3 + Glue). Needs LOCALSTACK_AUTH_TOKEN; Glue is a # LocalStack Pro feature (Ultimate tier and above). +# • Lakekeeper + Kafka Connect — forge-derived Iceberg sink configs on a +# real worker against a real Lakekeeper REST catalog. Keyless, in its +# own job (`lakekeeper-integration`). # # WHY THIS LANE RUNS NIGHTLY, NOT ON A LABEL AND NOT PER MERGE # ------------------------------------------------------------ @@ -310,3 +313,130 @@ jobs: if: always() run: | datahub docker nuke --keep-data 2>&1 | tail -5 || true + + # ──────────────────────────────────────────────────────────────────── + # Lakekeeper + Kafka Connect lane — forge's derived Iceberg sink + # configs on a real worker. + # + # What tests/integration/test_lakekeeper_kafka_connect_live.py proves: + # - a `location.catalog: lakekeeper` expose, derived through forge's + # real code, streams records into a real Lakekeeper REST catalog, + # which vends STS credentials for Silo (S3 + STS); + # - the same worker refuses a sink carrying both + # `iceberg.catalog.type` and `iceberg.catalog.catalog-impl`, the + # startup crash every Glue sink hit before the deriver emitted + # impl XOR type; + # - forge's derived Glue config gets past that refusal. + # + # Keyless: the stack is self-contained (Postgres, Lakekeeper, Silo, + # KRaft Kafka, cp-kafka-connect, every image pinned by digest), so + # there is no `environment:` and no deployment record. It still gates + # on the same label as its siblings so un-vouched PR code gets no + # runner. + # + # The test downloads the Apache Iceberg sink ZIP itself and verifies + # its sha1. The unzipped plugin is cached here by version + sha1; if + # the test's pin moves and this key does not, the stale cache is + # simply re-downloaded over (its .sha1 marker no longer matches). + lakekeeper-integration: + name: Lakekeeper + Kafka Connect integration + runs-on: ubuntu-latest + timeout-minutes: 45 + if: >- + github.event_name == 'schedule' || + github.event_name == 'workflow_dispatch' || + contains(github.event.pull_request.labels.*.name, 'ci:integration-emulated') + permissions: + contents: read + env: + ICEBERG_SINK_VERSION: "1.9.2" + # The ZIP's published sha1, not a credential. + ICEBERG_SINK_SHA1: 52dab2ff2b9659008deec803d1c1e92813c5ffa7 # pragma: allowlist secret + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: "3.12" + cache: pip + + # The test imports only forge's pure derivation modules and drives the + # stack through `docker compose` and urllib: the base install plus the + # pytest plugin pyproject.toml configures is all it needs. + - name: Install fluid-build + run: | + python -m pip install --upgrade pip + python -m pip install -e . "pytest>=7.4" "pytest-timeout>=2.3" + + - name: Restore the Iceberg sink plugin + id: plugin-cache + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ runner.temp }}/iceberg-kafka-connect + key: iceberg-kafka-connect-${{ env.ICEBERG_SINK_VERSION }}-${{ env.ICEBERG_SINK_SHA1 }} + + - name: Run the Lakekeeper + Kafka Connect live tests + timeout-minutes: 35 + env: + FLUID_TEST_LAKEKEEPER: "1" + FLUID_LK_PLUGIN_CACHE: ${{ runner.temp }}/iceberg-kafka-connect + # A fixed project name so the steps below can collect logs from, and + # tear down, a stack the test could not clean up (a timeout kill). + FLUID_LK_PROJECT: fluid-lk-ci + FLUID_LK_LOG_DIR: ${{ runner.temp }}/lakekeeper-logs + run: | + python -m pytest -v -m emulated_heavy \ + tests/integration/test_lakekeeper_kafka_connect_live.py \ + --junitxml=lakekeeper.xml + + # pytest exits 0 on an all-skipped run. This does not. + - name: Assert the lane actually exercised Lakekeeper + if: always() + run: | + python scripts/ci/assert_lane_coverage.py lakekeeper.xml \ + --require "Lakekeeper + Kafka Connect=tests/integration/test_lakekeeper_kafka_connect_live.py" + + # Saved whether or not the tests passed: the download does not depend on + # them. Only a verified install is saved; the test's one-shot writes the + # .sha1 marker last, after `sha1sum -c` passed. + - name: Check the plugin cache is complete + id: plugin-ready + if: always() && steps.plugin-cache.outputs.cache-hit != 'true' + env: + CACHE_DIR: ${{ runner.temp }}/iceberg-kafka-connect + run: | + if [ "$(cat "$CACHE_DIR/.sha1" 2>/dev/null)" = "$ICEBERG_SINK_SHA1" ]; then + echo "ready=true" >> "$GITHUB_OUTPUT" + fi + + - name: Save the Iceberg sink plugin + if: always() && steps.plugin-ready.outputs.ready == 'true' + uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ runner.temp }}/iceberg-kafka-connect + key: ${{ steps.plugin-cache.outputs.cache-primary-key }} + + # The test writes compose logs to FLUID_LK_LOG_DIR when it fails; this + # catches the case it could not, a run killed before its teardown. + - name: Collect logs from a stack left behind + if: failure() + env: + LOG_DIR: ${{ runner.temp }}/lakekeeper-logs + run: | + mkdir -p "$LOG_DIR" + docker compose -p fluid-lk-ci logs --no-color --tail=500 \ + > "$LOG_DIR/fluid-lk-ci-left-behind.log" 2>&1 || true + + - name: Upload JUnit + compose logs + if: failure() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: lakekeeper-integration-logs + path: | + lakekeeper.xml + ${{ runner.temp }}/lakekeeper-logs/ + if-no-files-found: ignore + retention-days: 14 + + - name: Tear down a stack left behind + if: always() + run: docker compose -p fluid-lk-ci down -v --remove-orphans || true diff --git a/CHANGELOG.md b/CHANGELOG.md index 1a27185a..eee10d00 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,8 +7,26 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Upgrade notes + +- An AWS contract that declared `location.catalog` other than `glue` (for example + `rest`, `iceberg_rest`, `polaris`, `unity`, `nessie` or `lakekeeper`) on an Iceberg + expose holds a Glue database and table in its OpenTofu state that earlier releases + created for it. `fluid apply` now stops before planning with + `iceberg_catalog_move_blocked` and prints the `tofu state rm` commands that release + them, instead of planning to destroy them: destroying a Glue database deletes every + table in it. The resources stay in AWS; run the commands, then apply again. + ### Changed +- CI: the emulated-heavy lane runs a real Kafka Connect worker (cp-kafka-connect 7.9.10 + with the Apache Iceberg sink 1.9.2) against a real Lakekeeper (v0.13.6, credentials + vended over STS from an S3-compatible store). It streams records through the config + forge derives for `catalog: lakekeeper`, shows that a config carrying both + `iceberg.catalog.type` and `catalog-impl` fails on the worker, and shows forge's Glue + config gets past that check. `assert_lane_coverage.py` fails the job if every test + skipped. Run it locally with `FLUID_TEST_LAKEKEEPER=1` (Docker required). + - CI: `release.yml` can publish through a TestPyPI outage. A manual run with `skip_testpypi: true` skips the TestPyPI upload and its install check and publishes the existing tag straight to PyPI; `verify-pypi` still installs and @@ -29,6 +47,54 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed +- **A Kafka Connect Iceberg sink on AWS Glue starts.** The derived connector config set + both `iceberg.catalog.type` and `iceberg.catalog.catalog-impl`, and Apache Iceberg's + `CatalogUtil` refuses that ("both type and catalog-impl are set"), so the sink failed + at startup. It now sets `catalog-impl` or `type`, never both, as the Debezium Server + sink already did. `fluid validate` also refuses an `iceberg_catalog_overrides` or + hand-written sink config that would add the other key back. +- **Every emitter reads `location.catalog` the same way.** The streaming sinks, dbt + `catalogs.yml`, the Snowflake, AWS and Confluent IaC, the native AWS planner, `fluid + policy compile`, `fluid diff` and `fluid test` each classified the free-string value by + hand, and disagreed: `catalog: lakekeeper` streamed over REST while dbt wrote a + Snowflake-managed table and the AWS IaC created a Glue table of the same name. One + table in `providers/_iceberg_catalog.py` now classifies it; spellings fold case and + `-`/`_` (`iceberg-rest` is `rest`, `snowflake` is `snowflake-managed`). + - AWS: an Iceberg expose in another catalog gets its S3 bucket but no Glue database, + table, import, Glue IAM grant or Lake Formation resource, and `fluid diff` and + `fluid test` no longer look for it in Glue. `governance.lakeFormation`, column + restrictions and row filters on such an expose are refused at `fluid validate` and + at apply, by catalog name, instead of being emitted against a Glue table that does + not exist. An absent catalog or `catalog: glue` emits exactly what it did. + - Snowflake: `lakekeeper` and the `iceberg-rest` spelling are external catalogs + (`catalog_type: iceberg_rest`, no EXTERNAL VOLUME), and `fluid validate` no longer + demands an `s3://` or `gs://` warehouse for them. `hive`, `jdbc`, `hadoop` and + `dynamodb`, which Snowflake has no catalog integration for, are a validate error and + are left out of `catalogs.yml` instead of becoming a Snowflake-managed table. + - dbt-bigquery: an expose naming a catalog other than `bigquery` is left out of + `catalogs.yml` with a warning instead of becoming a BigLake table. + - Confluent Tableflow: a `location.catalog` other than `glue` is a validate error; it + was ignored and the table was published to Glue. + - `catalog: dynamodb` reaches Iceberg as `catalog-impl` (`DynamoDbCatalog`); Iceberg + has no `dynamodb` type. +- **`fluid validate` refuses an Iceberg catalog value it does not know**, and lists the + accepted ones. The emitters used to fall back differently, so a typo split one table + across catalogs. `fluid apply` on AWS refuses it too (`unknown-iceberg-catalog`). +- **The streaming-sink checks cover every catalog.** `uri` and `warehouse` were required + only for the literal `rest`; every REST catalog (and Nessie) now needs them. A + `sink.catalog` that disagrees with the expose's catalog is refused, since dbt and the + IaC read only the expose. A warehouse override equal to the REST binding's warehouse + no longer warns that "the static Glue table may differ". The Kafka Connect runner and + the embedded Debezium Server runner run the same checks before they create anything. +- **A crash in an Iceberg or Confluent gate fails `fluid validate`.** It was printed only + with `--verbose`, and the contract passed unchecked. +- **`fluid validate` catches two `catalog: snowflake` Iceberg exposes that derive one + EXTERNAL VOLUME on different storage**, instead of `fluid apply` failing mid-emit. +- **`fluid policy compile` grants an Iceberg table where its catalog lives.** Every + Iceberg expose off GCP compiled to AWS S3 and Glue grants, whatever its platform. A + Snowflake-managed table now compiles to Snowflake grants, and a table in another + catalog gets no Glue grant and a warning to enforce access in that catalog. + - **A Lake Formation tag the contract does not define is associated, and the module validates.** A key in a binding's `governance.lakeFormation.tags` with no `governance.lakeFormation.tagDefinitions` entry got a `depends_on` on an diff --git a/RFC-streaming-extension.md b/RFC-streaming-extension.md index 28e81c35..5826c475 100644 --- a/RFC-streaming-extension.md +++ b/RFC-streaming-extension.md @@ -216,13 +216,22 @@ re-key is tested for zero key-loss — §10). | Key | Value | |---|---| -| `iceberg.catalog.type` | `glue` | +| `iceberg.catalog.type` | ~~`glue`~~ **not emitted** (correction below) | | `iceberg.catalog.catalog-impl` | `org.apache.iceberg.aws.glue.GlueCatalog` | | `iceberg.catalog.io-impl` | `org.apache.iceberg.aws.s3.S3FileIO` | | `iceberg.catalog.warehouse` | the **exact** `s3://{bucket}/{path}` `get_iceberg_warehouse` builds for the static table | | `iceberg.catalog.client.region` | from `binding.location.region` | | credentials | **none emitted** — DefaultCredentialsProvider chain (env / instance-profile / IRSA) | +> **CORRECTION (2026-10-01).** This table listed both `iceberg.catalog.type` +> and `iceberg.catalog.catalog-impl`, while §6.8 #1 says to reject a config with +> both. §6.8 is right: Iceberg's `CatalogUtil` refuses a catalog that sets +> `type` and `catalog-impl` together. The deriver emits **`catalog-impl` XOR +> `type`**: Glue (and DynamoDB, which has no `type`) get `catalog-impl` only, +> and every other kind gets `type` only. Both come from the catalog-kind table +> in `providers/_iceberg_catalog.py` (`catalog_kind_info`), which dbt, the IaC +> emitters and the policy compiler read too. + REST (`type=rest`, warehouse = catalog **name**, requires `s3.*` creds via `secret_ref`) and GCP (`GCSFileIO`) are **tagged-union variants deferred to PR7**. Each variant declares its own required-key set, validated at plan time. @@ -329,6 +338,18 @@ seam (`cli/plan.py`, after action-type parse, before `plan.json` serialization): streaming-only fallback — exactly where we'd assumed "auto-create handles it." Validator adds: streaming-only + auto-create ⇒ require a namespace-ensure action. + **CORRECTION (2026-10-01) to the correction above.** The observation is + kept as the record of what the spike saw, but its cause was the runtime, + not the design: the spike ran the Tabular `io.tabular` **0.6.19** runtime + (§14), not the Apache sink v1 ships against. The Apache sink creates the + namespace, and each of its parents, when it auto-creates a table, since + Iceberg **1.6.0** (apache/iceberg#10186: + `IcebergWriterFactory.createNamespaceIfNotExist`, called from + `autoCreateTable`; absent in 1.5.2). So **no ensure-namespace step is + needed** and the validator requirement above is **withdrawn**. One residue: + the sink ignores a `ForbiddenException` from `createNamespace`, so a + catalog principal without create-namespace rights still needs the + namespace created for it, or the table create fails. - **Operator override present:** when a hand-written `iceberg.catalog.warehouse` is set, **defer to the operator (warn, not fail)** — preserves operator-wins. @@ -521,6 +542,13 @@ by diffing the 0.6.19 README against the 1.11.0 docs — so the observed run validates the entire key surface; only those two constants differ and both are independently doc-confirmed for 1.11.0. +**Version caveat, corrected (2026-10-01):** an identical key surface did not +mean identical behaviour. 0.6.19 predates the namespace auto-create the Apache +sink gained in Iceberg 1.6.0, which is why correction A below was observed (see +§6.8 #6). And 0.6.19 is no longer the only published runtime: the Apache +Software Foundation now publishes the Apache sink on Confluent Hub +(`iceberg/iceberg-kafka-connect`, version 1.9.2). + **Result: PASS.** 15 rows physically committed to `default.events`; schema auto-inferred (`amount:double, name:string, id:long, region:string`); 2 snapshots; connector + task `RUNNING`. The exact config that worked is the same shape the @@ -532,7 +560,7 @@ converters). | # | Observed | RFC correction | |---|---|---| -| A | `auto-create-enabled=true` created the table but failed with `NoSuchNamespaceException: Namespace default does not exist` until the namespace was created explicitly. | The **auto-create fallback is incomplete** — forge must ensure the **namespace/database** exists even build-only. Folded into §6.8 #6. | +| A | `auto-create-enabled=true` created the table but failed with `NoSuchNamespaceException: Namespace default does not exist` until the namespace was created explicitly. | The **auto-create fallback is incomplete** — forge must ensure the **namespace/database** exists even build-only. Folded into §6.8 #6. **Withdrawn 2026-10-01:** a 0.6.19 behaviour; the Apache sink creates the namespace and its parents on auto-create since Iceberg 1.6.0 (§6.8 #6). | | B | `connector.state=RUNNING` while `tasks[0].state=FAILED` with a full trace. | The runner **must inspect task state + trace**, not just connector state. Confirmed §11 (now marked spike-observed). | | C | Schemaless JSON required `value.converter=JsonConverter` + `schemas.enable=false`; a task restart re-delivered records (duplicates) with no EOS. | The deriver must **co-emit converters** (§6.2, was implicit) and EOS config is **necessary not optional** (§11). | @@ -630,6 +658,11 @@ reference plugin). "Engine" becomes a binding concern: Kafka-Connect spike found (§14 A: auto-create makes the table, not the namespace) — a satisfying cross-topology consistency: **the writer owns the table, forge owns the namespace.** + **CORRECTION (2026-10-01):** the Tableflow conclusion stands on its own + evidence (the IAM policy has no `glue:CreateDatabase`), but the parallel does + not: §14 A was a 0.6.19 runtime behaviour, and the Apache sink creates the + namespace itself since Iceberg 1.6.0 (§6.8 #6). The split is + Tableflow-specific, not cross-topology. - **(b) Managed-class feasibility — ANSWERED: NO.** Confluent Cloud has **no fully-managed Iceberg sink connector**; the OSS `org.apache.iceberg.connect` class runs only via "Bring Your Own Connector" (custom upload). The managed @@ -666,7 +699,7 @@ sweep (mirrors the existing `FLUID_IAC_LIVE_{AWS,GCP,SNOWFLAKE}` gates). | Key | Role | forge derivation | |---|---|---| | `connector.class` | sink class | constant `org.apache.iceberg.connect.IcebergSinkConnector` | -| `iceberg.catalog.type` / `catalog-impl` | catalog kind | from `binding.platform` | +| `iceberg.catalog.type` / `catalog-impl` | catalog kind | from the catalog kind (`location.catalog`, else the `binding.platform` default); `catalog-impl` XOR `type` (corrected 2026-10-01, §6.3) | | `iceberg.catalog.warehouse` | warehouse root | `get_iceberg_warehouse()` (shared with static path) | | `iceberg.catalog.io-impl` | FileIO | per-platform (`S3FileIO` for AWS) | | `iceberg.catalog.client.region` | region | `binding.location.region` | @@ -676,9 +709,9 @@ sweep (mirrors the existing `FLUID_IAC_LIVE_{AWS,GCP,SNOWFLAKE}` gates). | `iceberg.tables.default-partition-by` | partitioning | `icebergConfig.partitionSpec` | | `iceberg.control.topic` | EOS control | `_iceberg-control-{product_id}` (spike-validated custom topic) | | `iceberg.coordinator.transactional.prefix` | EOS | `iceberg-coord-{product_id}` | -| `iceberg.tables.auto-create-enabled` | build-only fallback | `true` for streaming-only — **but namespace must pre-exist** (§14 A) | +| `iceberg.tables.auto-create-enabled` | build-only fallback | `true` for streaming-only; ~~**but namespace must pre-exist** (§14 A)~~ the sink creates the namespace and its parents too, since Iceberg 1.6.0 (corrected 2026-10-01, §6.8 #6) | | `key.converter` / `value.converter` | record decode | `JsonConverter` + `schemas.enable=false` (schemaless JSON; §14 C) | -| (catalog namespace) | table parent | **forge must `ensure-namespace`** — auto-create does NOT (§14 A) | +| (catalog namespace) | table parent | ~~**forge must `ensure-namespace`** — auto-create does NOT (§14 A)~~ created by the sink on auto-create since Iceberg 1.6.0 (apache/iceberg#10186); no forge step (corrected 2026-10-01, §6.8 #6) | ## Appendix B — verification log diff --git a/docs/INTEGRATION_TESTING.md b/docs/INTEGRATION_TESTING.md index 4c361d37..fbef40e5 100644 --- a/docs/INTEGRATION_TESTING.md +++ b/docs/INTEGRATION_TESTING.md @@ -138,6 +138,37 @@ python scripts/cleanup_bigquery_test_artifacts.py python scripts/cleanup_aws_test_artifacts.py ``` +## Heavy emulated lane: Lakekeeper + Kafka Connect + +`integration-emulated-heavy.yml::lakekeeper-integration` runs `tests/integration/test_lakekeeper_kafka_connect_live.py` against a throwaway Docker stack: Postgres and Lakekeeper (an Iceberg REST catalog), Silo (S3 + STS), a single-node KRaft Kafka, and `cp-kafka-connect` with the Apache Iceberg sink 1.9.2. Every image is pinned by digest and the job holds no secrets. + +Each sink config is derived only through forge's own code (`find_iceberg_expose_binding` → `resolve_iceberg_catalog` → `emit_iceberg_sink_config`), then handed to the real worker: + +- A `location.catalog: lakekeeper` expose streams records into Lakekeeper. The config carries `type=rest`, the warehouse by name, and no FileIO or storage keys, because Lakekeeper vends STS credentials. The sink creates the namespace and the table, and the test waits until the current snapshot holds every record. +- A sink carrying both `iceberg.catalog.type` and `iceberg.catalog.catalog-impl` fails at task start with `Cannot create catalog iceberg, both type and catalog-impl are set`, in the task status trace and in the worker log. Every Glue sink failed this way until the deriver emitted one key or the other. +- forge's derived Glue config (`catalog-impl` only) starts on the same worker. It writes nothing, so no AWS account is involved. + +The lane runs nightly, on `workflow_dispatch`, and on a PR carrying the `ci:integration-emulated` label. `scripts/ci/assert_lane_coverage.py` fails the job if the tests skipped. + +To run it locally you need Docker with about 2 GB of memory to spare. Once the images are pulled, a run takes about three minutes, most of it the worker's first commit: + +```bash +pip install -e ".[dev]" + +FLUID_TEST_LAKEKEEPER=1 \ +FLUID_LK_PLUGIN_CACHE=~/.cache/fluid/iceberg-kafka-connect \ + pytest -v -m emulated_heavy tests/integration/test_lakekeeper_kafka_connect_live.py +``` + +| Variable | Effect | +|---|---| +| `FLUID_TEST_LAKEKEEPER=1` | Opts in. Without it, or without a reachable Docker daemon, the module skips. | +| `FLUID_LK_PLUGIN_CACHE` | Host directory for the unzipped sink plugin (about 165 MB), reused across runs. Unset, a compose volume holds it and teardown deletes it, so every run downloads it again. | +| `FLUID_LK_PROJECT` | Compose project name. Defaults to `fluid-lk-`. | +| `FLUID_LK_LOG_DIR` | Where to write the compose logs when a test fails. They are printed either way. | + +The stack publishes only Lakekeeper and Connect, on free loopback ports, and the test runs `docker compose down -v --remove-orphans` when it finishes, pass or fail. + ## How CI enforces the safety properties The `.github/workflows/actionlint.yml` workflow runs on every push and PR (no secrets needed). It fails CI if: diff --git a/fluid_build/build_runners/debezium/runner.py b/fluid_build/build_runners/debezium/runner.py index 58f586c0..270f0d00 100644 --- a/fluid_build/build_runners/debezium/runner.py +++ b/fluid_build/build_runners/debezium/runner.py @@ -522,8 +522,15 @@ def _execute_debezium_server( find_iceberg_expose_binding, resolve_iceberg_catalog, ) + from ..kafka_connect.iceberg_sink_validation import iceberg_sink_preflight from .iceberg_sink import emit_debezium_iceberg_sink_config + # Fail closed BEFORE application.properties is written: the same + # checks `fluid validate` runs, so a sink it rejects never boots. + preflight_error = iceberg_sink_preflight(ctx.contract, ctx.build_id, log=LOG) + if preflight_error: + return _failed(ctx, started_at, t_start, preflight_error) + binding = find_iceberg_expose_binding(ctx.contract) if binding is not None: resolved = resolve_iceberg_catalog( diff --git a/fluid_build/build_runners/kafka_connect/iceberg_sink.py b/fluid_build/build_runners/kafka_connect/iceberg_sink.py index 2627c512..2066eb79 100644 --- a/fluid_build/build_runners/kafka_connect/iceberg_sink.py +++ b/fluid_build/build_runners/kafka_connect/iceberg_sink.py @@ -96,9 +96,13 @@ def emit_iceberg_sink_config( } # ── catalog block (prefix-passthrough: derive a few, forward the rest) ── - cfg["iceberg.catalog.type"] = resolved.catalog_type + # catalog-impl XOR type, as the Debezium twin does: Iceberg's + # CatalogUtil.loadCatalog throws "both type and catalog-impl are set" + # when it gets both, so a Glue sink emitted with both never started. if resolved.catalog_impl: cfg["iceberg.catalog.catalog-impl"] = resolved.catalog_impl + else: + cfg["iceberg.catalog.type"] = resolved.catalog_type if resolved.warehouse: cfg["iceberg.catalog.warehouse"] = resolved.warehouse if resolved.io_impl: diff --git a/fluid_build/build_runners/kafka_connect/iceberg_sink_validation.py b/fluid_build/build_runners/kafka_connect/iceberg_sink_validation.py index b519a70b..654f1cc8 100644 --- a/fluid_build/build_runners/kafka_connect/iceberg_sink_validation.py +++ b/fluid_build/build_runners/kafka_connect/iceberg_sink_validation.py @@ -16,11 +16,36 @@ These catch the connector's silent-fail-at-first-record traps BEFORE any apply: a sink with no matching Iceberg expose (the deriver would no-op or mis-target), -an incomplete catalog tagged-union, the v1-deferred upsert mode, dynamic routing -without a route field, and an operator warehouse override that diverges from the -binding (which would split the streaming write from the static Glue table). It is -a pure function returning (errors, warnings) — the validate stage routes errors -to its collector and surfaces warnings, mirroring product_types.py. +the v1-deferred upsert mode, dynamic routing without a route field, and the +catalog traps below. It is a pure function returning (errors, warnings) — the +validate stage routes errors to its collector and surfaces warnings, mirroring +product_types.py. The Kafka-Connect and Debezium-Server runners run the SAME +checks through :func:`iceberg_sink_preflight` right before they derive a sink, +so a contract ``fluid validate`` rejects fails its run before any Connect REST +call or ``application.properties`` write, rather than only when someone +remembered to validate first. + +The catalog checks are TABLE-DRIVEN: each one reads the kind's row from +``providers/_iceberg_catalog.py`` (the classification the sink deriver, dbt +``catalogs.yml`` and the IaC emitters share) instead of matching literals. +This validator used to check only ``catalog == "rest"``, so ``catalog: +lakekeeper`` passed with no ``uri``, streamed over REST, and dbt wrote a +Snowflake-managed table for the same expose. Per sink build: + +* the kind must be in the table: an unknown value gets each emitter's historic + fallback, and those fallbacks disagree with one another; +* ``sink.catalog`` must agree with the expose's catalog, because dbt and the + IaC read only the expose; +* every ``binding.location`` key in the row's ``sink_requires`` must be set + (Glue keeps its advisory region warning: the warehouse falls back); +* the runtime must ship the catalog's client (the stock Apache Iceberg Kafka + Connect runtime has no Nessie client); +* an operator override must not move the warehouse away from the binding, nor + leave the connector carrying both ``type`` and ``catalog-impl``. + +Debezium builds in ``bring-your-own`` / ``managed`` mode create only the SOURCE +connector (no sink is derived), so none of this applies to them; an embedded +Debezium Server build is checked whenever it would derive an Iceberg sink. Why imperative Python and not JSON-Schema if/then: these are CROSS-OBJECT checks (a build's sink ↔ a different expose's binding; a computed warehouse vs an @@ -36,122 +61,368 @@ from __future__ import annotations -from typing import Any, List, Mapping, Tuple +import logging +from dataclasses import dataclass +from typing import AbstractSet, Any, Dict, List, Mapping, Optional, Sequence, Tuple + +from ...providers._iceberg_catalog import ( + FAMILY_GLUE, + FAMILY_REST, + FAMILY_UNKNOWN, + CatalogKind, + binding_catalog_kind, + canonical_catalog_kind, + catalog_kind_info, + iceberg_catalog_kind, + iceberg_sink_exposes, + known_catalog_kinds, + resolve_iceberg_catalog, +) + +LOG = logging.getLogger("fluid.acquire.iceberg_sink") + +_Issues = Tuple[List[str], List[str]] + +#: ``only_build`` default: check every build. A sentinel, not ``None``, because +#: a build with an explicit ``id: null`` runs with ``ctx.build_id is None``. +_ALL_BUILDS: Any = object() + + +@dataclass(frozen=True) +class _SinkRuntime: + """How one build's runner assembles its Iceberg sink config. + + The two runtimes spell the catalog selector differently (Kafka Connect + prefixes ``iceberg.catalog.``; Debezium Server takes bare keys) and merge + different operator maps over the derived block, so the override checks read + these instead of assuming the Kafka-Connect shape. + """ + + engine: str + #: Does the runner derive a catalog block from the expose for this build? + derives: bool + type_key: str + impl_key: str + #: ``(label, map)`` merged over the derived config, in merge order. + overrides: Tuple[Tuple[str, Mapping[str, Any]], ...] + #: ``(label, value)`` of the warehouse override the zero-drift check reads. + warehouse_override: Tuple[str, Any] -_OBJECT_STORE_PREFIXES = ("s3://", "s3a://", "gs://", "abfss://") -# binding.format aliases that normalize to canonical "iceberg" (mirror _common). -_ICEBERG_FORMATS = {"iceberg", "iceberg_table", "iceberg-table"} +def _sink_runtime(build: Mapping[str, Any]) -> Optional[_SinkRuntime]: + """The runtime that writes an Iceberg sink for ``build``, or ``None``. -def _is_iceberg_binding(exposure: Any) -> bool: - if not isinstance(exposure, Mapping): - return False - binding = exposure.get("binding") or {} - # A confluent expose is a MANAGED Tableflow output (the Confluent IaC plugin - # + validate_confluent_binding own it), not a self-managed Kafka-Connect sink - # target. Excluding it keeps this validator from demanding REST/Glue catalog - # fields the managed path doesn't use (RFC §15). - if str(binding.get("platform") or "").lower() == "confluent": - return False - return str(binding.get("format") or "").lower() in _ICEBERG_FORMATS + Mirrors each runner's own gate. The Debezium runner derives from + ``server.sink.type`` (default ``iceberg``) in embedded mode and never reads + ``sink.format``, so selecting its builds by ``sink.format`` alone would let + a deriving build skip every check while one that never derives (a + bring-your-own build creates only the source connector) drew errors about + a catalog nothing writes to. + """ + props = build.get("properties") or {} + sink = props.get("sink") or {} + declares_iceberg = str(sink.get("format") or "").lower() == "iceberg" + engine = str(build.get("engine") or "").strip().lower() + + if engine == "debezium": + dbz = props.get("debezium") or {} + if (dbz.get("deployment") or {}).get("mode", "bring-your-own") != "embedded": + return None + server_sink = (dbz.get("server") or {}).get("sink") or {} + if server_sink.get("type", "iceberg") != "iceberg": + return None + derives = bool(server_sink.get("iceberg_sink_enabled", "config" not in server_sink)) + if not (declares_iceberg or derives): + return None + config = server_sink.get("config") or {} + label = "debezium.server.sink.config" + return _SinkRuntime( + engine=engine, + derives=derives, + type_key="type", + impl_key="catalog-impl", + overrides=((label, config),), + warehouse_override=(label, config.get("warehouse")), + ) + + if not declares_iceberg: + return None + kc = props.get("kafka-connect") or {} + handwritten = kc.get("sink_connector_config") + derives = bool(kc.get("iceberg_sink_enabled", handwritten is None)) + catalog_overrides = kc.get("iceberg_catalog_overrides") or {} + # ``iceberg_catalog_overrides`` is applied inside the deriver, so it only + # reaches the connector when the runner derives; a hand-written + # ``sink_connector_config`` is merged over the result either way. + overrides: Tuple[Tuple[str, Mapping[str, Any]], ...] = () + if derives: + overrides += (("iceberg_catalog_overrides", catalog_overrides),) + if isinstance(handwritten, Mapping): + overrides += (("sink_connector_config", handwritten),) + return _SinkRuntime( + engine=engine, + derives=derives, + type_key="iceberg.catalog.type", + impl_key="iceberg.catalog.catalog-impl", + overrides=overrides, + warehouse_override=( + "iceberg_catalog_overrides", + catalog_overrides.get("iceberg.catalog.warehouse"), + ), + ) def validate_iceberg_sink(contract: Mapping[str, Any]) -> Tuple[List[str], List[str]]: """Return (errors, warnings) for every Iceberg streaming-sink build.""" + return _validate(contract) + + +def iceberg_sink_preflight( + contract: Mapping[str, Any], + build_id: Any, + *, + log: Optional[logging.Logger] = None, +) -> Optional[str]: + """Run-time twin of ``fluid validate``: ONE build's sink errors, or ``None``. + + A runner calls this right before it derives a sink, so a contract the + validator rejects fails the run before any Connect REST call or file write + instead of shipping a connector that never starts (both selector keys) or + writes into a different catalog than dbt reads. ``build_id`` is the run + context's id (``build.get("id", "unknown")``); the build is selected by it, + not by matching message text. Warnings are logged and never fatal, the same + split as ``fluid validate`` without ``--strict``. + """ + errors, warnings = _validate(contract, only_build=build_id) + for msg in warnings: + (log or LOG).warning("iceberg_sink.preflight.warning build=%s %s", build_id, msg) + if not errors: + return None + return "iceberg sink preflight failed (see `fluid validate`): " + "; ".join(errors) + + +def _validate(contract: Mapping[str, Any], *, only_build: Any = _ALL_BUILDS) -> _Issues: errors: List[str] = [] warnings: List[str] = [] - exposes = [e for e in (contract.get("exposes") or []) if isinstance(e, Mapping)] - iceberg_exposes = [e for e in exposes if _is_iceberg_binding(e)] + iceberg_exposes = iceberg_sink_exposes(contract) expose_ids = {e.get("exposeId") or e.get("id") for e in iceberg_exposes} for build in contract.get("builds") or []: if not isinstance(build, Mapping): continue - props = build.get("properties") or {} - sink = props.get("sink") or {} - if str(sink.get("format") or "").lower() != "iceberg": - continue # not an Iceberg sink build - - bid = build.get("id", "?") - kc = props.get("kafka-connect") or {} - streaming = kc.get("streamingSink") or kc.get("streaming_sink") or {} - - # 1. build -> expose join: an Iceberg sink needs a matching Iceberg - # expose (the deriver resolves the catalog identity from it). HARD. - if not iceberg_exposes: - errors.append( - f"iceberg sink (build {bid!r}) has no expose with binding.format=iceberg; " - "the connector has no table identity to write to" - ) + # ``build_acquisition_run_context`` ids a build ``build.get("id", "unknown")``. + if only_build is not _ALL_BUILDS and build.get("id", "unknown") != only_build: continue - outputs = build.get("outputs") or [] - if outputs and not (set(outputs) & expose_ids): - warnings.append( - f"iceberg sink (build {bid!r}) outputs {list(outputs)} don't reference the " - f"Iceberg expose(s) {sorted(x for x in expose_ids if x)}; the join is implicit" - ) - binding = iceberg_exposes[0].get("binding") or {} - loc = binding.get("location") or {} + runtime = _sink_runtime(build) + if runtime is None: + continue # not a build that writes an Iceberg sink + _check_build(build, runtime, iceberg_exposes, expose_ids, errors, warnings) - # 2. upsert is deferred to v2 (locked v1 decision) — gate, don't silently - # append-only. HARD. - if streaming.get("upsertMode") is True: - errors.append( - f"iceberg sink (build {bid!r}): streamingSink.upsertMode is not supported in " - "v1 (CDC/upsert deferred); remove it or use append mode" - ) + return errors, warnings + + +def _check_build( + build: Mapping[str, Any], + runtime: _SinkRuntime, + iceberg_exposes: Sequence[Mapping[str, Any]], + expose_ids: AbstractSet[Any], + errors: List[str], + warnings: List[str], +) -> None: + props = build.get("properties") or {} + sink = props.get("sink") or {} + bid = build.get("id", "?") + kc = props.get("kafka-connect") or {} + streaming = kc.get("streamingSink") or kc.get("streaming_sink") or {} + + # 1. build -> expose join: an Iceberg sink needs a matching Iceberg + # expose (the deriver resolves the catalog identity from it). HARD. + if not iceberg_exposes: + errors.append( + f"iceberg sink (build {bid!r}) has no expose with binding.format=iceberg; " + "the connector has no table identity to write to" + ) + return + outputs = build.get("outputs") or [] + if outputs and not (set(outputs) & expose_ids): + warnings.append( + f"iceberg sink (build {bid!r}) outputs {list(outputs)} don't reference the " + f"Iceberg expose(s) {sorted(x for x in expose_ids if x)}; the join is implicit" + ) + binding = iceberg_exposes[0].get("binding") or {} + + # 2. upsert is deferred to v2 (locked v1 decision) — gate, don't silently + # append-only. HARD. + if streaming.get("upsertMode") is True: + errors.append( + f"iceberg sink (build {bid!r}): streamingSink.upsertMode is not supported in " + "v1 (CDC/upsert deferred); remove it or use append mode" + ) + + # 3. dynamic routing needs a route field, else records with no target are + # silently dropped. HARD. + if streaming.get("dynamicEnabled") is True and not streaming.get("routeField"): + errors.append( + f"iceberg sink (build {bid!r}): streamingSink.dynamicEnabled requires " + "streamingSink.routeField" + ) + + _check_catalog(bid, binding, sink, runtime, errors, warnings) + + +def _check_catalog( + bid: Any, + binding: Mapping[str, Any], + sink: Mapping[str, Any], + runtime: _SinkRuntime, + errors: List[str], + warnings: List[str], +) -> None: + """The catalog checks (4-6), every one read off the kind's table row.""" + loc = binding.get("location") or {} + kind = iceberg_catalog_kind(binding, sink) + info = catalog_kind_info(kind) + sink_kind = canonical_catalog_kind(sink.get("catalog")) + + # An unknown kind gets each emitter's historic fallback (REST for the sink, + # Snowflake-managed for dbt), so the per-kind checks below would be guesses + # about a row that does not exist. Refuse the value instead. HARD. + if info.family == FAMILY_UNKNOWN: + from_sink = bool(sink_kind) + raw = sink.get("catalog") if from_sink else loc.get("catalog") + where = "sink.catalog" if from_sink else "binding.location.catalog" + errors.append( + f"iceberg sink (build {bid!r}): unknown Iceberg catalog {raw!r} in {where}; " + f"use one of: {', '.join(known_catalog_kinds())}" + ) + return - # 3. dynamic routing needs a route field, else records with no target are - # silently dropped. HARD. - if streaming.get("dynamicEnabled") is True and not streaming.get("routeField"): + # The sink must write through the catalog dbt and the IaC read. Both read + # only the expose, so a ``sink.catalog`` that differs streams into one + # catalog while the static table and the models live in another. Compared + # canonically: ``iceberg-rest`` and ``rest`` are the same catalog. HARD. + expose_kind = binding_catalog_kind(binding) + if sink_kind and sink_kind != expose_kind: + expose_source = ( + "binding.location.catalog" + if canonical_catalog_kind(loc.get("catalog")) + else f"the {binding.get('platform')!r} platform default" + ) + errors.append( + f"iceberg sink (build {bid!r}): sink.catalog {sink.get('catalog')!r} disagrees " + f"with the expose's catalog {expose_kind!r} ({expose_source}); the sink would " + f"write through {sink_kind} while dbt and the IaC read {expose_kind}. Set " + f"binding.location.catalog: {sink_kind} and drop sink.catalog" + ) + + # 4. catalog tagged-union completeness, from the row's ``sink_requires``. + # HARD: forge can't derive a REST uri or a catalog name. Glue requires + # nothing (the warehouse falls back), so its region stays advisory. + for key in info.sink_requires: + if not loc.get(key): + suffix = ( + " (the catalog name)" if key == "warehouse" and info.family == FAMILY_REST else "" + ) errors.append( - f"iceberg sink (build {bid!r}): streamingSink.dynamicEnabled requires " - "streamingSink.routeField" + f"iceberg sink (build {bid!r}): {kind} catalog requires " + f"binding.location.{key}{suffix}" ) + if info.family == FAMILY_GLUE and not loc.get("region"): + warnings.append( + f"iceberg sink (build {bid!r}): glue catalog without binding.location.region; " + "the connector needs iceberg.catalog.client.region" + ) - # 4. catalog tagged-union completeness. HARD for REST (forge can't derive - # a uri/warehouse); advisory for Glue (warehouse falls back). - catalog_kind = str( - sink.get("catalog") - or loc.get("catalog") - or ("glue" if str(binding.get("platform") or "").lower() == "aws" else "rest") - ).lower() - if catalog_kind == "rest": - if not loc.get("uri"): - errors.append( - f"iceberg sink (build {bid!r}): rest catalog requires binding.location.uri" - ) - if not loc.get("warehouse"): - errors.append( - f"iceberg sink (build {bid!r}): rest catalog requires " - "binding.location.warehouse (the catalog name)" - ) - elif catalog_kind == "glue": - if not loc.get("region"): - warnings.append( - f"iceberg sink (build {bid!r}): glue catalog without binding.location.region; " - "the connector needs iceberg.catalog.client.region" - ) - - # 5. zero-drift cross-check (consumes PR1's same_warehouse): if the - # operator overrides the warehouse, it must still match the binding — - # else the streaming write and the static Glue table diverge. Operator - # wins (warn, not fail), per the locked decision. - overrides = kc.get("iceberg_catalog_overrides") or {} - override_wh = overrides.get("iceberg.catalog.warehouse") - if override_wh: - from fluid_build.providers.aws.util.warehouse import ( - get_iceberg_warehouse, - same_warehouse, - ) + # The stock Apache Iceberg Kafka Connect runtime bundles the AWS, GCP and + # Azure modules (Hive only in its -hive- distribution) but no iceberg-nessie + # (apache/iceberg kafka-connect/build.gradle), so ``type=nessie`` fails to + # load NessieCatalog on a stock worker. Advisory: a custom image may add it. + if info.name == "nessie" and runtime.engine == "kafka-connect": + warnings.append( + f"iceberg sink (build {bid!r}): the stock Apache Iceberg Kafka Connect runtime " + "does not bundle iceberg-nessie (apache/iceberg kafka-connect/build.gradle); " + "add the iceberg-nessie jar to the worker's connector plugin directory, or " + "the sink cannot load NessieCatalog" + ) - derived_wh = get_iceberg_warehouse(loc, account_ref="") - if not same_warehouse(override_wh, derived_wh): - warnings.append( - f"iceberg sink (build {bid!r}): iceberg_catalog_overrides warehouse " - f"{override_wh!r} diverges from the binding warehouse {derived_wh!r}; " - "the connector will use the override but the static Glue table may differ" - ) + _check_warehouse_override(bid, binding, sink, kind, info, runtime, warnings) + _check_selector_overrides(bid, kind, info, runtime, errors) - return errors, warnings + +def _check_warehouse_override( + bid: Any, + binding: Mapping[str, Any], + sink: Mapping[str, Any], + kind: str, + info: CatalogKind, + runtime: _SinkRuntime, + warnings: List[str], +) -> None: + """5. zero-drift cross-check (consumes PR1's same_warehouse). + + If the operator overrides the warehouse it must still match the binding, + else the streaming write and what dbt / the IaC address diverge. Operator + wins (warn, not fail), per the locked decision. The comparison is against + the warehouse the deriver itself resolves: for Glue that is PR1's canonical + ``s3://`` writer (unchanged), for every other catalog it is + ``location.warehouse``. Comparing a REST catalog's override against the + Glue writer used to report a matching override as diverging. + """ + label, override_wh = runtime.warehouse_override + if not override_wh: + return + from fluid_build.providers.aws.util.warehouse import same_warehouse + + derived_wh = resolve_iceberg_catalog(binding, sink=sink, account_ref="").warehouse + if not derived_wh and "warehouse" in info.sink_requires: + return # check 4 already refused the missing warehouse + if same_warehouse(override_wh, derived_wh): + return + if info.family == FAMILY_GLUE: + warnings.append( + f"iceberg sink (build {bid!r}): {label} warehouse " + f"{override_wh!r} diverges from the binding warehouse {derived_wh!r}; " + "the connector will use the override but the static Glue table may differ" + ) + else: + warnings.append( + f"iceberg sink (build {bid!r}): {label} warehouse " + f"{override_wh!r} diverges from the binding warehouse {derived_wh!r} of the " + f"{kind} catalog; the sink will write there while dbt and the IaC address " + "the binding's warehouse" + ) + + +def _check_selector_overrides( + bid: Any, + kind: str, + info: CatalogKind, + runtime: _SinkRuntime, + errors: List[str], +) -> None: + """6. the merged config must carry ``type`` XOR ``catalog-impl``. + + Iceberg's ``CatalogUtil.buildIcebergCatalog`` throws "both type and + catalog-impl are set" when it gets both, so the sink never starts. The + deriver emits exactly one; an override of the OTHER key (``type`` over a + derived Glue ``catalog-impl``, say) silently re-creates the crash, because + override maps are merged last. Presence is what counts: CatalogUtil checks + the key for null, so even an empty value trips it. HARD. + """ + setters: Dict[str, str] = {} + if runtime.derives: + derived = runtime.impl_key if info.catalog_impl else runtime.type_key + setters[derived] = f"the derived {kind} catalog config" + for label, mapping in runtime.overrides: + for key in (runtime.type_key, runtime.impl_key): + if key in mapping: + setters[key] = label + if runtime.type_key in setters and runtime.impl_key in setters: + errors.append( + f"iceberg sink (build {bid!r}): the sink config would carry both " + f"{runtime.type_key} (from {setters[runtime.type_key]}) and {runtime.impl_key} " + f"(from {setters[runtime.impl_key]}); Iceberg's CatalogUtil refuses a catalog " + "with both type and catalog-impl set, so the sink never starts. Keep one, and " + "change catalogs with binding.location.catalog rather than an override" + ) diff --git a/fluid_build/build_runners/kafka_connect/runner.py b/fluid_build/build_runners/kafka_connect/runner.py index af2895b9..ea9eb1da 100644 --- a/fluid_build/build_runners/kafka_connect/runner.py +++ b/fluid_build/build_runners/kafka_connect/runner.py @@ -345,6 +345,14 @@ def _execute(ctx: RunContext, runner: KafkaConnectRunner) -> RunResult: sink_config = kc_props.get("sink_connector_config") iceberg_enabled = kc_props.get("iceberg_sink_enabled", sink_config is None) if iceberg_enabled and str(ctx.sink.format or "").lower() == "iceberg": + # Fail closed BEFORE any Connect REST call: the same checks `fluid + # validate` runs, so a sink it rejects (a Lakekeeper binding with no + # uri, both catalog selectors set) never reaches the cluster. + from .iceberg_sink_validation import iceberg_sink_preflight + + preflight_error = iceberg_sink_preflight(ctx.contract, ctx.build_id, log=LOG) + if preflight_error: + return _failed(ctx, started_at, t_start, preflight_error) binding = _find_iceberg_expose_binding(ctx.contract) if binding is not None: from ...providers._iceberg_catalog import resolve_iceberg_catalog diff --git a/fluid_build/cli/_apply_opentofu_engine.py b/fluid_build/cli/_apply_opentofu_engine.py index a87c91be..edee1fe1 100644 --- a/fluid_build/cli/_apply_opentofu_engine.py +++ b/fluid_build/cli/_apply_opentofu_engine.py @@ -215,6 +215,11 @@ def apply_via_opentofu(args, logger: logging.Logger) -> int: contract, str(workdir), env, args, logger, plugin=plugin, actions=actions ) + # Pre-plan Iceberg catalog-move guard: Glue resources an older forge-cli + # created for an Iceberg table that lives in Lakekeeper / a REST catalog + # would otherwise be planned for DESTROY (iac/catalog_moves.py). + _guard_catalog_moves(plugin, contract, provider, str(workdir), env, logger, actions=actions) + _adopt_existing(plugin, contract, actions, str(workdir), env, logger) plan = runner.tofu_plan(str(workdir), env=env) @@ -888,6 +893,55 @@ def _guard_packaging_transitions( ) +def _guard_catalog_moves( + plugin: Any, + contract: Mapping[str, Any], + provider: str, + workdir: str, + env: Mapping[str, str], + logger: logging.Logger, + *, + actions: Any = (), +) -> None: + """Fail closed when the plan would destroy Glue resources of an Iceberg table + that now lives in another catalog. + + Thin CLI adapter over ``iac.catalog_moves.guard_catalog_moves``, the shape + of :func:`_guard_packaging_transitions`: detection and remediation text live + in ``iac/``, and this side owns the ``CLIError`` and the audit event. + + A no-op off AWS, for every contract without an AWS Iceberg expose in a + non-Glue catalog (checked before the state is even listed), and for a + fresh workdir. A probe that fails is logged and skipped rather than failing + the apply: the data-loss gate still stands behind it. + """ + if provider != "aws": + return + from fluid_build.iac.catalog_moves import ( + CatalogMoveError, + guard_catalog_moves, + moved_iceberg_exposes, + ) + + if not moved_iceberg_exposes(contract): + return + state = runner.tofu_state_list(workdir, env=env) + if not state: + return + try: + guard_catalog_moves(plugin, contract, state, workdir=workdir, actions=actions) + except CatalogMoveError as exc: + # Structured audit event BEFORE the raise, as for a packaging transition. + info(logger, "iceberg_catalog_move_blocked", **exc.event_fields()) + raise CLIError( + 1, + "iceberg_catalog_move_blocked", + {"kind": exc.kind, "error": str(exc), "remediation": list(exc.remediation)}, + ) + except Exception as exc: # defensive: a failed probe must not fail the apply + logger.debug("opentofu: catalog-move probe failed: %s", exc) + + def _report_suppressed_drift( plugin: Any, contract: Mapping[str, Any], diff --git a/fluid_build/cli/_diff_live.py b/fluid_build/cli/_diff_live.py index 4a3ad909..eca88b6d 100644 --- a/fluid_build/cli/_diff_live.py +++ b/fluid_build/cli/_diff_live.py @@ -755,14 +755,29 @@ def _inspect_glue( base: ExposeLiveResult, contract: Mapping[str, Any], ) -> ExposeLiveResult: - from fluid_build.iac.providers.aws import _GLUE_CATALOG_FORMATS + from fluid_build.iac.providers.aws import _glue_cataloged + from fluid_build.providers._iceberg_catalog import ( + binding_catalog_kind, + is_glue_cataloged, + ) loc = binding.get("location") or {} database, table = loc.get("database"), loc.get("table") # Same default as ``AwsIacPlugin.emit``: a binding with no format is # provisioned as parquet. fmt = str(binding.get("format") or "parquet").lower() - if not (database and table) or fmt not in _GLUE_CATALOG_FORMATS: + if not is_glue_cataloged(binding): + # An Iceberg table in Lakekeeper (or another REST / Nessie catalog): + # apply creates no Glue table for it, and a Glue table of the same name + # would be someone else's, so neither "absent" nor a comparison with it + # would be true. No AWS call. + base.status = NOT_CHECKED + base.detail = ( + f"table lives in Iceberg catalog {binding_catalog_kind(binding)}; " + "Glue is not inspected" + ) + return base + if not (database and table) or not _glue_cataloged(binding, fmt): # Apply creates a Glue table only for a file/lakehouse format with a # database and a table; anything else is a bare bucket prefix. base.status = NOT_CHECKED diff --git a/fluid_build/cli/contract_validation.py b/fluid_build/cli/contract_validation.py index b65079d3..2525e081 100644 --- a/fluid_build/cli/contract_validation.py +++ b/fluid_build/cli/contract_validation.py @@ -947,6 +947,8 @@ def _validate_against_actual_resource(self, expose: Dict[str, Any], path: str) - location = binding.get("location", {}) props = location.get("properties", {}) self._validate_bigquery_resource(expose, path, props) + elif self.provider_name == "aws" and self._iceberg_outside_glue(expose, path): + return elif self.provider_name in ("snowflake", "aws", "local"): self._validate_generic_resource(expose, path) else: @@ -1083,6 +1085,32 @@ def _validate_bigquery_resource( documentation_url="https://agenticstiger.github.io/forge_docs/advanced/production-troubleshooting.html", ) + def _iceberg_outside_glue(self, expose: Dict[str, Any], path: str) -> bool: + """Report an Iceberg table cataloged outside Glue as not checked. + + The AWS provider reads AWS Glue, and a Lakekeeper/REST/Polaris table is + not there: ``fluid apply`` creates no Glue table for it. Looking it up + anyway would fail every such contract with "does not exist in AWS Glue + catalog". Say what was not checked, and why, instead. + """ + from fluid_build.providers._iceberg_catalog import ( + binding_catalog_kind, + is_glue_cataloged, + ) + + binding = expose.get("binding") + if not isinstance(binding, dict) or is_glue_cataloged(binding): + return False + self.report.add_issue( + "info", + "binding", + f"Table not checked: it lives in Iceberg catalog " + f"'{binding_catalog_kind(binding)}', not AWS Glue, and the AWS " + "validation provider reads Glue", + path, + ) + return True + def _validate_generic_resource(self, expose: Dict[str, Any], path: str) -> None: """Validate a resource using the active validation provider (Snowflake, AWS, local).""" if not self.validation_provider: diff --git a/fluid_build/cli/validate.py b/fluid_build/cli/validate.py index a07816fa..da7dcade 100644 --- a/fluid_build/cli/validate.py +++ b/fluid_build/cli/validate.py @@ -736,8 +736,15 @@ def _run_contract_rules( # --- Iceberg streaming-sink checks (RFC-streaming-extension §6.8) -- # Catch the connector's silent-fail-at-first-record traps at validate # time: a sink with no Iceberg expose, the v1-deferred upsert mode, - # dynamic routing without a route field, an incomplete REST catalog, or - # an operator warehouse override that diverges from the binding. + # dynamic routing without a route field, a catalog kind the shared + # table does not know or whose required location keys are missing, a + # sink.catalog that disagrees with the expose, or an operator override + # that moves the warehouse or sets both catalog selectors. + # + # This gate and the two below fail CLOSED: a crash inside one is an + # error, not a --verbose-only note. It used to be the latter, so a + # validator that raised on some contract shape passed that contract + # as valid with the gate silently switched off. try: from fluid_build.build_runners.kafka_connect.iceberg_sink_validation import ( validate_iceberg_sink, @@ -749,9 +756,12 @@ def _run_contract_rules( validation_result.is_valid = False for msg in ice_warnings: validation_result.add_warning(msg) - except Exception as exc: # pragma: no cover — defensive - if args.verbose: - info(logger, f"Iceberg sink check skipped: {exc}") + except Exception as exc: # noqa: BLE001 — reported, never swallowed + validation_result.add_error( + f"Iceberg sink check could not run ({type(exc).__name__}: {exc}); " + "the contract was NOT checked for streaming-sink defects" + ) + validation_result.is_valid = False # --- Confluent Tableflow binding checks (RFC-streaming-extension §15) -- # A confluent-bound Iceberg expose carries hard Tableflow inputs @@ -767,9 +777,12 @@ def _run_contract_rules( validation_result.is_valid = False for msg in cf_warnings: validation_result.add_warning(msg) - except Exception as exc: # pragma: no cover — defensive - if args.verbose: - info(logger, f"Confluent binding check skipped: {exc}") + except Exception as exc: # noqa: BLE001 — reported, never swallowed + validation_result.add_error( + f"Confluent binding check could not run ({type(exc).__name__}: {exc}); " + "the contract was NOT checked for Tableflow binding defects" + ) + validation_result.is_valid = False # --- Iceberg prerequisite checks (anti-no-op gate) ------------------ # The Snowflake and GCP IaC emitters are emit-when-derivable: an @@ -786,9 +799,12 @@ def _run_contract_rules( validation_result.is_valid = False for msg in ice_warnings: validation_result.add_warning(msg) - except Exception as exc: # pragma: no cover (defensive) - if args.verbose: - info(logger, f"Iceberg binding check skipped: {exc}") + except Exception as exc: # noqa: BLE001 — reported, never swallowed + validation_result.add_error( + f"Iceberg binding check could not run ({type(exc).__name__}: {exc}); " + "the contract was NOT checked for Iceberg prerequisite defects" + ) + validation_result.is_valid = False # --- GCP binding checks (anti-no-op gate) --------------------------- # The GCP IaC emitter resolves each expose to a BigQuery / GCS / diff --git a/fluid_build/engines/dbt/catalogs_yml.py b/fluid_build/engines/dbt/catalogs_yml.py index 3331fb64..5ef02868 100644 --- a/fluid_build/engines/dbt/catalogs_yml.py +++ b/fluid_build/engines/dbt/catalogs_yml.py @@ -47,30 +47,47 @@ here, and emitting a guessed shape is worse than emitting nothing, so it returns ``None`` and the file is simply not written. -**Zero drift.** Catalog identity comes from -:func:`fluid_build.providers._iceberg_catalog.resolve_iceberg_catalog`, the same -resolver the streaming Iceberg sink uses, so the dbt catalog and the Kafka -Connect / Debezium sinks can never disagree about which table they mean. The +**Zero drift.** Which catalog an expose lives in comes from the catalog-kind +table in :mod:`fluid_build.providers._iceberg_catalog` +(:func:`~fluid_build.providers._iceberg_catalog.binding_catalog_kind` read +through :func:`~fluid_build.providers._iceberg_catalog.catalog_kind_info`), the +same table the streaming Iceberg sink's resolver and the AWS / Snowflake IaC +emitters read, so dbt, the Kafka Connect / Debezium sinks and the provisioned +infrastructure can never disagree about which catalog a table is in. The external volume name comes from :func:`fluid_build.providers._iceberg_catalog.iceberg_external_volume_name`, which the Snowflake IaC emitter also calls, so the volume FLUID creates is exactly the volume this file references. + +**Unreachable catalogs.** An expose whose catalog the adapter has no +integration for (Hive, JDBC, Hadoop or DynamoDB on Snowflake; anything but +BigLake metastore on BigQuery) is omitted with a warning rather than mapped to +the nearest shape. A guessed integration has dbt write a second table under +the name the catalog already owns; ``fluid validate`` is where the combination +is refused. """ from __future__ import annotations import hashlib +import logging from typing import Any, Dict, List, Mapping, Optional import yaml from fluid_build.providers._iceberg_catalog import ( - EXTERNAL_ICEBERG_CATALOGS, + FAMILY_GLUE, + CatalogKind, + binding_catalog_kind, + canonical_catalog_kind, + catalog_kind_info, iceberg_external_volume_name, iceberg_storage_uri, - resolve_iceberg_catalog, + known_catalog_kinds, ) +logger = logging.getLogger(__name__) + _HEADER = ( "# Generated by fluid generate. Do not edit manually.\n# Regenerate with: fluid generate\n\n" ) @@ -84,18 +101,26 @@ _CATALOG_TYPE_BUILT_IN = "built_in" _CATALOG_TYPE_REST = "iceberg_rest" -#: The external-vs-managed split. Shared with the Snowflake IaC emitter via -#: ``_iceberg_catalog.EXTERNAL_ICEBERG_CATALOGS``: the two sides must -#: partition identically or this file references a volume the IaC never -#: creates. Anything in the set maps to ``iceberg_rest``; an absent or -#: unlisted catalog maps to ``built_in``. -_EXTERNAL_CATALOGS = EXTERNAL_ICEBERG_CATALOGS +#: The external-vs-managed split is not decided in this module. It is the +#: ``snowflake_catalog_type`` column of the catalog-kind table, which the +#: Snowflake IaC emitter partitions on too: ``iceberg_rest`` for a catalog +#: external to Snowflake, ``built_in`` for Snowflake-managed (including an +#: absent catalog on ``platform: snowflake``, and an unknown value, the +#: historic fallback ``fluid validate`` reports), and ``None`` for a catalog +#: Snowflake has no integration for. The two sides must partition identically +#: or this file references a volume the IaC never creates. A hand-kept set here +#: is how ``catalog: lakekeeper`` used to fall through to ``built_in`` while +#: the streaming sink wrote the same table over REST. #: BigQuery's only catalog type. dbt's docs are explicit that the metastore #: itself needs no setup because it is built into BigQuery: the prerequisite #: is the GCS bucket, which the GCP IaC emitter creates. _CATALOG_TYPE_BIGLAKE = "biglake_metastore" +#: The catalog kind ``biglake_metastore`` IS. dbt-bigquery cannot point a +#: model at any other Iceberg catalog, so an expose naming one is omitted. +_BIGLAKE_KIND = "bigquery" + #: Adapters whose shape is verified against dbt's documentation. Databricks #: is deliberately absent. _SUPPORTED_ADAPTERS = frozenset({"snowflake", "bigquery"}) @@ -128,35 +153,60 @@ def _catalog_names(expose_id: str, seen: set) -> tuple[str, str]: return f"{safe}_catalog", f"{safe}_write_integration" -def _adapter_properties( - resolved: Any, - location: Mapping[str, Any], - *, - catalog_type: str, -) -> Dict[str, Any]: +def _adapter_properties(info: CatalogKind, location: Mapping[str, Any]) -> Dict[str, Any]: """Only properties we can actually derive. Never invent defaults. dbt applies its own defaults for anything omitted, so emitting a guess here would silently override the adapter rather than defer to it. """ props: Dict[str, Any] = {} - if catalog_type != _CATALOG_TYPE_REST: + if info.snowflake_catalog_type != _CATALOG_TYPE_REST: return props # A REST catalog on Snowflake is reached through a catalog-linked database. linked = location.get("glue_database") or location.get("database") if linked: props["catalog_linked_database"] = str(linked) - if str(getattr(resolved, "catalog_type", "") or "").lower() == "glue": + # Glue only. dbt documents ``glue`` as the one value, and dbt-snowflake's + # create macro branches on exactly ``== 'glue'`` (Glue needs a multi-step + # create instead of CTAS). Its v2-catalog translation also maps unity to + # ``unity``, but no macro reads any other value, so emitting it would add + # an undocumented key that changes nothing. + if info.family == FAMILY_GLUE: props["catalog_linked_database_type"] = "glue" return props +def _reachable_kinds(adapter: str) -> str: + """The catalog kinds ``adapter`` can write through, for a warning.""" + if adapter == "bigquery": + return f"{_BIGLAKE_KIND} (or omit location.catalog)" + kinds = [ + k + for k in known_catalog_kinds() + if canonical_catalog_kind(k) == k and catalog_kind_info(k).snowflake_catalog_type + ] + return ", ".join(kinds) + " (or omit location.catalog for a Snowflake-managed table)" + + +def _warn_unreachable(expose_id: str, kind: str, adapter: str) -> None: + logger.warning( + "catalogs.yml: omitting expose %r: dbt-%s has no catalog integration for " + "location.catalog %r, and a guessed one would write a second table under " + "the name that catalog owns. Use one of: %s.", + expose_id, + adapter, + kind, + _reachable_kinds(adapter), + ) + + def _bigquery_write_integration( contract: Mapping[str, Any], binding: Mapping[str, Any], *, integration_name: str, + expose_id: str = "", ) -> Optional[Dict[str, Any]]: """BigQuery's shape: ``biglake_metastore`` over a ``gs://`` URI. @@ -164,7 +214,19 @@ def _bigquery_write_integration( the name of an object that must already exist, and ``file_format`` is required. Returns ``None`` when no bucket is derivable, since a ``biglake_metastore`` integration without storage is not usable. + + Also returns ``None`` when ``location.catalog`` names a catalog other than + BigLake metastore. ``biglake_metastore`` is dbt-bigquery's only catalog + type, so emitting it for ``catalog: lakekeeper`` had dbt create a BigLake + table under the name the streaming sink commits to in Lakekeeper. This + reads the explicit value only, not :func:`binding_catalog_kind`: an absent + catalog has always meant BigLake here, whatever the platform default says. """ + kind = canonical_catalog_kind((binding.get("location") or {}).get("catalog")) + if kind and kind != _BIGLAKE_KIND: + _warn_unreachable(expose_id, kind, "bigquery") + return None + storage_uri = iceberg_storage_uri(binding, scheme="gs") if not storage_uri: return None @@ -182,14 +244,28 @@ def _write_integration( expose: Mapping[str, Any], *, integration_name: str, -) -> Dict[str, Any]: + expose_id: str = "", +) -> Optional[Dict[str, Any]]: + """Snowflake's shape, ``built_in`` or ``iceberg_rest`` per the kind table. + + Returns ``None`` for a catalog Snowflake has no integration for (Hive, + JDBC, Hadoop, DynamoDB), which the caller omits exactly as it omits an + underivable BigQuery expose. + """ binding = expose.get("binding") or {} location = binding.get("location") or {} - resolved = resolve_iceberg_catalog(binding, contract=contract) - - explicit_catalog = str(location.get("catalog") or "").lower() - is_external = explicit_catalog in _EXTERNAL_CATALOGS - catalog_type = _CATALOG_TYPE_REST if is_external else _CATALOG_TYPE_BUILT_IN + # The binding's kind, not the raw ``location.catalog`` string: it folds + # spelling (``Lakekeeper``, ``ICEBERG_REST``) and applies the platform + # default the IaC emitters apply. An absent catalog on ``platform: + # snowflake`` is Snowflake-managed (built_in). On ``platform: aws`` it is + # the Glue table the AWS IaC creates; ``built_in`` there named a volume + # nothing creates, since the Snowflake IaC only serves Snowflake exposes. + kind = binding_catalog_kind(binding) + info = catalog_kind_info(kind) + catalog_type = info.snowflake_catalog_type + if catalog_type is None: + _warn_unreachable(expose_id, kind, "snowflake") + return None integration: Dict[str, Any] = { "name": integration_name, @@ -203,7 +279,7 @@ def _write_integration( # bindingLocation is additionalProperties:false and has no slot for it. integration["external_volume"] = iceberg_external_volume_name(contract, binding) - props = _adapter_properties(resolved, location, catalog_type=catalog_type) + props = _adapter_properties(info, location) if props: integration["adapter_properties"] = props return integration @@ -244,14 +320,20 @@ def generate_catalogs_yml( catalog_name, integration_name = _catalog_names(expose_id, seen_names) if adapter == "bigquery": integration = _bigquery_write_integration( - contract, expose.get("binding") or {}, integration_name=integration_name + contract, + expose.get("binding") or {}, + integration_name=integration_name, + expose_id=expose_id, ) - if integration is None: - # No derivable bucket, so a biglake_metastore integration - # would name storage that does not exist. - continue else: - integration = _write_integration(contract, expose, integration_name=integration_name) + integration = _write_integration( + contract, expose, integration_name=integration_name, expose_id=expose_id + ) + if integration is None: + # No integration this adapter can express: a BigQuery expose with + # no derivable bucket (biglake_metastore would name storage that + # does not exist), or a catalog the adapter cannot reach. + continue catalogs.append( { "name": catalog_name, diff --git a/fluid_build/iac/catalog_moves.py b/fluid_build/iac/catalog_moves.py new file mode 100644 index 00000000..fbbe2279 --- /dev/null +++ b/fluid_build/iac/catalog_moves.py @@ -0,0 +1,226 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Pre-plan guard for an AWS Iceberg table whose catalog is not Glue. + +Until this release the AWS emitter ignored ``location.catalog`` and created a +Glue database and table for every Iceberg expose, ``catalog: lakekeeper`` +included, while the streaming sink wrote the table through Lakekeeper's REST +API. The emitter now creates no Glue resource for an Iceberg table another +catalog owns (``providers/aws.py::_glue_cataloged``). For a contract the old +release applied, those resources are in OpenTofu state and no longer in the +configuration, so the next ``tofu plan`` plans to DESTROY them. Destroying an +``aws_glue_catalog_database`` is Glue ``DeleteDatabase``, which deletes every +table in the database with it, tables created outside FLUID included. The +data-loss gate would stop that plan, but its remedy is ``--allow-data-loss``, +which is the one thing an operator must not do here. + +So, the same shape as the ownership-transition guard (``iac/transition.py``): +the change is diffed against prior state *before* ``tofu plan``, and the apply +fails closed with the ``tofu state rm`` commands that release this contract's +claim. ``state rm`` touches zero bytes in AWS; the Glue resources stay where +they are, for the operator to delete by hand once nothing reads them. + +Which addresses: the Glue resources the AWS plugin WOULD emit if every such +expose still named the Glue catalog, minus the ones it emits for the contract +as written (a Glue database a parquet expose still uses stays declared, so it +is not flagged), intersected with the state. The candidate list is the +plugin's own emit, so no address derivation is duplicated here. + +Pure except for the caller-supplied state listing and plugin; no ``tofu`` +shell-out, no ``cli`` imports (the CLI layer owns ``CLIError`` translation and +the audit event), as in ``iac/transition.py``. +""" + +from __future__ import annotations + +import copy +import shlex +from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple + +from ..providers._iceberg_catalog import binding_catalog_kind, is_glue_cataloged +from .provider_match import is_cloud + +__all__ = [ + "GLUE_CATALOG_RESOURCE_TYPES", + "CatalogMoveError", + "detect_catalog_moves", + "guard_catalog_moves", + "moved_iceberg_exposes", +] + +#: The resources the AWS emitter creates for a Glue-cataloged table, and the +#: only ones a catalog move removes from the configuration. Lake Formation +#: resources name the table too, but an Iceberg expose in another catalog has +#: its ``governance.lakeFormation`` refused at emit, before this guard runs. +GLUE_CATALOG_RESOURCE_TYPES: Tuple[str, ...] = ( + "aws_glue_catalog_database", + "aws_glue_catalog_table", +) + + +class CatalogMoveError(RuntimeError): + """Prior state holds Glue resources for an Iceberg table now in another catalog. + + ``kind`` is a stable, greppable tag (the ``PackagingTransitionError.kind`` + discipline); ``remediation`` carries the ``tofu state rm`` commands. + """ + + kind = "iceberg-catalog-move" + + def __init__( + self, + message: str, + *, + addresses: Sequence[str] = (), + exposes: Sequence[Tuple[str, str]] = (), + remediation: Sequence[str] = (), + ): + super().__init__(message) + self.addresses: Tuple[str, ...] = tuple(addresses) + self.exposes: Tuple[Tuple[str, str], ...] = tuple(exposes) + self.remediation: Tuple[str, ...] = tuple(remediation) + + def event_fields(self) -> Dict[str, Any]: + """Structured payload for the run record's audit event.""" + return { + "kind": self.kind, + "addresses": list(self.addresses), + "exposes": [{"expose": e, "catalog": k} for e, k in self.exposes], + "remediation": list(self.remediation), + } + + +def _moves_off_glue(binding: Any) -> bool: + """An AWS Iceberg binding with a Glue database whose table lives elsewhere. + + Only these change between the old emit and the new one: everything else + Glue catalogs (a parquet table, an Iceberg table with no ``catalog`` or + ``catalog: glue``) gets exactly the Glue resources it got before. + """ + if not isinstance(binding, Mapping) or not is_cloud(binding, "aws"): + return False + loc = binding.get("location") + if not isinstance(loc, Mapping) or not loc.get("database"): + return False + return not is_glue_cataloged(binding) + + +def moved_iceberg_exposes(contract: Mapping[str, Any]) -> Tuple[Tuple[str, str], ...]: + """``(expose id, catalog kind)`` for every AWS Iceberg expose not in Glue. + + Empty for every contract the guard cannot concern, which is the fast path: + the apply engine reads no state and emits nothing for those. + """ + out: List[Tuple[str, str]] = [] + for index, exposure in enumerate(contract.get("exposes") or []): + if not isinstance(exposure, Mapping): + continue + binding = exposure.get("binding") + if _moves_off_glue(binding): + expose_id = exposure.get("exposeId") or exposure.get("id") or index + out.append((str(expose_id), binding_catalog_kind(binding))) + return tuple(out) + + +def _as_glue(contract: Mapping[str, Any]) -> Dict[str, Any]: + """A copy of ``contract`` in which every moved expose names the Glue catalog. + + What the previous release emitted for it: that release created the Glue + resources whatever ``location.catalog`` said. + """ + doc = copy.deepcopy(dict(contract)) + for exposure in doc.get("exposes") or []: + if isinstance(exposure, Mapping) and _moves_off_glue(exposure.get("binding")): + exposure["binding"]["location"]["catalog"] = "glue" + return doc + + +def _glue_addresses(resources: Mapping[str, Any]) -> set: + return { + f"{resource_type}.{name}" + for resource_type in GLUE_CATALOG_RESOURCE_TYPES + for name in (resources.get(resource_type) or {}) + } + + +def detect_catalog_moves( + plugin: Any, + contract: Mapping[str, Any], + state_addresses: Iterable[str], + *, + actions: Iterable[Mapping[str, Any]] = (), +) -> Tuple[str, ...]: + """The Glue addresses in state that the next plan would destroy, sorted. + + Matched exactly against ``tofu state list``: the module ``fluid`` writes + has no child modules, so a ``module.``-prefixed address is not one its + configuration declares and this emit change cannot be what removes it. + + A provable no-op (no emit at all) for a contract with no AWS Iceberg expose + in another catalog, and for an empty state. + """ + if not moved_iceberg_exposes(contract): + return () + in_state = {a.strip() for a in state_addresses or () if isinstance(a, str) and a.strip()} + if not in_state: + return () + actions = list(actions or ()) + was = _glue_addresses(plugin.emit(_as_glue(contract), actions)) + now = _glue_addresses(plugin.emit(contract, actions)) + return tuple(sorted((was - now) & in_state)) + + +def _state_rm_commands(addresses: Sequence[str], *, workdir: Optional[str]) -> Tuple[str, ...]: + """Copy-pasteable ``tofu state rm`` commands, as ``transition.state_rm_commands``.""" + chdir = f" -chdir={shlex.quote(workdir)}" if workdir else "" + return tuple(f"tofu{chdir} state rm {shlex.quote(a)}" for a in addresses) + + +def guard_catalog_moves( + plugin: Any, + contract: Mapping[str, Any], + state_addresses: Iterable[str], + *, + workdir: Optional[str] = None, + actions: Iterable[Mapping[str, Any]] = (), +) -> None: + """Raise :class:`CatalogMoveError` when the plan would destroy Glue resources of a + moved table; a no-op otherwise (see :func:`detect_catalog_moves`).""" + addresses = detect_catalog_moves(plugin, contract, state_addresses, actions=actions) + if not addresses: + return + exposes = moved_iceberg_exposes(contract) + commands = _state_rm_commands(addresses, workdir=workdir) + raise CatalogMoveError( + "iceberg catalog move blocked — this contract's OpenTofu state holds " + f"{len(addresses)} Glue catalog resource(s) for Iceberg table(s) that now live in " + "another catalog:\n\n" + + "\n".join(f" exposes[{e}]: location.catalog {k}" for e, k in exposes) + + "\n\n" + + "\n".join(f" {a}" for a in addresses) + + "\n\nforge-cli no longer creates a Glue table for an Iceberg table in a non-Glue " + "catalog (it was a second claim on the table's name, without its metadata), so " + "applying now would plan to DESTROY these resources, and destroying a Glue " + "database deletes every table in it, including tables created outside FLUID.\n\n" + "Drop each from this contract's state, then re-run apply:\n\n" + + "\n".join(f" {command}" for command in commands) + + "\n\n`tofu state rm` touches ZERO bytes of infrastructure: the resources stay in " + "AWS, and only this contract's claim on them is released. Delete a Glue table by " + "hand once nothing reads it, and a Glue database only once it holds nothing else. If " + "the table belongs in Glue, remove location.catalog (or set it to 'glue') instead.", + addresses=addresses, + exposes=exposes, + remediation=commands, + ) diff --git a/fluid_build/iac/governance_validation.py b/fluid_build/iac/governance_validation.py index c1cdf543..75cafa9e 100644 --- a/fluid_build/iac/governance_validation.py +++ b/fluid_build/iac/governance_validation.py @@ -21,7 +21,8 @@ a BigQuery table, and a Lake Formation tag the module cannot create or associate as declared (a definition with no values, two definitions that are one resource, a value the contract's definition of the tag does not allow, tags on a binding with no -Glue table); ``fluid plan`` does not run the emitter. This runs the SAME derivations +Glue table), and Lake Formation on an Iceberg table that lives in a catalog other +than Glue; ``fluid plan`` does not run the emitter. This runs the SAME derivations (``iac/providers/gcp_governance.py``, ``lf_governance`` and ``lf_tag_associations`` in ``iac/providers/aws.py``, ``iac/column_access.py``, ``iac/principals.py``) so ``fluid validate`` reports each refusal at stage 2, with the same message, instead of @@ -32,6 +33,7 @@ from typing import Any, List, Mapping, Tuple +from ..providers._iceberg_catalog import binding_catalog_kind, is_glue_cataloged from .access import normalize_access_grants from .base import UnsupportedBindingError from .principals import gcp_grants, principal_map @@ -69,17 +71,25 @@ def _lf_tag_definition_errors(contract: Mapping[str, Any]) -> List[str]: def _aws_refusal( contract: Mapping[str, Any], exposure: Mapping[str, Any], binding: Mapping[str, Any], index: int ) -> str: - """What the AWS emitter refuses this expose with, or ``""``: ``lf_governance``, - then ``lf_tag_associations``, in the emitter's order. + """What the AWS emitter refuses this expose with, or ``""``: Lake Formation on + a table outside Glue, ``lf_governance``, then ``lf_tag_associations``, in the + emitter's order. A name the masked view's SQL cannot quote raises a plain ``ValueError`` (``validate_ident``), on which the emitter stops too. It is reported here as the expose's error: escaping, it would end :func:`validate_governance`, and ``fluid validate`` would report no governance finding at all. """ - from .providers.aws import lf_governance, lf_tag_associations + from .providers.aws import ( + lf_governance, + lf_tag_associations, + refuse_lake_formation_on_external_catalog, + ) try: + # Lake Formation on an Iceberg table another catalog owns would + # otherwise be dropped at apply: refused first, as the emitter does. + refuse_lake_formation_on_external_catalog(exposure, binding, index) lf_governance(exposure, binding, index) lf_tag_associations(contract, exposure, binding, index) except UnsupportedBindingError as exc: @@ -104,6 +114,9 @@ def validate_governance(contract: Mapping[str, Any]) -> Tuple[List[str], List[st warnings: List[str] = [] grants = normalize_access_grants(contract) unenforced: List[str] = [] + # AWS Iceberg tables in another catalog: Lake Formation cannot reach them, + # so the remedy the warning below gives would itself be refused. + elsewhere: List[str] = [] try: _gov.refuse_mixed_dataset_encryption(contract) except UnsupportedBindingError as exc: @@ -122,7 +135,11 @@ def validate_governance(contract: Mapping[str, Any]) -> Tuple[List[str], List[st if refusal: errors.append(refusal) elif not _lf_grants(binding): - unenforced.append(str(exposure.get("exposeId") or index)) + expose_id = str(exposure.get("exposeId") or index) + if is_glue_cataloged(binding): + unenforced.append(expose_id) + else: + elsewhere.append(f"{expose_id} ({binding_catalog_kind(binding)})") continue if not _gov.gcp_owned(binding): continue @@ -148,6 +165,13 @@ def validate_governance(contract: Mapping[str, Any]) -> Tuple[List[str], List[st "governance.lakeFormation.grants, the AWS form of who may read the table. Add " "them to the aws overlay's binding (column restrictions then narrow them)." ) + if grants and elsewhere: + warnings.append( + f"accessPolicy.grants are not enforced on aws binding(s) {', '.join(elsewhere)}: " + "their Iceberg tables live in a catalog other than Glue, and the AWS emitter " + "writes access only as Lake Formation grants on a Glue table. Grant access in " + "that catalog." + ) return errors, warnings diff --git a/fluid_build/iac/iceberg_validation.py b/fluid_build/iac/iceberg_validation.py index 3850f21d..24fed8b7 100644 --- a/fluid_build/iac/iceberg_validation.py +++ b/fluid_build/iac/iceberg_validation.py @@ -25,7 +25,20 @@ validator explains. The two MUST agree about what counts as derivable, or the gate either blocks a contract that would have worked or waves through one that silently emits nothing. Every check here mirrors a specific skip -branch, and the tests assert the pairing in both directions. +branch, and the tests assert the pairing in both directions: an ERROR marks a +skip the user can fix, a WARNING marks a skip that is deliberate (a catalog +whose Snowflake integration needs a secret the credential-free module cannot +carry), and a clean result means something was emitted. + +The emitters and this gate classify ``location.catalog`` through ONE table, +``_iceberg_catalog.catalog_kind_info``. Before it, each side lowered the raw +string and compared it to its own hand-kept list, so ``catalog: lakekeeper`` +was Snowflake-managed to the IaC (which built an EXTERNAL VOLUME for it) and +got a false "needs an s3:// or gs:// warehouse" error here, while the +streaming sink wrote the same table over REST. The one check with no skip +branch to mirror is the unknown-kind refusal: every emitter keeps a historic +fallback for a value it does not know, and the fallbacks disagree, so the +gate refuses the value itself. What is deliberately NOT checked: whether a given catalog accepts, requires or forbids a client-supplied storage location. dbt's own EPIC (dbt-labs/ @@ -78,12 +91,63 @@ def validate_iceberg_bindings( """ errors: List[str] = [] warnings: List[str] = [] + _check_catalog_kinds(contract, errors) _check_snowflake(contract, errors, warnings) _check_volume_collisions(contract, errors) _check_gcp(contract, errors, warnings) return errors, warnings +def _check_catalog_kinds(contract: Mapping[str, Any], errors: List[str]) -> None: + """A ``location.catalog`` value outside the shared kind table, on any platform. + + No emitter skips an unknown kind; each falls back, and the fallbacks + disagree. The streaming sink writes it over the REST protocol, while dbt + ``catalogs.yml`` and the Snowflake IaC treat it as Snowflake-managed and + create an EXTERNAL VOLUME for it. A typo (``lakekeper``) therefore streams + into one catalog while dbt writes a second table somewhere else, which is + the bug class the kind table exists to end. The fallbacks are what make an + unknown value dangerous, so the gate refuses the value itself, once, and + the per-platform checks below stay silent about it. + + Confluent exposes are left to ``validate_confluent_binding``, which + refuses every non-Glue catalog: listing the REST kinds here would point + the user at values that gate rejects too. + """ + from fluid_build.providers._iceberg_catalog import ( + FAMILY_UNKNOWN, + canonical_catalog_kind, + catalog_kind_info, + is_iceberg_format, + known_catalog_kinds, + ) + + for exposure in contract.get("exposes") or []: + if not isinstance(exposure, Mapping): + continue + binding = exposure.get("binding") or {} + if not isinstance(binding, Mapping) or is_cloud(binding, "confluent"): + continue + # The streaming sink's format predicate, which is the widest of the + # emitters': every spelling some emitter reads a catalog for is checked. + if not is_iceberg_format(binding.get("format")): + continue + loc = binding.get("location") or {} + raw = loc.get("catalog") if isinstance(loc, Mapping) else None + if not canonical_catalog_kind(raw): + continue + if catalog_kind_info(raw).family == FAMILY_UNKNOWN: + eid = exposure.get("exposeId") or exposure.get("id") or "?" + errors.append( + f"expose '{eid}': binding.location.catalog '{raw}' is not a catalog " + f"kind FLUID knows. Use one of: {', '.join(known_catalog_kinds())}. " + "Each emitter guesses differently for an unknown value (the " + "streaming sink writes over Iceberg REST, dbt and the Snowflake IaC " + "treat it as Snowflake-managed), so the table would be written to " + "two different catalogs." + ) + + def _check_volume_collisions(contract: Mapping[str, Any], errors: List[str]) -> None: """Two exposes deriving one volume name with different storage. @@ -92,15 +156,22 @@ def _check_volume_collisions(contract: Mapping[str, Any], errors: List[str]) -> "must never be quiet". Raising mid-emit is loud but late: it surfaces at ``fluid apply`` rather than ``fluid validate``. Catch it here so the contract is rejected before anyone provisions anything. + + Only a kind the emitter builds a volume for is checked, which is every + ``built_in`` row: absent, ``snowflake`` and an unknown value alike. This + used to skip ANY non-empty ``catalog``, so two ``catalog: snowflake`` + exposes on different storage passed validate and then raised mid-emit. """ from fluid_build.providers._iceberg_catalog import ( + binding_catalog_kind, + catalog_kind_info, iceberg_external_volume_name, iceberg_storage_uri, ) seen: dict = {} for eid, binding, loc in _iceberg_exposures(contract, "snowflake"): - if str(loc.get("catalog") or "").lower(): + if catalog_kind_info(binding_catalog_kind(binding)).snowflake_catalog_type != "built_in": continue try: name = iceberg_external_volume_name(contract, binding) @@ -125,9 +196,18 @@ def _check_volume_collisions(contract: Mapping[str, Any], errors: List[str]) -> def _check_snowflake(contract: Mapping[str, Any], errors: List[str], warnings: List[str]) -> None: - """Mirror ``snowflake._emit_iceberg_prereqs``'s skip branches.""" + """Mirror ``snowflake._emit_iceberg_prereqs``'s skip branches. + + Both sides switch on the same row of the kind table: Glue gets the + integration checks, ``iceberg_rest`` kinds the deferred warning, a kind + Snowflake cannot integrate at all an error, and ``built_in`` the EXTERNAL + VOLUME checks. + """ from fluid_build.providers._iceberg_catalog import ( - EXTERNAL_ICEBERG_CATALOGS, + FAMILY_GLUE, + FAMILY_UNKNOWN, + binding_catalog_kind, + catalog_kind_info, iceberg_external_volume_is_override, iceberg_external_volume_name, iceberg_storage_provider, @@ -135,8 +215,9 @@ def _check_snowflake(contract: Mapping[str, Any], errors: List[str], warnings: L for eid, binding, loc in _iceberg_exposures(contract, "snowflake"): catalog = str(loc.get("catalog") or "").lower() + kind = catalog_kind_info(binding_catalog_kind(binding)) - if catalog == "glue": + if kind.family == FAMILY_GLUE: missing = [ label for key, label in ( @@ -155,7 +236,17 @@ def _check_snowflake(contract: Mapping[str, Any], errors: List[str], warnings: L ) continue - if catalog in EXTERNAL_ICEBERG_CATALOGS: + if kind.family == FAMILY_UNKNOWN: + # Refused once by _check_catalog_kinds. The managed checks below + # would also ask for storage that a mistyped external catalog + # (``lakekeper``) does not need, which reads as a second problem. + continue + + if kind.snowflake_catalog_type == "iceberg_rest": + # rest / lakekeeper / polaris / unity / nessie / bigquery. A + # Lakekeeper warehouse is a catalog NAME, so the managed checks + # below used to reject it with a false "needs an s3:// or gs:// + # warehouse" error while the emitter built a volume nobody used. warnings.append( f"expose '{eid}': catalog '{catalog}' is understood but FLUID does " "not emit its Snowflake CATALOG INTEGRATION yet, because that " @@ -164,6 +255,19 @@ def _check_snowflake(contract: Mapping[str, Any], errors: List[str], warnings: L ) continue + if kind.snowflake_catalog_type is None: + # hive / jdbc / hadoop / dynamodb: no CATALOG_SOURCE exists for + # them, so there is nothing to defer. The emitter emits nothing. + errors.append( + f"expose '{eid}': Snowflake has no catalog integration for a " + f"{kind.name} catalog (it integrates Glue, object storage, Open " + "Catalog/Polaris and Iceberg REST catalogs), so no table Snowflake " + "can read is reachable through it. Use catalog: glue, an Iceberg " + "REST catalog (rest, lakekeeper, polaris, unity), or omit catalog " + "for a Snowflake-managed table." + ) + continue + if iceberg_external_volume_is_override(binding): # Operator-owned volume: FLUID emits no CREATE, by design. The name # still has to be a legal identifier, because both the dbt @@ -203,19 +307,54 @@ def _check_snowflake(contract: Mapping[str, Any], errors: List[str], warnings: L def _check_gcp(contract: Mapping[str, Any], errors: List[str], warnings: List[str]) -> None: - """Mirror ``gcp._emit_iceberg_storage``'s skip branch.""" - from fluid_build.providers._iceberg_catalog import iceberg_bucket_name + """Mirror ``gcp._emit_iceberg_storage``'s skip branch. + + The BigLake errors (a warehouse that is not GCS, or no storage at all) + hold only for a BigQuery-cataloged table, the absent-``catalog`` default + included, because there dbt names the bucket and FLUID must create it. + An external catalog (``lakekeeper``, ``rest``, ``nessie``...) owns its + storage, and its warehouse is often a catalog NAME (``demo``), so those + errors used to reject a correct contract. For it, only an object-store + scheme other than ``gs://`` is wrong: a ``platform: gcp`` table cannot + live in S3 or ADLS. The kind is read from ``location.catalog`` itself, + not :func:`binding_catalog_kind`, whose non-AWS default is ``rest``. + """ + from fluid_build.providers._iceberg_catalog import ( + FAMILY_UNKNOWN, + canonical_catalog_kind, + catalog_kind_info, + iceberg_bucket_name, + is_object_store_uri, + ) for eid, binding, loc in _iceberg_exposures(contract, "gcp"): if iceberg_bucket_name(binding): continue warehouse = str(loc.get("warehouse") or "") + kind = canonical_catalog_kind(loc.get("catalog")) + external = bool(kind) and kind != "bigquery" + if external and catalog_kind_info(kind).family == FAMILY_UNKNOWN: + # Refused once by _check_catalog_kinds. + continue if warehouse.startswith("gs://"): errors.append( f"expose '{eid}': binding.location.warehouse is '{warehouse}', which " "names no bucket. Use gs:/// so FLUID can " "create the bucket dbt's catalogs.yml points at." ) + elif external: + # gcs:// is the alternate GCS spelling the shared FileIO table + # (_iceberg_catalog._OBJECT_STORE_FILE_IO) maps to GCSFileIO. + if is_object_store_uri(warehouse) and not warehouse.lower().startswith( + ("gs://", "gcs://") + ): + errors.append( + f"expose '{eid}': binding.location.warehouse is '{warehouse}', " + "which is not Google Cloud Storage. A platform: gcp Iceberg table " + f"in a {kind} catalog takes the catalog's warehouse name (the " + "catalog owns the storage) or a gs:/// " + "location FLUID creates the bucket for." + ) elif warehouse: errors.append( f"expose '{eid}': binding.location.warehouse is '{warehouse}', but a " diff --git a/fluid_build/iac/providers/aws.py b/fluid_build/iac/providers/aws.py index f90cfae0..b054869d 100644 --- a/fluid_build/iac/providers/aws.py +++ b/fluid_build/iac/providers/aws.py @@ -20,6 +20,11 @@ ``CREATE EXTERNAL SCHEMA`` bridge so Redshift queries the same Glue catalog via Spectrum. A pure function of the contract; no credentials, no network. +An Iceberg expose whose ``location.catalog`` names another catalog +(Lakekeeper, a REST catalog, Polaris, Nessie...) gets its S3 bucket but no Glue +database or table: the table lives in that catalog, which the streaming sink and +dbt write to as well (see :func:`_glue_cataloged`). + **Packaging modes (RFC-packaging-modes.md file 3).** ``resolve_packaging`` decides per container kind whether this contract owns the container: @@ -53,6 +58,14 @@ import yaml +from ...providers._iceberg_catalog import ( + FAMILY_UNKNOWN, + binding_catalog_kind, + catalog_kind_info, + is_glue_cataloged, + is_iceberg_format, + known_catalog_kinds, +) from ...providers._sql_safety import quote_string_literal, validate_ident from ...providers.aws.util import warehouse as _warehouse from .. import column_access @@ -402,6 +415,11 @@ def emit( loc = binding.get("location") or {} fmt = binding.get("format") or "parquet" schema = (exposure.get("contract") or {}).get("schema") or [] + # Lake Formation on an Iceberg table another catalog owns is + # refused before any of it is derived: every LF resource names the + # Glue table, and there is none (see :func:`_glue_cataloged`). + refuse_lake_formation_on_external_catalog(exposure, binding, index) + refuse_unknown_iceberg_catalog(exposure, binding, index) # The contract's column restrictions, as each Lake Formation grant's # excluded columns (``iac/column_access.py``), its row filters, each on # the read grant of its principal, and the protected views its masked @@ -424,6 +442,7 @@ def emit( schema, cid, tags, + binding=binding, contract=contract, placement=placement, labels=gov_labels, @@ -574,11 +593,13 @@ def _add(address: str, resource_id: str) -> None: seen.add(address) blocks.append(ImportBlock(to=address, id=resource_id)) - # Resolve catalog_id once — only when needed (Glue catalog refs). + # Resolve catalog_id once — only when needed (Glue catalog refs). The + # raw format here, not the emit's parquet default: unchanged from before + # the catalog half of the gate was added. contract_needs_catalog_id = any( is_cloud(b.get("binding") or {}, "aws") and ((b.get("binding") or {}).get("location") or {}).get("database") - and str(((b.get("binding") or {}).get("format")) or "").lower() in _GLUE_CATALOG_FORMATS + and _glue_cataloged(b.get("binding") or {}, (b.get("binding") or {}).get("format")) for b in contract.get("exposes") or [] ) catalog_id = _resolve_catalog_id() if contract_needs_catalog_id else "" @@ -597,8 +618,10 @@ def _add(address: str, resource_id: str) -> None: namespace = loc.get("namespace") or loc.get("workgroup") # Glue catalog resources — only file/lakehouse formats use - # the Glue catalog (mirrors ``_emit_glue``'s gate). - if database and catalog_id and str(fmt or "").lower() in _GLUE_CATALOG_FORMATS: + # the Glue catalog (mirrors ``_emit_glue``'s gate). An Iceberg + # table in another catalog is not imported: a Glue table of its + # name would be adopted into state as if the contract owned it. + if database and catalog_id and _glue_cataloged(binding, fmt): db_key = safe_ident(f"{cid}_{database}") if not placement.database_referenced: # provider id: ``{catalog_id}:{name}`` @@ -734,6 +757,92 @@ def provider_block_for(self, contract: Mapping[str, Any]) -> Dict[str, Any]: {"iceberg", "parquet", "csv", "json", "avro", "orc", "delta"} ) + +def _glue_cataloged(binding: Mapping[str, Any], fmt: Any) -> bool: + """THE gate of every resource that is, or names, the binding's Glue table. + + Two halves. ``fmt`` (as the caller resolved it) must be a format Glue + catalogs (:data:`_GLUE_CATALOG_FORMATS`), and an Iceberg table must also + live in the Glue catalog (``_iceberg_catalog.is_glue_cataloged``: no + ``location.catalog`` on AWS means Glue). A ``catalog: lakekeeper`` (REST, + Polaris, Unity, Nessie...) table lives in THAT catalog: the sink streams to + it over REST and dbt writes it there, while this emitter used to create a + second, metadata-less Glue table of the same name beside it. + + Every gate reads this one predicate: the Glue database and table + (:func:`_emit_glue`), their imports (:meth:`AwsIacPlugin.discover_imports`), + the Lake Formation grants, tags and filters (:func:`_emit_lakeformation`), + the LF bucket policy, the column checks and ``fluid diff``'s Glue + inspector. A Lake Formation resource that passes a gate the Glue table did + not references an undeclared ``aws_glue_catalog_table`` and ``tofu + validate`` fails. The S3 bucket is not gated: a REST catalog's storage + profile still needs it. + """ + return str(fmt or "").lower() in _GLUE_CATALOG_FORMATS and is_glue_cataloged(binding) + + +def refuse_lake_formation_on_external_catalog( + exposure: Mapping[str, Any], binding: Mapping[str, Any], index: int = 0 +) -> None: + """Refuse ``governance.lakeFormation`` on an Iceberg table another catalog owns. + + Lake Formation governs tables of the Glue Data Catalog. Every per-expose LF + resource names the Glue table (grants, LF-tags, data-cells filters) and + :func:`_glue_cataloged` emits none for a Lakekeeper or other REST-catalog + table, so the block would be dropped without a word: the grants a reviewer + read in the contract would control nothing. Called by + :meth:`AwsIacPlugin.emit` and by ``governance_validation.validate_governance``, + so it is refused at ``fluid validate`` with the message ``fluid apply`` gives. + """ + governance = binding.get("governance") + if not isinstance(governance, Mapping) or not governance.get("lakeFormation"): + return + if is_glue_cataloged(binding): + return + kind = binding_catalog_kind(binding) + raise UnsupportedBindingError( + "lake-formation-needs-glue-catalog", + f"exposes[{_expose_id(exposure) or index}] declares governance.lakeFormation, but " + f"its Iceberg table lives in the {kind!r} catalog (location.catalog), not in AWS " + "Glue. Lake Formation governs only Glue Data Catalog tables, so its grants, LF-tags " + "and filters would have no table to name and would not be applied.", + ( + f"Govern access in the {kind} catalog itself and remove governance.lakeFormation " + "from this binding.", + "Remove location.catalog (or set it to 'glue') to keep the table in the Glue " + "catalog under Lake Formation.", + ), + ) + + +def refuse_unknown_iceberg_catalog( + exposure: Mapping[str, Any], binding: Mapping[str, Any], index: int = 0 +) -> None: + """Refuse an Iceberg ``location.catalog`` value no row of the table knows. + + ``fluid validate`` already reports it, but ``fluid apply`` does not run + that gate, and on AWS the answer decides whether a Glue table exists: a + typo (``catalog: glu``) would otherwise apply with no Glue table and no + word. Fail closed with the accepted spellings instead. + """ + if not is_iceberg_format(binding.get("format")): + return + loc = binding.get("location") + raw = loc.get("catalog") if isinstance(loc, Mapping) else None + if not raw or catalog_kind_info(raw).family != FAMILY_UNKNOWN: + return + raise UnsupportedBindingError( + "unknown-iceberg-catalog", + f"exposes[{_expose_id(exposure) or index}] names Iceberg catalog {raw!r} " + "(location.catalog), which FLUID does not know, so it cannot tell whether the " + "table belongs in AWS Glue.", + ( + "Use one of: " + ", ".join(known_catalog_kinds()) + ".", + "Remove location.catalog to keep the table in the Glue catalog.", + ), + ) + + #: Hive storage classes per file format, for query engines (Athena, Spark) #: that read a Glue table through them. Iceberg is read through its metadata, #: not these, and is deliberately absent. @@ -792,6 +901,7 @@ def _emit_glue( cid: str, tags: Dict[str, str], *, + binding: Mapping[str, Any], contract: Optional[Mapping[str, Any]] = None, placement: _Placement = _LEGACY_PLACEMENT, labels: Optional[Mapping[str, str]] = None, @@ -801,8 +911,9 @@ def _emit_glue( return # Only file/lakehouse formats use the Glue catalog as their # storage-and-schema registry. Redshift-flavoured bindings (whose - # ``database`` is internal to the workgroup) skip this emit. - if str(fmt or "").lower() not in _GLUE_CATALOG_FORMATS: + # ``database`` is internal to the workgroup) skip this emit, and so does + # an Iceberg table another catalog owns (see :func:`_glue_cataloged`). + if not _glue_cataloged(binding, fmt): return db_name = safe_ident(f"{cid}_{database}") if not placement.database_referenced: @@ -1554,10 +1665,12 @@ def _wire_aws_deps(resources: Dict[str, Any], cid: str) -> None: # none for a tag owned outside the contract (its resource is not here). # - Empty governance blocks emit nothing — every existing contract # stays at zero LF surface area. -# - LF is Glue-catalog-backed, so the per-exposure emit only fires for -# formats in ``_GLUE_CATALOG_FORMATS``. Redshift / Kinesis / Lambda -# bindings ignore any governance.lakeFormation block by design (LF -# doesn't manage those resources). +# - LF is Glue-catalog-backed, so the per-exposure emit only fires where +# ``_glue_cataloged`` does. Redshift / Kinesis / Lambda bindings ignore +# any governance.lakeFormation block by design (LF doesn't manage those +# resources); an Iceberg table in another catalog has its block REFUSED +# (``refuse_lake_formation_on_external_catalog``), because there the +# author asked LF to govern a table that only looks like one it could. def _contract_uses_lakeformation(contract: Mapping[str, Any]) -> bool: @@ -1701,9 +1814,11 @@ def lf_tag_associations( return [] lf_tags = [{"key": str(k), "value": str(v)} for k, v in tags.items() if v] fmt = str(binding.get("format") or "parquet") - if not lf_tags or fmt.lower() not in _GLUE_CATALOG_FORMATS: - # Not a Glue-catalog format: _emit_lakeformation ignores the whole - # governance.lakeFormation block of such a binding, by design. + if not lf_tags or not _glue_cataloged(binding, fmt): + # Not a Glue-catalog table: _emit_lakeformation ignores the whole + # governance.lakeFormation block of such a binding, by design (an Iceberg + # table in another catalog is refused before this, by + # :func:`refuse_lake_formation_on_external_catalog`). return [] where = f"exposes[{exposure.get('exposeId') or index}] governance.lakeFormation.tags" loc = binding.get("location") or {} @@ -1931,14 +2046,14 @@ def _lf_bucket_policy( * ``all-grantees``: a statement for every ``arn:`` grantee — the emit before this field existed, byte for byte. - ``None`` when there is no ``governance.lakeFormation`` block, the format is - not Glue-catalog-backed, no database or bucket is bound, no grant names an - ``arn:`` principal, or the mode is ``none``. + ``None`` when there is no ``governance.lakeFormation`` block, the table is + not in the Glue catalog (:func:`_glue_cataloged`), no database or bucket is + bound, no grant names an ``arn:`` principal, or the mode is ``none``. """ gov = (binding.get("governance") or {}).get("lakeFormation") or {} if not gov: return None - if str(fmt or "").lower() not in _GLUE_CATALOG_FORMATS or not loc.get("database"): + if not _glue_cataloged(binding, fmt) or not loc.get("database"): return None mode = _lf_bucket_policy_mode(gov) principals = _lf_bucket_principals(gov) @@ -2189,12 +2304,31 @@ def lf_column_exclusions( The exclusions reach a grant only as ``table_with_columns`` on a Glue-catalog table, so a restriction on another format, or on a binding that names no database and table, would be dropped by :func:`_emit_lakeformation`. + + An Iceberg table in another catalog (Lakekeeper, REST...) is refused first, + by name: ``column_access`` would otherwise tell the author to add Lake + Formation grants, which :func:`refuse_lake_formation_on_external_catalog` + then refuses. """ + if column_access.restrictions_for(exposure, index) and not is_glue_cataloged(binding): + kind = binding_catalog_kind(binding) + raise UnsupportedBindingError( + "column-restriction-unenforceable", + f"exposes[{exposure.get('exposeId') or index}] restricts columns, but its Iceberg " + f"table lives in the {kind!r} catalog (location.catalog), not in AWS Glue; the AWS " + "emitter enforces column restrictions only through Lake Formation grants on a " + "Glue-catalog table, so nothing would enforce them.", + ( + f"Enforce the restriction in the {kind} catalog's own access control.", + "Remove location.catalog (or set it to 'glue') to keep the table in the Glue " + "catalog, where Lake Formation grants carry the restriction.", + ), + ) exclusions = column_access.lf_exclusions(exposure, binding, index) loc = binding.get("location") or {} fmt = str(binding.get("format") or "parquet") if exclusions is not None and ( - fmt.lower() not in _GLUE_CATALOG_FORMATS or not loc.get("database") or not loc.get("table") + not _glue_cataloged(binding, fmt) or not loc.get("database") or not loc.get("table") ): raise UnsupportedBindingError( "column-restriction-unenforceable", @@ -2225,9 +2359,26 @@ def lf_row_filters( if not filters: return {} where = f"exposes[{exposure.get('exposeId') or index}].policy.authz.rowFilters" + if not is_glue_cataloged(binding): + # Refused by name, as column restrictions are: the generic message below + # would send the author to Lake Formation grants, which + # :func:`refuse_lake_formation_on_external_catalog` then refuses. + kind = binding_catalog_kind(binding) + raise UnsupportedBindingError( + "row-filter-unenforceable", + f"{where} filters rows, but its Iceberg table lives in the {kind!r} catalog " + "(location.catalog), not in AWS Glue; the AWS emitter enforces row filters only " + "as Lake Formation data cells filters on a Glue-catalog table, so nothing would " + "enforce it.", + ( + f"Enforce the filter in the {kind} catalog's own access control.", + "Remove location.catalog (or set it to 'glue') to keep the table in the Glue " + "catalog, where a data cells filter carries the row filter.", + ), + ) loc = binding.get("location") or {} fmt = str(binding.get("format") or "parquet") - if fmt.lower() not in _GLUE_CATALOG_FORMATS or not loc.get("database") or not loc.get("table"): + if not _glue_cataloged(binding, fmt) or not loc.get("database") or not loc.get("table"): formats = sorted(_GLUE_CATALOG_FORMATS) raise UnsupportedBindingError( "row-filter-unenforceable", @@ -2527,8 +2678,10 @@ def _emit_lakeformation( return # LF only meaningfully manages access to Glue-catalog-backed formats # (file formats on S3). Redshift/Kinesis bindings have their own - # access-control models and are skipped here. - if str(fmt or "").lower() not in _GLUE_CATALOG_FORMATS: + # access-control models and are skipped here. An Iceberg table in another + # catalog never reaches this: ``emit`` refuses its LF block first + # (:func:`refuse_lake_formation_on_external_catalog`). + if not _glue_cataloged(binding, fmt): return database = loc.get("database") @@ -2888,7 +3041,7 @@ def _check_lf_grant_columns( (:func:`lf_column_exclusions`): what the emitter writes, so what is checked. """ gov = (binding.get("governance") or {}).get("lakeFormation") or {} - if not gov or str(fmt or "").lower() not in _GLUE_CATALOG_FORMATS: + if not gov or not _glue_cataloged(binding, fmt): return if not loc.get("database") or not loc.get("table"): return diff --git a/fluid_build/iac/providers/confluent.py b/fluid_build/iac/providers/confluent.py index 94f819ca..9fbd0e30 100644 --- a/fluid_build/iac/providers/confluent.py +++ b/fluid_build/iac/providers/confluent.py @@ -40,6 +40,7 @@ from typing import Any, Dict, Iterable, List, Mapping, Tuple +from ...providers._iceberg_catalog import binding_catalog_kind, canonical_catalog_kind from ...providers.aws.util.warehouse import normalize_location from ..importer import ImportBlock from ..naming import safe_ident, tofu_ref @@ -70,6 +71,28 @@ def _topic_name(loc: Mapping[str, Any], exposure: Mapping[str, Any]) -> str: return loc.get("topic") or loc.get("table") or exposure.get("exposeId") or "topic" +def _non_glue_catalog(binding: Mapping[str, Any]) -> str: + """The canonical kind of an explicit, non-Glue ``location.catalog``, else ``""``. + + FLUID's Tableflow emitter publishes only to AWS Glue (the + ``aws_glue`` block of ``confluent_catalog_integration``). Tableflow's + Snowflake Open Catalog (Polaris) and Unity Catalog integrations + authenticate with a client secret this credential-free module cannot + carry, and Tableflow has no integration for Lakekeeper or a generic + Iceberg REST catalog. A non-Glue ``catalog`` used to be ignored, so + ``catalog: lakekeeper`` published the table to Glue: a second, metadata- + less claim on the name, in a catalog the contract never named. The + emitter and ``validate_confluent_binding`` both read this, so the gate + refuses exactly what the emitter declines to publish. An absent + ``catalog`` is Glue here, as it always was. + """ + loc = binding.get("location") or {} + if not canonical_catalog_kind(loc.get("catalog")): + return "" + kind = binding_catalog_kind(binding) + return "" if kind == "glue" else kind + + def _resource_name(cid: str, loc: Mapping[str, Any], exposure: Mapping[str, Any]) -> str: """The OpenTofu resource name for an exposure's Tableflow resources. The emitter and ``validate_confluent_binding`` MUST agree so the validator's @@ -136,6 +159,8 @@ def _emit_tableflow( Skips silently when a hard input is absent — ``validate_confluent_binding`` surfaces a clean error at validate time, and plan-binding guarantees a validated contract by apply, so this only guards a partial/unvalidated dict. + A non-Glue ``location.catalog`` skips just the catalog integration, on the + same :func:`_non_glue_catalog` condition the validator refuses. """ environment_id = loc.get("environment_id") cluster_id = loc.get("kafka_cluster_id") @@ -173,7 +198,10 @@ def _emit_tableflow( } # 3. Catalog integration — publishes the Iceberg table to AWS Glue. The Glue - # database must pre-exist (Tableflow does not create it). + # database must pre-exist (Tableflow does not create it). Never for a + # contract that names another catalog: see _non_glue_catalog. + if _non_glue_catalog(exposure.get("binding") or {}): + return glue: Dict[str, Any] = {"provider_integration_id": pi_ref} if database: glue["custom_database"] = database @@ -215,7 +243,18 @@ def validate_confluent_binding(contract: Mapping[str, Any]) -> Tuple[List[str], ): if not loc.get(key): errors.append(f"expose '{eid}': platform=confluent requires {label}") - if not loc.get("database"): + foreign = _non_glue_catalog(binding) + if foreign: + errors.append( + f"expose '{eid}': binding.location.catalog is '{loc.get('catalog')}' " + f"({foreign}), but FLUID's Tableflow emitter publishes only to AWS " + "Glue. Tableflow's Polaris (Snowflake Open Catalog) and Unity Catalog " + "integrations authenticate with a client secret the credential-free " + "module cannot carry, and Tableflow has no integration for Lakekeeper " + "or a generic Iceberg REST catalog. Set catalog: glue or omit it." + ) + # The Glue-database warning would only restate the error above. + elif not loc.get("database"): warnings.append( f"expose '{eid}': no binding.location.database — the AWS Glue database must " f"pre-exist and be named as custom_database so Tableflow publishes there" diff --git a/fluid_build/iac/providers/snowflake.py b/fluid_build/iac/providers/snowflake.py index e71cab2c..8eb486e4 100644 --- a/fluid_build/iac/providers/snowflake.py +++ b/fluid_build/iac/providers/snowflake.py @@ -48,8 +48,10 @@ from ...observability.secret_redactor import collect_secret_values, redact_secret_text from ...providers._iceberg_catalog import ( - EXTERNAL_ICEBERG_CATALOGS, + FAMILY_GLUE, STORAGE_PROVIDERS, + binding_catalog_kind, + catalog_kind_info, iceberg_external_volume_is_override, iceberg_external_volume_name, ) @@ -737,24 +739,36 @@ def _emit_iceberg_prereqs( Emit-when-derivable, matching this module's pure-emitter shape (no logging, no I/O): - - Snowflake-managed catalog (no external ``location.catalog``): a - ``snowflake_external_volume``. Needs a ``location.warehouse`` with an - ``s3://`` or ``gs://`` scheme (or a ``bucket``, treated as S3); S3 also - needs ``location.iam_role_arn``. ``allow_writes`` is pinned ``"true"``, - which the provider requires for Iceberg tables using Snowflake as the - catalog. - - ``location.catalog: glue``: a ``snowflake_catalog_integration_aws_glue``. - Needs ``location.iam_role_arn`` (the role Snowflake assumes) and - ``location.account`` (the AWS account id). - - Other external catalogs (rest / polaris / unity): nothing yet. Their - integrations authenticate with OAuth client secrets or bearer tokens, - and this emitter's ``.tf.json`` is credential-free by invariant, so - wiring them needs a variables design first. Documented follow-up. + The branch is the ``snowflake_catalog_type`` column of the shared kind + table (:func:`fluid_build.providers._iceberg_catalog.catalog_kind_info`), + the same column dbt's ``catalogs.yml`` writes as ``catalog_type``, so the + two sides partition every spelling identically: - The external-vs-managed split uses the SAME - ``EXTERNAL_ICEBERG_CATALOGS`` set as the dbt emitter, so an unlisted - catalog value (``snowflake``, say) is Snowflake-managed to both sides - rather than dbt referencing a volume this side never creates. + - ``glue``: a ``snowflake_catalog_integration_aws_glue``. Needs + ``location.iam_role_arn`` (the role Snowflake assumes) and + ``location.account`` (the AWS account id). + - ``iceberg_rest`` (rest / iceberg_rest / lakekeeper / polaris / unity / + nessie / bigquery): nothing yet. Their integrations authenticate with + OAuth client secrets or bearer tokens, and this emitter's ``.tf.json`` + is credential-free by invariant, so wiring them needs a variables + design first. Documented follow-up; ``fluid validate`` warns. + - ``None`` (hive / jdbc / hadoop / dynamodb): nothing, ever. Snowflake + has no catalog integration for them, and ``fluid validate`` refuses + the contract. + - ``built_in`` (no ``location.catalog``, ``snowflake``, or a value the + table does not know): a ``snowflake_external_volume``. Needs a + ``location.warehouse`` with an ``s3://`` or ``gs://`` scheme (or a + ``bucket``, treated as S3); S3 also needs ``location.iam_role_arn``. + ``allow_writes`` is pinned ``"true"``, which the provider requires for + Iceberg tables using Snowflake as the catalog. An unknown value lands + here because dbt's fallback for it is ``built_in`` too, so dbt never + references a volume this side does not create; ``fluid validate`` + refuses the value itself. + + This used to lower the raw string and test it against a hand-kept set, + which missed ``lakekeeper``: a Lakekeeper table got an EXTERNAL VOLUME + and a Snowflake-managed dbt table while its streaming sink wrote to + Lakekeeper over REST. An explicit ``binding.icebergConfig.properties.external_volume`` means "I already have a volume": dbt references it and NO resource is emitted @@ -769,9 +783,9 @@ def _emit_iceberg_prereqs( if str(fmt or "").lower() not in _ICEBERG_FORMATS: return - catalog = str(loc.get("catalog") or "").lower() + kind = catalog_kind_info(binding_catalog_kind(binding)) - if catalog == "glue": + if kind.family == FAMILY_GLUE: role_arn = loc.get("iam_role_arn") account = loc.get("account") if not (role_arn and account): @@ -789,8 +803,9 @@ def _emit_iceberg_prereqs( ) return - if catalog in EXTERNAL_ICEBERG_CATALOGS: - # rest / polaris / unity / nessie: secret-bearing auth, see docstring. + if kind.snowflake_catalog_type != "built_in": + # ``iceberg_rest``: secret-bearing auth, see docstring. ``None``: + # Snowflake cannot integrate the catalog at all. return if iceberg_external_volume_is_override(binding): @@ -798,7 +813,7 @@ def _emit_iceberg_prereqs( return # Snowflake-managed (Horizon) catalog: the EXTERNAL VOLUME path. An - # unlisted ``catalog`` value lands here too, mirroring the dbt emitter's + # unknown ``catalog`` value lands here too, mirroring the dbt emitter's # built_in fallback. warehouse = str(loc.get("warehouse") or "") base_url = "" diff --git a/fluid_build/policy/compiler.py b/fluid_build/policy/compiler.py index c7cf4e37..1b6d64a0 100644 --- a/fluid_build/policy/compiler.py +++ b/fluid_build/policy/compiler.py @@ -58,7 +58,9 @@ def compile_policy(contract: dict) -> Tuple[List[Dict[str, Any]], List[str]]: Reads binding.platform from the contract schema to determine the provider and generate appropriate bindings. The provider and project are embedded in the output so downstream tools (policy-apply) don't - need separate flags. + need separate flags. An Iceberg expose is granted where its catalog + lives: Glue grants only for a Glue-cataloged table (see + :func:`_compile_catalog_managed_iceberg` for every other catalog). Returns: (bindings, warnings) where bindings is a list of IAM binding dicts @@ -87,6 +89,10 @@ def compile_policy(contract: dict) -> Tuple[List[Dict[str, Any]], List[str]]: if platform == "gcp" or fmt in ("bigquery_table", "gcs_parquet_files", "gcs_file"): _compile_gcp_bindings(bindings, fmt, loc, principal, permissions) + elif not _glue_cataloged(binding): + _compile_catalog_managed_iceberg( + bindings, warnings, exp, binding, principal, permissions + ) elif platform == "aws" or fmt in ("s3_file", "iceberg", "parquet"): _compile_aws_bindings(bindings, fmt, loc, principal, permissions) elif platform == "snowflake" or fmt == "snowflake_table": @@ -142,8 +148,80 @@ def _compile_gcp_bindings(bindings, fmt, loc, principal, permissions): ) -def _compile_aws_bindings(bindings, fmt, loc, principal, permissions): - """Generate AWS IAM policy statements.""" +def _glue_cataloged(binding) -> bool: + """Is this expose's table registered in AWS Glue? + + True for every non-Iceberg format and for an Iceberg table whose catalog + is Glue (named, or the AWS default). The format check used to be enough + on its own: every ``iceberg`` expose was routed to the AWS compiler, so a + Lakekeeper table, a Snowflake-managed table and a ``platform: local`` + table each got a ``glue.table`` grant on a Glue table that does not + exist. Reads the classification every emitter shares, imported lazily so + this module stays cheap to import. + """ + from ..providers._iceberg_catalog import is_glue_cataloged + + if not isinstance(binding.get("location") or {}, dict): + return True # malformed: keep the historic routing, schema reports it + return is_glue_cataloged(binding) + + +def _compile_catalog_managed_iceberg(bindings, warnings, exp, binding, principal, permissions): + """Compile an Iceberg expose whose catalog is not AWS Glue. + + The catalog owns the table, so the table-level grant belongs to the + catalog's own access control (Snowflake RBAC for a Snowflake-managed + table; Lakekeeper / Polaris / Unity authorization for a REST catalog). + Snowflake RBAC is compiled here. A grant this compiler cannot express is + reported as a warning, the same channel as an unsupported platform, never + dropped: a contract that reads as "principal X may read this table" must + not compile to silence. + + On AWS the bucket the binding names is still the table's storage, so its + S3 statement is emitted; the Glue statement is not. + """ + from ..iac.provider_match import canonical_cloud + from ..providers._iceberg_catalog import ( + FAMILY_SNOWFLAKE_MANAGED, + binding_catalog_kind, + catalog_kind_info, + ) + + kind = binding_catalog_kind(binding) + fmt = binding.get("format", "") + loc = binding.get("location") or {} + cloud = canonical_cloud(binding.get("platform")) + snowflake_managed = catalog_kind_info(kind).family == FAMILY_SNOWFLAKE_MANAGED + + if snowflake_managed or cloud == "snowflake": + _compile_snowflake_bindings(bindings, fmt, loc, principal, permissions) + elif cloud == "aws": + _compile_aws_bindings(bindings, fmt, loc, principal, permissions, glue=False) + + if snowflake_managed: + return + expose_id = exp.get("exposeId") or exp.get("id") or "?" + perms = list(permissions) + if cloud == "snowflake": + warnings.append( + f"Iceberg expose '{expose_id}' is cataloged in '{kind}': the Snowflake grant " + f"compiled for {principal} covers Snowflake readers only. Enforce {perms} for " + f"every other engine in the '{kind}' catalog's own access control." + ) + else: + warnings.append( + f"Iceberg expose '{expose_id}' is cataloged in '{kind}', not AWS Glue, so no " + f"table grant was compiled for {principal} {perms}. Enforce it in the " + f"'{kind}' catalog's own access control." + ) + + +def _compile_aws_bindings(bindings, fmt, loc, principal, permissions, *, glue=True): + """Generate AWS IAM policy statements. + + ``glue=False`` emits only the S3 statement, for an Iceberg table whose + catalog is not Glue (:func:`_compile_catalog_managed_iceberg`). + """ bucket = loc.get("bucket") if bucket: perm_key = ( @@ -166,7 +244,7 @@ def _compile_aws_bindings(bindings, fmt, loc, principal, permissions): # Glue/Athena bindings database = loc.get("database") or loc.get("dataset") table = loc.get("table") - if database: + if glue and database: perm_key = ( "manage" if any(p in permissions for p in ("write", "insert", "update", "delete")) diff --git a/fluid_build/providers/_iceberg_catalog.py b/fluid_build/providers/_iceberg_catalog.py index a8eb9687..e432d00e 100644 --- a/fluid_build/providers/_iceberg_catalog.py +++ b/fluid_build/providers/_iceberg_catalog.py @@ -28,6 +28,7 @@ import hashlib import re from dataclasses import dataclass, field +from types import MappingProxyType from typing import Any, Dict, Mapping, Optional, Tuple from ._sql_safety import validate_ident @@ -36,41 +37,281 @@ # Apache Iceberg runtime class names (pinned to the connector surface validated # in the OSS spike — RFC §14). Bumping the Iceberg runtime may change these. GLUE_CATALOG_IMPL = "org.apache.iceberg.aws.glue.GlueCatalog" +DYNAMODB_CATALOG_IMPL = "org.apache.iceberg.aws.dynamodb.DynamoDbCatalog" S3_FILE_IO = "org.apache.iceberg.aws.s3.S3FileIO" GCS_FILE_IO = "org.apache.iceberg.gcp.gcs.GCSFileIO" ADLS_FILE_IO = "org.apache.iceberg.azure.adlsv2.ADLSFileIO" -# Iceberg catalog types the runtime recognizes for ``iceberg.catalog.type``. -# Anything else (polaris / snowflake-managed / unity — all REST-fronted) maps to -# ``rest`` so the connector talks to it over the REST protocol. -_KNOWN_CATALOG_TYPES = frozenset( - {"rest", "hive", "hadoop", "jdbc", "nessie", "bigquery", "dynamodb"} +# --------------------------------------------------------------------------- +# Catalog kinds — THE one classification every emitter reads +# --------------------------------------------------------------------------- +# +# ``binding.location.catalog`` (and ``sink.catalog``) is a free string, and +# before this table four emitters each classified it by hand: the sink +# deriver mapped unknown kinds to ``rest``, dbt ``catalogs.yml`` and the +# Snowflake IaC mapped them to Snowflake-managed, the sink validator checked +# only the literal ``rest``, and the AWS IaC ignored the catalog and emitted a +# Glue table for every Iceberg expose. ``catalog: lakekeeper`` therefore +# streamed over REST while dbt wrote a Snowflake-managed table. One row per +# kind, and every consumer derives its answer from the row. +# +# The shape borrows dbt-adapters' ``_V2_TO_V1_TYPE`` hook (a user-facing kind +# mapped to each emitter's wire value, identity by default; dbt-snowflake +# impl.py) and Airbyte's typed Polaris preset (REST on the wire, but the +# warehouse is a catalog NAME). Wire values come only from Apache Iceberg's +# ``CatalogUtil`` set (hadoop/hive/rest/glue/nessie/jdbc/bigquery): no engine +# or SDK has a ``lakekeeper`` type, and CatalogUtil throws on one. + +#: AWS Glue Data Catalog through the native ``GlueCatalog``. +FAMILY_GLUE = "glue" +#: The Iceberg REST protocol: ``uri`` + ``warehouse`` (a catalog name). +FAMILY_REST = "rest" +#: A runtime-native, non-REST client (Nessie's own API, Hive metastore, JDBC...). +FAMILY_NATIVE = "native" +#: Snowflake's built-in (Horizon) catalog over a Snowflake EXTERNAL VOLUME. +FAMILY_SNOWFLAKE_MANAGED = "snowflake-managed" +#: A value this table does not know. ``fluid validate`` rejects it. +FAMILY_UNKNOWN = "unknown" + + +@dataclass(frozen=True) +class CatalogKind: + """What one ``location.catalog`` value means to every emitter.""" + + name: str + family: str + #: ``iceberg.catalog.type`` (Kafka Connect) / ``type`` (Debezium Server). + #: Mutually exclusive with ``catalog_impl``: Iceberg's ``CatalogUtil`` + #: throws when both are set. + runtime_type: Optional[str] + #: ``catalog-impl`` for a catalog the runtime has no ``type`` for. + catalog_impl: Optional[str] + #: dbt ``catalogs.yml`` on Snowflake: ``iceberg_rest`` (a catalog external + #: to Snowflake, reached through a catalog integration), ``built_in`` + #: (Snowflake-managed), or ``None`` when Snowflake has no integration for + #: it. The Snowflake IaC emitter partitions on the same column. + snowflake_catalog_type: Optional[str] + #: ``binding.location`` keys a streaming sink needs for this catalog. + sink_requires: Tuple[str, ...] = () + + @property + def speaks_rest(self) -> bool: + """Does the writer reach this catalog over the Iceberg REST protocol?""" + return self.runtime_type == "rest" + + +_REST_REQUIRES = ("uri", "warehouse") + +_CATALOG_KINDS: Mapping[str, CatalogKind] = MappingProxyType( + { + k.name: k + for k in ( + CatalogKind("glue", FAMILY_GLUE, None, GLUE_CATALOG_IMPL, "iceberg_rest"), + CatalogKind("rest", FAMILY_REST, "rest", None, "iceberg_rest", _REST_REQUIRES), + CatalogKind( + "lakekeeper", + FAMILY_REST, + "rest", + None, + "iceberg_rest", + _REST_REQUIRES, + ), + CatalogKind( + "polaris", + FAMILY_REST, + "rest", + None, + "iceberg_rest", + _REST_REQUIRES, + ), + CatalogKind( + "unity", + FAMILY_REST, + "rest", + None, + "iceberg_rest", + _REST_REQUIRES, + ), + # Native NessieCatalog for the sinks (uri ends /api/v1|v2, the + # warehouse is an object-store location), while Snowflake reaches + # Nessie through its Iceberg REST endpoint. The stock Apache Kafka + # Connect runtime does not bundle iceberg-nessie. + CatalogKind("nessie", FAMILY_NATIVE, "nessie", None, "iceberg_rest", _REST_REQUIRES), + # BigLake metastore. ``type=bigquery`` exists in Iceberg >= 1.10. + CatalogKind("bigquery", FAMILY_NATIVE, "bigquery", None, "iceberg_rest"), + CatalogKind("hive", FAMILY_NATIVE, "hive", None, None), + CatalogKind("jdbc", FAMILY_NATIVE, "jdbc", None, None, ("uri",)), + CatalogKind("hadoop", FAMILY_NATIVE, "hadoop", None, None, ("warehouse",)), + # CatalogUtil has no ``dynamodb`` type: it is reached by impl only. + CatalogKind("dynamodb", FAMILY_NATIVE, None, DYNAMODB_CATALOG_IMPL, None), + # Horizon over an EXTERNAL VOLUME; a streaming writer reaches it + # through Snowflake's Iceberg REST endpoint. + CatalogKind( + "snowflake-managed", + FAMILY_SNOWFLAKE_MANAGED, + "rest", + None, + "built_in", + _REST_REQUIRES, + ), + ) + } +) + +#: Accepted spellings for a canonical kind (after case and ``-``/``_`` folding). +_CATALOG_ALIASES: Mapping[str, str] = MappingProxyType( + {"iceberg-rest": "rest", "snowflake": "snowflake-managed"} ) -#: ``location.catalog`` values that mean "a catalog EXTERNAL to Snowflake". -#: THE shared predicate for the two halves of the dbt Iceberg loop: the dbt +#: A value outside the table. Emitters keep their historic fallbacks for it +#: (REST for a sink, Snowflake-managed for dbt) and ``fluid validate`` refuses it. +_UNKNOWN_KIND = CatalogKind("", FAMILY_UNKNOWN, "rest", None, "built_in") + + +def _fold(value: Any) -> str: + return str(value or "").strip().lower().replace("_", "-") + + +def canonical_catalog_kind(value: Any) -> str: + """The canonical kind for a ``catalog`` value: ``""`` when absent, the + table's name for a known kind or alias, else the folded value as given.""" + folded = _fold(value) + return _CATALOG_ALIASES.get(folded, folded) + + +def catalog_kind_info(value: Any) -> CatalogKind: + """The table row for ``value`` (any spelling), or the UNKNOWN row.""" + return _CATALOG_KINDS.get(canonical_catalog_kind(value), _UNKNOWN_KIND) + + +def known_catalog_kinds() -> Tuple[str, ...]: + """Every accepted spelling, canonical names first — for error messages.""" + return tuple(sorted(_CATALOG_KINDS)) + tuple(sorted(_CATALOG_ALIASES)) + + +def default_catalog_kind(binding: Mapping[str, Any]) -> str: + """The kind an Iceberg binding gets when it names none: Glue on AWS, + Snowflake-managed on Snowflake, a REST catalog anywhere else. + + GCP is the one platform where an absent catalog means two things: the + streaming sink has always written a REST catalog there, while dbt-bigquery + and the GCP IaC create a BigLake table. Changing the default would change + every existing GCP sink's ``iceberg.catalog.type`` (and ``bigquery`` needs + Iceberg >= 1.10, newer than the published Kafka Connect sink), so the + BigQuery paths read only an EXPLICIT ``location.catalog`` instead. + """ + from ..iac.provider_match import canonical_cloud + + cloud = canonical_cloud(binding.get("platform")) + if cloud == "aws": + return "glue" + if cloud == "snowflake": + return "snowflake-managed" + return "rest" + + +def binding_catalog_kind(binding: Mapping[str, Any]) -> str: + """The canonical kind of an expose binding: ``location.catalog``, else the + platform default. What dbt and the IaC emitters read.""" + loc = binding.get("location") or {} + return canonical_catalog_kind(loc.get("catalog")) or default_catalog_kind(binding) + + +def iceberg_catalog_kind(binding: Mapping[str, Any], sink: Any = None) -> str: + """The canonical kind a streaming sink writes through: ``sink.catalog``, + then ``location.catalog``, then the platform default. + + ``sink`` may be a ``SinkSpec`` or the raw ``sink`` mapping. ``fluid + validate`` refuses a ``sink.catalog`` that disagrees with the expose, + because dbt and the IaC emitters read only the expose. + """ + if isinstance(sink, Mapping): + explicit = sink.get("catalog") + else: + explicit = getattr(sink, "catalog", None) if sink is not None else None + return canonical_catalog_kind(explicit) or binding_catalog_kind(binding) + + +#: ``binding.format`` spellings that mean Iceberg (mirrors the sink validator). +_ICEBERG_FORMATS = frozenset({"iceberg", "iceberg-table"}) + + +def is_iceberg_format(fmt: Any) -> bool: + return _fold(fmt) in _ICEBERG_FORMATS + + +def is_glue_cataloged(binding: Mapping[str, Any]) -> bool: + """Is this binding's table registered in AWS Glue? + + Any non-Iceberg format Glue catalogs (parquet, csv, ...) is; an Iceberg + table is only when its catalog is Glue. A Lakekeeper or other REST-catalog + table lives in that catalog, so a static Glue table for it would be a + second, metadata-less claim on the same name. + """ + if not is_iceberg_format(binding.get("format")): + return True + return binding_catalog_kind(binding) == "glue" + + +#: ``location.catalog`` values that mean "a catalog EXTERNAL to Snowflake" +#: (every spelling, aliases included). Derived from the table: the dbt #: ``catalogs.yml`` emitter maps these to ``catalog_type: iceberg_rest`` and -#: everything else to ``built_in``; the Snowflake IaC emitter must partition -#: identically or one side references infrastructure the other never creates -#: (a ``catalog: snowflake`` binding, say, must be Snowflake-managed to BOTH). +#: the Snowflake IaC emitter skips the EXTERNAL VOLUME for them, so both read +#: :func:`catalog_kind_info` rather than hand-maintaining a list. EXTERNAL_ICEBERG_CATALOGS = frozenset( - {"glue", "polaris", "unity", "rest", "iceberg_rest", "nessie"} + {n for n, k in _CATALOG_KINDS.items() if k.snowflake_catalog_type == "iceberg_rest"} + | { + a + for a, n in _CATALOG_ALIASES.items() + if _CATALOG_KINDS[n].snowflake_catalog_type == "iceberg_rest" + } + | {"iceberg_rest"} ) +def iceberg_sink_exposes(contract: Mapping[str, Any]) -> Tuple[Mapping[str, Any], ...]: + """The exposes a self-managed streaming sink can write to, in order. + + An Iceberg-format expose that is not on ``platform: confluent`` (a + Confluent Tableflow expose is a MANAGED output its own plugin owns). The + runners and ``fluid validate`` both select through this, so the expose a + sink writes to is the one the validator checks. + """ + return tuple( + e + for e in (contract.get("exposes") or []) + if isinstance(e, Mapping) + and isinstance(e.get("binding") or {}, Mapping) + and is_iceberg_format((e.get("binding") or {}).get("format")) + and _fold((e.get("binding") or {}).get("platform")) != "confluent" + ) + + def find_iceberg_expose_binding(contract: Mapping[str, Any]) -> Optional[Dict[str, Any]]: """The expose ``binding`` carrying the Iceberg-table identity for a sink. Shared by the Kafka-Connect and Debezium-Server runners so both resolve the - SAME table identity (the RFC zero-drift spine). A simple ``format=iceberg`` - lookup; the validated build->expose join (build.outputs / exposeId) lands - with the plan-time validator (RFC §6.8 #5). + SAME table identity (the RFC zero-drift spine): the first of + :func:`iceberg_sink_exposes`. The validated build->expose join + (build.outputs / exposeId) lands with the plan-time validator (RFC §6.8 #5). """ - for exposure in contract.get("exposes") or []: - binding = exposure.get("binding") or {} - if str(binding.get("format") or "").lower() == "iceberg": - return binding - return None + exposes = iceberg_sink_exposes(contract) + return dict(exposes[0].get("binding") or {}) if exposes else None + + +#: Object-store URI schemes, and the Iceberg ``FileIO`` each one needs. THE one +#: scheme list: the resolver picks a FileIO from it and the sink validator +#: refuses a name-only warehouse that carries one. +_OBJECT_STORE_FILE_IO = ( + (("s3://", "s3a://", "s3n://"), S3_FILE_IO), + (("gs://", "gcs://"), GCS_FILE_IO), + (("abfs://", "abfss://"), ADLS_FILE_IO), +) + + +def is_object_store_uri(value: Any) -> bool: + """Does ``value`` carry an object-store scheme (s3://, gs://, abfss://...)?""" + return _io_impl_for_warehouse(str(value or "")) is not None def _io_impl_for_warehouse(warehouse: str) -> Optional[str]: @@ -80,14 +321,13 @@ def _io_impl_for_warehouse(warehouse: str) -> Optional[str]: works-in-REST-demo-fails-on-cloud trap). REST / Nessie / Hive catalogs can front any cloud, so the FileIO follows the WAREHOUSE scheme, not the catalog kind: ``s3://`` -> S3FileIO, ``gs://`` -> GCSFileIO, ``abfss://`` -> ADLSFileIO. + A warehouse NAME (Lakekeeper, Polaris) gets none: the catalog vends the + table's FileIO configuration with its metadata. """ w = (warehouse or "").lower() - if w.startswith(("s3://", "s3a://", "s3n://")): - return S3_FILE_IO - if w.startswith(("gs://", "gcs://")): - return GCS_FILE_IO - if w.startswith(("abfs://", "abfss://")): - return ADLS_FILE_IO + for schemes, file_io in _OBJECT_STORE_FILE_IO: + if w.startswith(schemes): + return file_io return None @@ -95,25 +335,17 @@ def _io_impl_for_warehouse(warehouse: str) -> Optional[str]: class ResolvedIcebergCatalog: """Canonical, provider-neutral Iceberg-table identity for a binding.""" - catalog_type: str # "glue" | "rest" | "nessie" | "hive" | ... + catalog_type: str # wire type: "glue" | "rest" | "nessie" | "hive" | ... warehouse: str # s3://|gs://|abfss:// path (glue/object-store) or catalog name (rest) fq_table: str # "." - catalog_impl: Optional[str] = None # GlueCatalog for glue + catalog_impl: Optional[str] = None # GlueCatalog for glue; XOR with catalog_type on the wire io_impl: Optional[str] = None # S3 / GCS / ADLS FileIO per warehouse scheme region: Optional[str] = None uri: Optional[str] = None # REST catalog endpoint id_columns: Tuple[str, ...] = () # -> iceberg.tables.default-id-columns partition_by: Tuple[str, ...] = () # -> iceberg.tables.default-partition-by extra_catalog_props: Mapping[str, str] = field(default_factory=dict) - - -def _catalog_kind(binding: Mapping[str, Any], sink: Any, loc: Mapping[str, Any]) -> str: - """glue / rest, in precedence: explicit sink.catalog > location.catalog > - platform default (aws -> glue, else rest).""" - explicit = (getattr(sink, "catalog", None) if sink is not None else None) or loc.get("catalog") - if explicit: - return str(explicit).lower() - return "glue" if str(binding.get("platform") or "").lower() == "aws" else "rest" + kind: str = "" # canonical catalog kind ("lakekeeper", "glue", ...) def _id_columns(contract: Optional[Mapping[str, Any]]) -> Tuple[str, ...]: @@ -144,12 +376,13 @@ def resolve_iceberg_catalog( database = loc.get("database") or binding.get("database") table = loc.get("table") or binding.get("table") fq_table = f"{database}.{table}" - kind = _catalog_kind(binding, sink, loc) + kind = iceberg_catalog_kind(binding, sink) + info = catalog_kind_info(kind) partition_by = tuple(getattr(sink, "partition_by", None) or loc.get("partitionBy") or ()) id_columns = _id_columns(contract) - if kind == "glue": + if info.family == FAMILY_GLUE: return ResolvedIcebergCatalog( catalog_type="glue", warehouse=get_iceberg_warehouse(loc, account_ref=account_ref), @@ -159,23 +392,26 @@ def resolve_iceberg_catalog( region=loc.get("region"), id_columns=id_columns, partition_by=partition_by, + kind=kind, ) - # Non-Glue catalog: REST / Nessie / Hive / Polaris / Snowflake Open Catalog / - # Unity (all REST-fronted) over any cloud storage. ``catalog_type`` is the - # kind when the runtime recognizes it (nessie / hive / rest / ...), else REST; - # the FileIO follows the WAREHOUSE scheme so GCS (gs://) and ADLS (abfss://) - # work, not just S3 (RFC §6.3 — PR7's REST + GCP profiles). + # Non-Glue catalog over any cloud storage. The wire type comes from the + # table (Lakekeeper / Polaris / Unity / Snowflake are all ``rest``; an + # unknown kind keeps the historic REST fallback, and ``fluid validate`` + # refuses it). The FileIO follows the WAREHOUSE scheme so GCS (gs://) and + # ADLS (abfss://) work, not just S3 (RFC §6.3 — PR7's REST + GCP profiles). warehouse = loc.get("warehouse") or "" return ResolvedIcebergCatalog( - catalog_type=kind if kind in _KNOWN_CATALOG_TYPES else "rest", + catalog_type=info.runtime_type or kind, warehouse=warehouse, fq_table=fq_table, + catalog_impl=info.catalog_impl, uri=loc.get("uri"), io_impl=_io_impl_for_warehouse(warehouse), region=loc.get("region"), id_columns=id_columns, partition_by=partition_by, + kind=kind, ) diff --git a/fluid_build/providers/aws/plan/planner.py b/fluid_build/providers/aws/plan/planner.py index decf1e7e..4d52d0ad 100644 --- a/fluid_build/providers/aws/plan/planner.py +++ b/fluid_build/providers/aws/plan/planner.py @@ -50,6 +50,36 @@ def _replacer(m: re.Match) -> str: return re.sub(r"\{\{\s*env\.(\S+?)\s*\}\}", _replacer, value) +def _glue_cataloged(binding: Mapping[str, Any]) -> bool: + """Does this AWS binding's table live in the Glue Data Catalog? + + False only for an Iceberg binding whose ``location.catalog`` names a + non-Glue catalog (``lakekeeper``, ``rest``, ``nessie``...). That table is + created and owned by its catalog, so a Glue database and Glue Iceberg table + for it would be a second, metadata-less claim on the same name: the bug + class where ``catalog: lakekeeper`` streamed over REST while the AWS plan + still provisioned a Glue table. An absent catalog on AWS is Glue, so every + existing contract plans exactly as before. + + Reads the one classification every emitter shares + (``providers/_iceberg_catalog.is_glue_cataloged``), imported lazily as + ``_plan_exposures`` does its format helpers. A non-mapping ``location`` (a + malformed contract this planner already tolerates) counts as no catalog. + """ + from fluid_build.providers._iceberg_catalog import is_glue_cataloged + + location = binding.get("location") + if not isinstance(location, Mapping): + location = {} + return is_glue_cataloged({**binding, "location": location}) + + +def _declared_catalog(binding: Mapping[str, Any]) -> Any: + """``location.catalog`` as written, for log events (``None`` when absent).""" + location = binding.get("location") + return location.get("catalog") if isinstance(location, Mapping) else None + + class AwsPlanner(BasePlanner): """AWS-specific planner — wires the 6-phase scaffold to the Glue / S3 / Athena / Redshift phase functions below. @@ -224,7 +254,29 @@ def _plan_infrastructure( platform = binding.get("platform", "").lower() # Handle different binding formats - if platform in ["aws", "glue", "athena"]: + if platform in ["aws", "glue", "athena"] and not _glue_cataloged(binding): + # A non-Glue Iceberg catalog owns its namespace, so no Glue + # database (and no ``{account}-fluid-data`` fallback bucket, which + # exists only to back that database). A bucket the binding names + # is still the table's storage, and the IaC emitter creates it too. + location = binding.get("location") or {} + raw_bucket = ( + location.get("bucket") if isinstance(location, dict) else None + ) or binding.get("bucket") + bucket_name = _resolve_env_templates(raw_bucket) if raw_bucket else None + if bucket_name and "{{" not in bucket_name and bucket_name not in buckets_created: + actions.append( + { + "op": "s3.ensure_bucket", + "id": f"bucket_{bucket_name}", + "bucket": bucket_name, + "region": region, + "tags": _get_resource_tags(contract), + } + ) + buckets_created.add(bucket_name) + + elif platform in ["aws", "glue", "athena"]: # Ensure Glue database exists (support nested and flat formats) location = binding.get("location") or {} database = location.get("database") if isinstance(location, dict) else None @@ -347,7 +399,22 @@ def _plan_iam_policies( continue platform = binding.get("platform", "").lower() - if platform in ["aws", "glue", "athena"]: + if platform in ["aws", "glue", "athena"] and not _glue_cataloged(binding): + # No Glue database exists for a non-Glue Iceberg table, so a Glue + # IAM binding would grant on nothing while the catalog that owns + # the table stays ungoverned. Say so rather than skip silently; + # the catalog's own authorization is where these policies apply. + logger.warning( + format_event( + "glue_iam_binding_skipped", + expose=exposure.get("exposeId") or exposure.get("id"), + catalog=_declared_catalog(binding), + reason="the table lives in a non-Glue Iceberg catalog; " + "enforce metadata.policies in that catalog", + ) + ) + + elif platform in ["aws", "glue", "athena"]: database = binding.get("database") if database: @@ -496,7 +563,18 @@ def _plan_exposures( platform = binding.get("platform", "").lower() contract_schema = exposure.get("contract", {}).get("schema", []) - if platform in ["aws", "glue", "athena"]: + if platform in ["aws", "glue", "athena"] and not _glue_cataloged(binding): + # The table lives in its own Iceberg catalog (Lakekeeper, a REST + # catalog...), whose writer creates it: no Glue Iceberg table here. + logger.debug( + format_event( + "glue_table_skipped", + expose=exposure_id, + catalog=_declared_catalog(binding), + ) + ) + + elif platform in ["aws", "glue", "athena"]: location = binding.get("location", {}) database = location.get("database") or binding.get("database") table = location.get("table") or binding.get("table") diff --git a/fluid_build/schemas/fluid-schema-0.7.6.json b/fluid_build/schemas/fluid-schema-0.7.6.json index 44c75802..dfba70d0 100644 --- a/fluid_build/schemas/fluid-schema-0.7.6.json +++ b/fluid_build/schemas/fluid-schema-0.7.6.json @@ -1943,7 +1943,7 @@ }, "catalog": { "type": "string", - "description": "NEW in v0.7.5: Iceberg catalog kind/profile selector — maps to iceberg.catalog.type (e.g. glue, rest, nessie, hive). REST-fronted catalogs (polaris / unity / snowflake open catalog) map to 'rest'. Consumed by providers/_iceberg_catalog.py and the streaming-sink validator." + "description": "NEW in v0.7.5: The Iceberg catalog that owns this table: glue, rest, lakekeeper, polaris, unity, nessie, bigquery, hive, jdbc, hadoop, dynamodb or snowflake-managed. Spellings fold case and '-'/'_', so iceberg_rest == rest and snowflake == snowflake-managed; fluid validate refuses any other value on an Iceberg expose. When absent: glue on platform aws, snowflake-managed on snowflake, rest elsewhere. The REST catalogs (rest, lakekeeper, polaris, unity) need location.uri and location.warehouse. On platform aws, a kind other than glue gets no Glue table and no Glue grant: that catalog creates and governs the table. Every emitter (streaming sinks, dbt catalogs.yml, the IaC, policy compile) reads one table, providers/_iceberg_catalog.py." }, "warehouse": { "type": "string", diff --git a/tests/build_runners/test_debezium_iceberg_sink.py b/tests/build_runners/test_debezium_iceberg_sink.py index c1b5173b..e0fa76c3 100644 --- a/tests/build_runners/test_debezium_iceberg_sink.py +++ b/tests/build_runners/test_debezium_iceberg_sink.py @@ -324,3 +324,99 @@ def test_properties_key_escapes_separators(): assert _escape_properties_key("debezium.sink.iceberg.warehouse") == ( "debezium.sink.iceberg.warehouse" ) + + +# ── runner preflight: fail closed BEFORE application.properties is written ── +# +# The embedded runner runs the same checks as `fluid validate` right before it +# derives the sink, so a sink the validator rejects never boots a server. + +_LAKEKEEPER_NO_URI = { + "platform": "local", + "format": "iceberg", + "location": { + "database": "bronze", + "table": "orders", + "catalog": "lakekeeper", + "warehouse": "analytics", + }, +} + + +def _props_path(contract: Dict[str, Any], tmp_path: Path) -> Path: + return tmp_path / ".fluid" / "debezium" / contract["id"] / "ingest" / "application.properties" + + +def _run_embedded(contract: Dict[str, Any], tmp_path: Path): + from fluid_build.build_runners._acquisition_common import build_acquisition_run_context + from fluid_build.build_runners.debezium.runner import DebeziumRunner + + ctx = build_acquisition_run_context(contract["builds"][0], contract, tmp_path) + return DebeziumRunner().run(ctx) + + +@pytest.mark.parametrize( + "sink, binding, expected", + [ + ( + {"type": "iceberg"}, + _LAKEKEEPER_NO_URI, + "lakekeeper catalog requires binding.location.uri", + ), + ( + # a hand-written `type` merged over the derived Glue catalog-impl + {"type": "iceberg", "iceberg_sink_enabled": True, "config": {"type": "rest"}}, + None, + "would carry both type", + ), + ], + ids=["missing-uri", "type-and-impl"], +) +def test_embedded_preflight_fails_before_writing_config(tmp_path: Path, sink, binding, expected): + from fluid_build.api.runner import RunState + + contract = _contract(sink=sink, binding=binding) + result = _run_embedded(contract, tmp_path) + assert result.state == RunState.FAILED + assert expected in (result.error or ""), result.error + assert "iceberg sink preflight failed" in result.error + assert not _props_path(contract, tmp_path).exists() + + assert execute_debezium_build(contract["builds"][0], contract, tmp_path, dry_run=False) == 1 + assert not _props_path(contract, tmp_path).exists() + + +def test_embedded_complete_lakekeeper_derives_rest(tmp_path: Path): + binding = { + **_LAKEKEEPER_NO_URI, + "location": {**_LAKEKEEPER_NO_URI["location"], "uri": "http://lakekeeper:8181/catalog"}, + } + text = _props_text(_contract(sink={"type": "iceberg"}, binding=binding), tmp_path) + assert "debezium.sink.iceberg.type=rest" in text + assert "debezium.sink.iceberg.uri=http://lakekeeper:8181/catalog" in text + assert "debezium.sink.iceberg.warehouse=analytics" in text + assert "debezium.sink.iceberg.catalog-impl=" not in text + + +def test_embedded_handwritten_config_skips_the_preflight(tmp_path: Path): + # Derivation is OFF for a hand-written config, so nothing is derived from + # the binding and the file stays what the operator wrote. + contract = _contract( + sink={"type": "iceberg", "config": {"catalog.name": "rest"}}, binding=_LAKEKEEPER_NO_URI + ) + text = _props_text(contract, tmp_path) + assert "debezium.sink.iceberg.catalog.name=rest" in text + + +def test_bring_your_own_mode_is_not_preflighted(kafka_connect_mock, tmp_path: Path): + # bring-your-own creates only the SOURCE connector; no sink is derived, so an + # incomplete Iceberg binding is not this build's concern. + contract = _contract(sink={"type": "iceberg"}, binding=_LAKEKEEPER_NO_URI) + contract["builds"][0]["properties"]["debezium"] = { + "deployment": {"mode": "bring-your-own", "server_url": "http://kafka-connect.test:8083"}, + "status_timeout_seconds": 1, + "poll_interval_seconds": 0.01, + } + rc = execute_debezium_build(contract["builds"][0], contract, tmp_path, dry_run=False) + assert rc == 0 + assert "create" in kafka_connect_mock.calls diff --git a/tests/build_runners/test_iceberg_sink_deriver.py b/tests/build_runners/test_iceberg_sink_deriver.py index 8d3a1499..2cfa1946 100644 --- a/tests/build_runners/test_iceberg_sink_deriver.py +++ b/tests/build_runners/test_iceberg_sink_deriver.py @@ -108,8 +108,10 @@ def test_deriver_core_keys_and_class(): assert "io.tabular" not in cfg["connector.class"] assert cfg["iceberg.tables"] == "sales.orders" assert cfg["topics"] == "orders" - assert cfg["iceberg.catalog.type"] == "glue" + # catalog-impl XOR type: Iceberg's CatalogUtil refuses a config carrying + # both, so a Glue sink that sent both never started. assert cfg["iceberg.catalog.catalog-impl"] == "org.apache.iceberg.aws.glue.GlueCatalog" + assert "iceberg.catalog.type" not in cfg assert cfg["iceberg.catalog.warehouse"] == "s3://lake/sales/orders/" assert cfg["iceberg.catalog.io-impl"] == "org.apache.iceberg.aws.s3.S3FileIO" assert cfg["iceberg.catalog.client.region"] == "us-east-1" @@ -280,3 +282,126 @@ def test_runner_default_off_when_handwritten_sink(kafka_connect_mock, tmp_path): cfg = kafka_connect_mock.connectors["snk"]["config"] assert cfg["connector.class"] == "io.confluent.connect.s3.S3SinkConnector" assert "iceberg.tables" not in cfg + + +# ── runner preflight: fail closed BEFORE any Connect REST call ────────────── +# +# The runner runs the same checks as `fluid validate` right before deriving, +# so a sink the validator rejects never reaches the cluster: before this a +# Lakekeeper binding with no uri was POSTed as a connector that failed at its +# first record, and a Glue sink with an overridden `type` never started. + +_LAKEKEEPER_NO_URI = { + "platform": "local", + "format": "iceberg", + "location": { + "database": "sales", + "table": "orders", + "catalog": "lakekeeper", + "warehouse": "analytics", + }, +} + + +def _run(contract, tmp_path): + from fluid_build.build_runners._acquisition_common import build_acquisition_run_context + from fluid_build.build_runners.kafka_connect.runner import KafkaConnectRunner + + ctx = build_acquisition_run_context(contract["builds"][0], contract, tmp_path) + return KafkaConnectRunner().run(ctx) + + +@pytest.mark.parametrize( + "mutate, expected", + [ + ( + lambda c: c["exposes"][0].__setitem__("binding", _LAKEKEEPER_NO_URI), + "lakekeeper catalog requires binding.location.uri", + ), + ( + lambda c: c["builds"][0]["properties"]["kafka-connect"].__setitem__( + "iceberg_catalog_overrides", {"iceberg.catalog.type": "rest"} + ), + "would carry both iceberg.catalog.type", + ), + ( + lambda c: c["builds"][0]["properties"]["sink"].__setitem__("catalog", "lakekeeper"), + "disagrees with the expose's catalog 'glue'", + ), + (lambda c: c.__setitem__("exposes", []), "has no expose with binding.format=iceberg"), + ], + ids=["missing-uri", "type-and-impl", "sink-vs-expose", "no-expose"], +) +def test_runner_preflight_fails_before_any_rest_call( + kafka_connect_mock, tmp_path, mutate, expected +): + from fluid_build.api.runner import RunState + from fluid_build.build_runners.kafka_connect.runner import execute_kafka_connect_build + + contract = _iceberg_contract() + mutate(contract) + + result = _run(contract, tmp_path) + assert result.state == RunState.FAILED + assert expected in (result.error or ""), result.error + assert "iceberg sink preflight failed" in result.error + # Nothing reached the cluster: not the sink, and not the SOURCE connector + # either (it would stream into a topic no sink drains). + assert kafka_connect_mock.calls == [] + assert kafka_connect_mock.connectors == {} + + assert execute_kafka_connect_build(contract["builds"][0], contract, tmp_path) == 1 + assert kafka_connect_mock.calls == [] + + +def test_runner_preflight_warning_logs_and_still_deploys(kafka_connect_mock, tmp_path, caplog): + from fluid_build.build_runners.kafka_connect.runner import execute_kafka_connect_build + + contract = _iceberg_contract() + contract["exposes"][0]["binding"] = { + **_LAKEKEEPER_NO_URI, + "location": { + **_LAKEKEEPER_NO_URI["location"], + "catalog": "nessie", + "uri": "http://nessie:19120/api/v2", + "warehouse": "s3://lake/warehouse", + }, + } + with caplog.at_level("WARNING", logger="fluid.acquire.kafka_connect"): + rc = execute_kafka_connect_build(contract["builds"][0], contract, tmp_path) + assert rc == 0 + assert any("iceberg-nessie" in r.getMessage() for r in caplog.records) + cfg = kafka_connect_mock.connectors["src-sink"]["config"] + assert cfg["iceberg.catalog.type"] == "nessie" + assert "iceberg.catalog.catalog-impl" not in cfg + assert cfg["iceberg.catalog.uri"] == "http://nessie:19120/api/v2" + + +def test_complete_lakekeeper_streams_over_rest(kafka_connect_mock, tmp_path): + from fluid_build.build_runners.kafka_connect.runner import execute_kafka_connect_build + + contract = _iceberg_contract() + contract["exposes"][0]["binding"] = { + **_LAKEKEEPER_NO_URI, + "location": {**_LAKEKEEPER_NO_URI["location"], "uri": "http://lakekeeper:8181/catalog"}, + } + assert execute_kafka_connect_build(contract["builds"][0], contract, tmp_path) == 0 + cfg = kafka_connect_mock.connectors["src-sink"]["config"] + assert cfg["iceberg.catalog.type"] == "rest" + assert cfg["iceberg.catalog.uri"] == "http://lakekeeper:8181/catalog" + # A warehouse NAME: the catalog vends FileIO, so none is forced. + assert "iceberg.catalog.io-impl" not in cfg + + +def test_runner_handwritten_sink_skips_the_preflight(kafka_connect_mock, tmp_path): + from fluid_build.build_runners.kafka_connect.runner import execute_kafka_connect_build + + # Derivation is OFF for a hand-written sink, so nothing is derived from the + # binding and the run stays byte-for-byte what it was (validate still flags it). + contract = _iceberg_contract(sink_connector_config={"connector.class": "x.Y", "topics": "t"}) + contract["exposes"][0]["binding"] = _LAKEKEEPER_NO_URI + assert execute_kafka_connect_build(contract["builds"][0], contract, tmp_path) == 0 + assert kafka_connect_mock.connectors["snk"]["config"] == { + "connector.class": "x.Y", + "topics": "t", + } diff --git a/tests/build_runners/test_iceberg_sink_validation.py b/tests/build_runners/test_iceberg_sink_validation.py index 8dbae208..c1fd6ab1 100644 --- a/tests/build_runners/test_iceberg_sink_validation.py +++ b/tests/build_runners/test_iceberg_sink_validation.py @@ -19,8 +19,15 @@ import pytest from fluid_build.build_runners.kafka_connect.iceberg_sink_validation import ( + iceberg_sink_preflight, validate_iceberg_sink, ) +from fluid_build.providers._iceberg_catalog import ( + FAMILY_REST, + canonical_catalog_kind, + catalog_kind_info, + known_catalog_kinds, +) pytestmark = [pytest.mark.unit] @@ -191,3 +198,393 @@ def test_confluent_expose_not_treated_as_kc_sink_target(): # correctly reports "no iceberg expose" rather than a bogus rest-catalog error assert any("no expose with binding.format=iceberg" in e for e in errs) assert not any("rest catalog requires" in e for e in errs) + + +# ── table-driven catalog checks (one classification for every emitter) ────── +# +# Before the shared kind table this validator checked only the literal "rest", +# so `catalog: lakekeeper` skipped every check, streamed over REST, and dbt +# wrote a Snowflake-managed table for the same expose. Every case below derives +# its expectation from the table row, so a new kind is covered by construction. + +CANONICAL_KINDS = sorted({canonical_catalog_kind(k) for k in known_catalog_kinds()}) +REQUIRING_KINDS = [k for k in CANONICAL_KINDS if catalog_kind_info(k).sink_requires] +COMPLETE = {"uri": "http://lakekeeper:8181/catalog", "warehouse": "analytics"} + + +def _loc_binding(catalog=None, *, platform="local", **location): + loc = {"database": "default", "table": "events", **location} + if catalog is not None: + loc["catalog"] = catalog + return {"platform": platform, "format": "iceberg", "location": loc} + + +def _sink_contract(binding, *, sink=None, kc=None, engine="kafka-connect", debezium=None): + """A one-build contract whose sink block, engine and runtime props vary.""" + props = { + "source": {"kind": "postgres", "mode": "incremental_append"}, + "sink": {"format": "iceberg", **(sink or {})}, + } + if engine == "debezium": + props["debezium"] = debezium if debezium is not None else {} + else: + props["kafka-connect"] = kc or {} + return { + "id": "b.x", + "builds": [ + {"id": "ingest", "pattern": "acquisition", "engine": engine, "properties": props} + ], + "exposes": [{"exposeId": "events", "kind": "table", "binding": binding}], + } + + +def _embedded(sink_block=None): + return {"deployment": {"mode": "embedded"}, "server": {"sink": sink_block or {}}} + + +@pytest.mark.parametrize("kind", REQUIRING_KINDS) +@pytest.mark.parametrize("missing", ["uri", "warehouse"]) +def test_every_requiring_kind_demands_its_location_keys(kind, missing): + row = catalog_kind_info(kind) + location = {k: v for k, v in COMPLETE.items() if k != missing} + errs = _errs(_sink_contract(_loc_binding(kind, **location))) + demanded = [e for e in errs if f"{kind} catalog requires binding.location.{missing}" in e] + assert bool(demanded) == (missing in row.sink_requires), errs + + +@pytest.mark.parametrize("kind", [k for k in CANONICAL_KINDS if k not in REQUIRING_KINDS]) +def test_kinds_requiring_nothing_demand_nothing(kind): + errs = _errs(_sink_contract(_loc_binding(kind, platform="aws", bucket="lake"))) + assert not any("catalog requires" in e for e in errs), errs + + +@pytest.mark.parametrize("kind", ["rest", "lakekeeper", "polaris", "unity"]) +def test_rest_family_warehouse_message_names_the_catalog_name(kind): + assert catalog_kind_info(kind).family == FAMILY_REST + errs = _errs(_sink_contract(_loc_binding(kind, uri="http://c:8181/catalog"))) + assert any( + f"{kind} catalog requires binding.location.warehouse (the catalog name)" in e for e in errs + ), errs + + +def test_complete_lakekeeper_is_clean(): + assert validate_iceberg_sink(_sink_contract(_loc_binding("lakekeeper", **COMPLETE))) == ( + [], + [], + ) + + +def test_lakekeeper_used_to_skip_every_check(): + # The bug: no uri and no warehouse, and the old literal-"rest" check passed it. + errs = _errs(_sink_contract(_loc_binding("lakekeeper"))) + assert any("lakekeeper catalog requires binding.location.uri" in e for e in errs) + assert any("lakekeeper catalog requires binding.location.warehouse" in e for e in errs) + + +# ── sink.catalog must agree with the expose (dbt + IaC read only the expose) ── + + +@pytest.mark.parametrize( + "sink_catalog, binding", + [ + # AWS expose with no catalog is Glue (the platform default). + ("lakekeeper", _loc_binding(None, platform="aws", bucket="lake", region="us-east-1")), + ("rest", _loc_binding("lakekeeper", **COMPLETE)), + ("nessie", _loc_binding("rest", uri="http://c:8181", warehouse="s3://b/w")), + ], +) +def test_sink_catalog_disagreeing_with_expose_errors(sink_catalog, binding): + errs = _errs(_sink_contract(binding, sink={"catalog": sink_catalog})) + hit = [e for e in errs if "disagrees with the expose's catalog" in e] + assert hit, errs + assert f"binding.location.catalog: {canonical_catalog_kind(sink_catalog)}" in hit[0] + + +def test_disagreement_names_the_platform_default_when_expose_has_no_catalog(): + binding = _loc_binding(None, platform="aws", bucket="lake", region="us-east-1") + errs = _errs(_sink_contract(binding, sink={"catalog": "lakekeeper"})) + assert any("'glue'" in e and "platform default" in e for e in errs), errs + + +@pytest.mark.parametrize( + "sink_catalog, location_catalog", + [ + ("iceberg-rest", "rest"), + ("ICEBERG_REST", "rest"), + ("rest", "iceberg_rest"), + ("snowflake", "snowflake_managed"), + ("Lakekeeper", "lakekeeper"), + ], +) +def test_aliases_of_the_same_catalog_agree(sink_catalog, location_catalog): + binding = _loc_binding(location_catalog, **COMPLETE) + errs = _errs(_sink_contract(binding, sink={"catalog": sink_catalog})) + assert not any("disagrees" in e for e in errs), errs + + +# ── unknown kinds are refused, never guessed ──────────────────────────────── + + +@pytest.mark.parametrize("where", ["location", "sink"]) +def test_unknown_kind_errors_naming_value_and_listing_kinds(where): + if where == "location": + contract = _sink_contract(_loc_binding("lakekeper", **COMPLETE)) + else: + contract = _sink_contract( + _loc_binding("lakekeeper", **COMPLETE), sink={"catalog": "gravitino"} + ) + errs = _errs(contract) + unknown = [e for e in errs if "unknown Iceberg catalog" in e] + assert len(unknown) == 1, errs + value, field = ( + ("'lakekeper'", "binding.location.catalog") + if where == "location" + else ( + "'gravitino'", + "sink.catalog", + ) + ) + assert value in unknown[0] and field in unknown[0] + for kind in known_catalog_kinds(): + assert kind in unknown[0] + # No per-kind guesses about a row that does not exist, and a typo is never + # offered back as the remedy. + assert not any("catalog requires" in e for e in errs) + assert not any("disagrees" in e for e in errs) + + +def test_generic_rest_accepts_an_object_store_warehouse(): + # Plain REST catalogs (the apache/iceberg-rest-fixture) take an s3:// warehouse. + binding = _loc_binding("rest", uri="http://iceberg:8181", warehouse="s3://bucket/warehouse/") + assert validate_iceberg_sink(_sink_contract(binding)) == ([], []) + + +# ── runtime support: the stock KC runtime has no Nessie client ────────────── + + +def test_nessie_on_kafka_connect_warns_about_the_missing_runtime_jar(): + binding = _loc_binding("nessie", uri="http://nessie:19120/api/v2", warehouse="s3://b/w") + errors, warnings = validate_iceberg_sink(_sink_contract(binding)) + assert errors == [] + assert any("does not bundle iceberg-nessie" in w for w in warnings), warnings + + +def test_nessie_on_debezium_server_has_no_kafka_connect_warning(): + binding = _loc_binding("nessie", uri="http://nessie:19120/api/v2", warehouse="s3://b/w") + contract = _sink_contract(binding, engine="debezium", debezium=_embedded()) + assert not any("iceberg-nessie" in w for w in _warns(contract)) + + +# ── check 5: the warehouse override compares against the RESOLVED warehouse ─ + + +def test_rest_override_equal_to_binding_warehouse_is_clean(): + # Regression: the override was compared against the Glue writer's + # s3:////
/ even for REST, so a MATCHING override "diverged". + binding = _loc_binding("lakekeeper", **COMPLETE) + kc = {"iceberg_catalog_overrides": {"iceberg.catalog.warehouse": "analytics"}} + assert validate_iceberg_sink(_sink_contract(binding, kc=kc)) == ([], []) + + +def test_rest_override_divergence_names_kind_and_not_glue(): + binding = _loc_binding("lakekeeper", **COMPLETE) + kc = {"iceberg_catalog_overrides": {"iceberg.catalog.warehouse": "other"}} + warns = _warns(_sink_contract(binding, kc=kc)) + hit = [w for w in warns if "diverges from the binding warehouse" in w] + assert len(hit) == 1, warns + assert "lakekeeper catalog" in hit[0] + assert "Glue" not in hit[0] + + +def test_glue_override_divergence_message_is_unchanged(): + # Byte-identical for Glue on AWS: the text operators and tooling already see. + warns = _warns( + _contract(kc={"iceberg_catalog_overrides": {"iceberg.catalog.warehouse": "s3://other/"}}) + ) + assert warns == [ + "iceberg sink (build 'ingest'): iceberg_catalog_overrides warehouse 's3://other/' " + "diverges from the binding warehouse 's3://lake/s/o/'; the connector will use the " + "override but the static Glue table may differ" + ] + + +def test_glue_region_warning_is_unchanged(): + binding = { + "platform": "aws", + "format": "iceberg", + "location": {"database": "s", "table": "o", "bucket": "lake"}, + } + assert _warns(_contract(binding=binding)) == [ + "iceberg sink (build 'ingest'): glue catalog without binding.location.region; " + "the connector needs iceberg.catalog.client.region" + ] + + +def test_missing_required_warehouse_is_not_also_reported_as_divergence(): + binding = _loc_binding("lakekeeper", uri="http://c:8181/catalog") + kc = {"iceberg_catalog_overrides": {"iceberg.catalog.warehouse": "analytics"}} + errors, warnings = validate_iceberg_sink(_sink_contract(binding, kc=kc)) + assert any("requires binding.location.warehouse" in e for e in errors) + assert not any("diverges" in w for w in warnings) + + +# ── check 6: type XOR catalog-impl (CatalogUtil refuses both) ─────────────── + +_GLUE_AWS = _loc_binding(None, platform="aws", bucket="lake", region="us-east-1") +_REST = _loc_binding("rest", uri="http://iceberg:8181", warehouse="s3://b/w") + + +@pytest.mark.parametrize( + "binding, kc, conflict", + [ + # derived Glue catalog-impl + an override type + (_GLUE_AWS, {"iceberg_catalog_overrides": {"iceberg.catalog.type": "rest"}}, True), + # derived REST type + an override catalog-impl + (_REST, {"iceberg_catalog_overrides": {"iceberg.catalog.catalog-impl": "x.Y"}}, True), + # presence is what CatalogUtil checks: an empty value still trips it + (_GLUE_AWS, {"iceberg_catalog_overrides": {"iceberg.catalog.type": ""}}, True), + # overriding the SAME key the deriver emits is fine + (_GLUE_AWS, {"iceberg_catalog_overrides": {"iceberg.catalog.catalog-impl": "x.Y"}}, False), + (_REST, {"iceberg_catalog_overrides": {"iceberg.catalog.type": "rest"}}, False), + # a hand-written sink config merged over a derived one (opt-in) + ( + _GLUE_AWS, + { + "iceberg_sink_enabled": True, + "sink_connector_config": {"iceberg.catalog.type": "glue"}, + }, + True, + ), + # derivation OFF: catalog overrides are never applied, only the hand-written map + ( + _GLUE_AWS, + { + "sink_connector_config": {"topics": "t"}, + "iceberg_catalog_overrides": {"iceberg.catalog.type": "rest"}, + }, + False, + ), + ( + _GLUE_AWS, + { + "sink_connector_config": { + "iceberg.catalog.type": "glue", + "iceberg.catalog.catalog-impl": "x.Y", + } + }, + True, + ), + ], +) +def test_overrides_must_not_set_both_catalog_selectors(binding, kc, conflict): + errs = _errs(_sink_contract(binding, kc=kc)) + hit = [e for e in errs if "would carry both" in e] + assert bool(hit) == conflict, errs + if conflict: + assert "iceberg.catalog.type" in hit[0] and "iceberg.catalog.catalog-impl" in hit[0] + assert "CatalogUtil" in hit[0] + + +def test_debezium_server_selector_conflict_uses_bare_keys(): + contract = _sink_contract( + _GLUE_AWS, + engine="debezium", + debezium=_embedded({"iceberg_sink_enabled": True, "config": {"type": "rest"}}), + ) + hit = [e for e in _errs(contract) if "would carry both" in e] + assert len(hit) == 1 + assert "type (from debezium.server.sink.config)" in hit[0] + assert "catalog-impl (from the derived glue catalog config)" in hit[0] + + +# ── Debezium: only a build that derives a sink is checked ─────────────────── + +_BROKEN = _loc_binding("lakekeeper") # no uri, no warehouse + + +@pytest.mark.parametrize("mode", ["bring-your-own", "managed", None]) +def test_debezium_source_only_modes_draw_no_sink_findings(mode): + debezium = {"deployment": {"mode": mode}} if mode else {} + contract = _sink_contract(_BROKEN, engine="debezium", debezium=debezium) + assert validate_iceberg_sink(contract) == ([], []) + contract["exposes"] = [] # not even the missing-expose join applies + assert validate_iceberg_sink(contract) == ([], []) + + +def test_debezium_embedded_iceberg_sink_is_checked(): + contract = _sink_contract(_BROKEN, engine="debezium", debezium=_embedded()) + assert any("lakekeeper catalog requires binding.location.uri" in e for e in _errs(contract)) + + +def test_debezium_embedded_deriving_build_is_checked_without_sink_format(): + # The Debezium runner derives from server.sink.type and never reads + # sink.format, so selecting by sink.format alone let this build skip. + contract = _sink_contract(_BROKEN, engine="debezium", debezium=_embedded()) + contract["builds"][0]["properties"].pop("sink") + assert any("lakekeeper catalog requires binding.location.uri" in e for e in _errs(contract)) + + +def test_debezium_embedded_non_iceberg_server_sink_is_not_checked(): + contract = _sink_contract( + _BROKEN, engine="debezium", debezium=_embedded({"type": "s3", "config": {"a": "b"}}) + ) + assert validate_iceberg_sink(contract) == ([], []) + + +@pytest.mark.parametrize("enabled", [None, True]) +def test_debezium_embedded_warehouse_override_is_cross_checked(enabled): + sink_block = {"config": {"warehouse": "s3://elsewhere/wh"}} + if enabled is not None: + sink_block["iceberg_sink_enabled"] = enabled + contract = _sink_contract(_GLUE_AWS, engine="debezium", debezium=_embedded(sink_block)) + warns = [w for w in _warns(contract) if "diverges from the binding warehouse" in w] + assert len(warns) == 1 + assert warns[0].startswith( + "iceberg sink (build 'ingest'): debezium.server.sink.config warehouse 's3://elsewhere/wh'" + ) + assert "static Glue table" in warns[0] + + +def test_debezium_embedded_matching_warehouse_override_is_clean(): + sink_block = { + "iceberg_sink_enabled": True, + "config": {"warehouse": "s3://lake/default/events/"}, + } + contract = _sink_contract(_GLUE_AWS, engine="debezium", debezium=_embedded(sink_block)) + assert not any("diverges" in w for w in _warns(contract)) + + +# ── runner preflight: ONE build's errors, warnings logged ─────────────────── + + +def test_preflight_selects_only_the_named_build(): + contract = _sink_contract(_loc_binding("lakekeeper", **COMPLETE)) + bad = contract["builds"][0] + bad["properties"]["kafka-connect"] = {"streamingSink": {"upsertMode": True}} + good = {**bad, "id": "good", "properties": {**bad["properties"], "kafka-connect": {}}} + contract["builds"].append(good) + # `fluid validate` sees the one defect ... + assert len(_errs(contract)) == 1 + # ... and each runner sees only its own build's. + msg = iceberg_sink_preflight(contract, "ingest") + assert msg is not None and "build 'ingest'" in msg and "upsertMode" in msg + assert iceberg_sink_preflight(contract, "good") is None + + +def test_preflight_clean_build_returns_none_and_logs_warnings(caplog): + binding = _loc_binding("nessie", uri="http://nessie:19120/api/v2", warehouse="s3://b/w") + with caplog.at_level("WARNING", logger="fluid.acquire.iceberg_sink"): + assert iceberg_sink_preflight(_sink_contract(binding), "ingest") is None + assert any("iceberg-nessie" in r.getMessage() for r in caplog.records) + + +def test_preflight_ignores_unknown_build_ids(): + assert iceberg_sink_preflight(_sink_contract(_BROKEN), "not-this-one") is None + + +def test_preflight_matches_the_run_context_id_of_an_id_less_build(): + contract = _sink_contract(_BROKEN) + del contract["builds"][0]["id"] + # build_acquisition_run_context ids it "unknown"; the message keeps "?". + msg = iceberg_sink_preflight(contract, "unknown") + assert msg is not None and "build '?'" in msg diff --git a/tests/cli/test_contract_validation_iceberg_catalog.py b/tests/cli/test_contract_validation_iceberg_catalog.py new file mode 100644 index 00000000..df95de3c --- /dev/null +++ b/tests/cli/test_contract_validation_iceberg_catalog.py @@ -0,0 +1,83 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""``fluid test`` on AWS reads Glue, so it must not look for a table that +``fluid apply`` deliberately keeps out of Glue. + +Before the catalog-kind table, AWS IaC created a (metadata-less) Glue table +for every Iceberg expose, so this lookup "passed" against the phantom. Now a +Lakekeeper or REST-catalog table has no Glue twin, and the lookup would fail +every such contract with "does not exist in AWS Glue catalog". +""" + +from __future__ import annotations + +from datetime import datetime +from pathlib import Path +from unittest.mock import MagicMock + +import pytest + +from fluid_build.cli.contract_validation import ContractValidator, ValidationReport + +pytestmark = pytest.mark.unit + + +def _validator() -> ContractValidator: + v = ContractValidator(Path("c.yaml"), provider_name="aws", use_cache=False, check_data=True) + v.report = ValidationReport( + contract_path="c.yaml", + contract_id="sales.orders", + contract_version="1", + validation_time=datetime(2026, 10, 1), + duration=0.0, + ) + v.validation_provider = MagicMock() + v.validation_provider.get_resource_schema.return_value = None + v.validation_provider.validate_resource.return_value = MagicMock(issues=[], success=True) + v.cache = None + v.history = None + return v + + +def _expose(fmt: str, catalog=None): + location = {"database": "sales", "table": "orders", "bucket": "lake"} + if catalog: + location["catalog"] = catalog + return { + "exposeId": "orders", + "binding": {"platform": "aws", "format": fmt, "location": location}, + } + + +@pytest.mark.parametrize("catalog", ["lakekeeper", "Lakekeeper", "rest", "iceberg_rest", "polaris"]) +def test_an_iceberg_table_in_another_catalog_is_not_looked_up_in_glue(catalog): + v = _validator() + v._validate_against_actual_resource(_expose("iceberg", catalog), "exposes[0]") + v.validation_provider.get_resource_schema.assert_not_called() + assert not v.report.get_errors() + info = [i for i in v.report.issues if i.severity == "info"] + assert len(info) == 1 + assert "not AWS Glue" in info[0].message + assert catalog.lower().replace("_", "-").replace("iceberg-rest", "rest") in info[0].message + + +@pytest.mark.parametrize( + "fmt,catalog", + [("iceberg", None), ("iceberg", "glue"), ("parquet", None), ("parquet", "lakekeeper")], +) +def test_glue_tables_are_still_checked(fmt, catalog): + v = _validator() + v._validate_against_actual_resource(_expose(fmt, catalog), "exposes[0]") + v.validation_provider.get_resource_schema.assert_called_once() diff --git a/tests/cli/test_diff_live_drift.py b/tests/cli/test_diff_live_drift.py index e2401374..a48493ba 100644 --- a/tests/cli/test_diff_live_drift.py +++ b/tests/cli/test_diff_live_drift.py @@ -753,6 +753,35 @@ def test_glue_access_denied_is_not_read_as_absent(workspace, built_providers, gl assert "AccessDeniedException" in expose["detail"] +@pytest.mark.parametrize( + ("catalog", "kind"), + [("lakekeeper", "lakekeeper"), ("iceberg_rest", "rest"), ("nessie", "nessie")], +) +def test_an_iceberg_table_in_another_catalog_is_not_read_from_glue( + workspace, built_providers, monkeypatch, catalog, kind +): + """Apply creates no Glue table for it, so a Glue table of that name would be + somebody else's: neither "absent" nor a comparison with it is true. Runs + everywhere, because no AWS call may be made.""" + from fluid_build.providers import aws_validation + + def _no_aws(*_a: Any, **_k: Any) -> None: + raise AssertionError("the Glue inspector reached AWS for a non-Glue Iceberg table") + + monkeypatch.setattr(aws_validation, "AWSValidationProvider", _no_aws) + binding = json.loads(json.dumps(AWS_BINDING)) + binding["format"] = "iceberg" + binding["location"]["catalog"] = catalog + contract = _write_contract(workspace, _contract(binding)) + + rc, event = _invoke(["diff", str(contract), "--out", "diff.json", "--exit-on-drift"]) + + (expose,) = _report(workspace / "diff.json")["live"]["exposes"] + assert (rc, event) == (0, None) + assert expose["status"] == "not_checked" + assert expose["detail"] == f"table lives in Iceberg catalog {kind}; Glue is not inspected" + + def test_aws_overlay_plans_in_the_binding_region_and_reaches_the_comparison( workspace, glue, monkeypatch ): diff --git a/tests/cli/test_validate_iceberg_gates_fail_closed.py b/tests/cli/test_validate_iceberg_gates_fail_closed.py new file mode 100644 index 00000000..2f2bfd88 --- /dev/null +++ b/tests/cli/test_validate_iceberg_gates_fail_closed.py @@ -0,0 +1,107 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""A crash inside an Iceberg / Confluent validate gate is an ERROR, always. + +``fluid validate`` wrapped each of these gates in ``except Exception`` and +reported the exception only under ``--verbose``. A validator that raised on +some contract shape therefore passed that contract as VALID with the gate +silently switched off: the silent-no-op class, in the one place whose job is +to refuse silent no-ops. The gates now fail closed. +""" + +from __future__ import annotations + +import logging +from types import SimpleNamespace + +import pytest + +from fluid_build.cli.validate import _run_contract_rules +from fluid_build.schema_manager import ValidationResult + +pytestmark = pytest.mark.unit + +#: (module, function, the label the error names) for each gate. +GATES = [ + ( + "fluid_build.build_runners.kafka_connect.iceberg_sink_validation", + "validate_iceberg_sink", + "Iceberg sink check", + ), + ( + "fluid_build.iac.providers.confluent", + "validate_confluent_binding", + "Confluent binding check", + ), + ("fluid_build.iac.iceberg_validation", "validate_iceberg_bindings", "Iceberg binding check"), +] + + +def _args(): + # --verbose OFF: the crash used to be visible only with it on. + return SimpleNamespace(verbose=False, quiet=True, strict=False, offline=True) + + +def _contract(): + return { + "fluidVersion": "0.7.6", + "kind": "DataProduct", + "id": "gold.orders", + "name": "orders", + "metadata": {"layer": "Gold", "owner": {"team": "dp", "email": "x@y.z"}}, + "exposes": [ + { + "exposeId": "orders", + "kind": "table", + "binding": { + "platform": "aws", + "format": "iceberg", + "location": { + "database": "sales", + "table": "orders", + "bucket": "lake", + "region": "us-east-1", + }, + }, + } + ], + } + + +def _run(): + result = ValidationResult(is_valid=True) + _run_contract_rules(_contract(), None, result, _args(), logging.getLogger("test")) + return result + + +@pytest.mark.parametrize("module, func, label", GATES, ids=[g[2] for g in GATES]) +def test_a_crashing_gate_is_reported_as_an_error(monkeypatch, module, func, label): + def _boom(contract): + raise KeyError("location") + + monkeypatch.setattr(f"{module}.{func}", _boom) + result = _run() + crash = [e for e in result.errors if e.startswith(f"{label} could not run")] + assert len(crash) == 1, result.errors + assert "KeyError" in crash[0] and "location" in crash[0] + assert "NOT checked" in crash[0] + assert result.is_valid is False + + +def test_healthy_gates_report_no_crash(): + # Negative control: the errors above come from the crash, not the contract. + result = _run() + for _, _, label in GATES: + assert not any(e.startswith(f"{label} could not run") for e in result.errors) diff --git a/tests/engines/test_dbt_catalogs_yml.py b/tests/engines/test_dbt_catalogs_yml.py index 25f466f0..32b6c070 100644 --- a/tests/engines/test_dbt_catalogs_yml.py +++ b/tests/engines/test_dbt_catalogs_yml.py @@ -23,13 +23,18 @@ from __future__ import annotations +import logging from typing import Any, Dict, Optional import pytest import yaml from fluid_build.engines.dbt.catalogs_yml import generate_catalogs_yml -from fluid_build.providers._iceberg_catalog import iceberg_external_volume_name +from fluid_build.providers._iceberg_catalog import ( + iceberg_external_volume_name, + is_glue_cataloged, + resolve_iceberg_catalog, +) pytestmark = pytest.mark.unit @@ -44,8 +49,10 @@ def _contract( catalog: Optional[str] = None, database: str = "ANALYTICS", contract_id: str = "gold.orders", + platform: str = "snowflake", + **extra_location: Any, ) -> Dict[str, Any]: - location: Dict[str, Any] = {"database": database, "table": "ORDERS"} + location: Dict[str, Any] = {"database": database, "table": "ORDERS", **extra_location} if catalog: location["catalog"] = catalog return { @@ -59,7 +66,7 @@ def _contract( "exposeId": "orders", "kind": "table", "binding": { - "platform": "snowflake", + "platform": platform, "format": fmt, "location": location, }, @@ -153,6 +160,186 @@ def test_non_glue_rest_omits_the_glue_only_property(self): assert "catalog_linked_database_type" not in integration.get("adapter_properties", {}) +class TestSnowflakeCatalogKinds: + """Snowflake's catalog type is the kind table's ``snowflake_catalog_type``. + + Before the table, this emitter kept its own lower-cased set, so + ``catalog: lakekeeper`` (and any non-canonical spelling) fell through to + ``built_in``: dbt wrote a Snowflake-managed table while the streaming sink + committed the same name to Lakekeeper over REST. + """ + + @pytest.mark.parametrize( + "catalog", ["lakekeeper", "Lakekeeper", "iceberg-rest", "ICEBERG_REST"] + ) + def test_rest_family_spellings_map_to_iceberg_rest(self, catalog): + integration = _integration(generate_catalogs_yml(_contract(catalog=catalog), _build())) + assert integration["catalog_type"] == "iceberg_rest" + assert "external_volume" not in integration + props = integration["adapter_properties"] + assert props["catalog_linked_database"] == "ANALYTICS" + assert "catalog_linked_database_type" not in props + + def test_lakekeeper_agrees_with_the_streaming_sink(self): + """The originating bug, end to end: both writers must mean a REST catalog.""" + contract = _contract(catalog="lakekeeper") + binding = contract["exposes"][0]["binding"] + assert resolve_iceberg_catalog(binding, contract=contract).catalog_type == "rest" + integration = _integration(generate_catalogs_yml(contract, _build())) + assert integration["catalog_type"] == "iceberg_rest" + + @pytest.mark.parametrize( + "catalog", ["snowflake", "Snowflake", "snowflake_managed", "snowflake-managed"] + ) + def test_snowflake_spellings_are_built_in(self, catalog): + contract = _contract(catalog=catalog) + integration = _integration(generate_catalogs_yml(contract, _build())) + assert integration["catalog_type"] == "built_in" + assert integration["external_volume"] == iceberg_external_volume_name( + contract, contract["exposes"][0]["binding"] + ) + + def test_unknown_kind_keeps_the_built_in_fallback(self): + """Historic fallback, kept so a typo does not silently flip the table + to a catalog-linked database; ``fluid validate`` reports the value.""" + integration = _integration( + generate_catalogs_yml(_contract(catalog="not-a-catalog"), _build()) + ) + assert integration["catalog_type"] == "built_in" + assert integration["external_volume"] + + @pytest.mark.parametrize("catalog", ["hive", "HIVE", "jdbc", "hadoop", "dynamodb"]) + def test_catalog_snowflake_cannot_reach_writes_no_file(self, catalog): + """No Snowflake catalog integration exists for these. ``built_in`` + would have dbt create a second, Snowflake-managed table under the name + the Hive metastore (or JDBC / Hadoop / DynamoDB catalog) owns.""" + assert generate_catalogs_yml(_contract(catalog=catalog), _build()) is None + + def test_unreachable_expose_is_omitted_beside_a_reachable_one(self): + contract = _contract(catalog="hive") + managed = {**contract["exposes"][0], "exposeId": "managed"} + managed["binding"] = {**managed["binding"], "location": {"database": "A", "table": "B"}} + contract["exposes"].append(managed) + doc = _parse(generate_catalogs_yml(contract, _build())) + assert [c["name"] for c in doc["catalogs"]] == ["managed_catalog"] + assert doc["catalogs"][0]["write_integrations"][0]["catalog_type"] == "built_in" + + def test_omission_is_logged_with_the_way_out(self, caplog): + with caplog.at_level(logging.WARNING, logger="fluid_build.engines.dbt.catalogs_yml"): + generate_catalogs_yml(_contract(catalog="hive"), _build()) + message = caplog.text + assert "'orders'" in message and "'hive'" in message + assert "lakekeeper" in message and "omit location.catalog" in message + + @pytest.mark.parametrize( + "platform, catalog", + [ + ("aws", None), + ("aws", "glue"), + ("aws", "lakekeeper"), + ("snowflake", None), + ("snowflake", "GLUE"), + ("snowflake", "polaris"), + ], + ) + def test_glue_link_type_tracks_the_aws_iac(self, platform, catalog): + """The AWS IaC creates a Glue table exactly when ``is_glue_cataloged``. + + dbt on Snowflake must reach that same table through a Glue + catalog-linked database. An absent catalog on ``platform: aws`` used + to emit ``built_in`` with a volume that nothing creates (the Snowflake + IaC only serves ``platform: snowflake`` exposes). + """ + contract = _contract(catalog=catalog, platform=platform) + binding = contract["exposes"][0]["binding"] + integration = _integration(generate_catalogs_yml(contract, _build())) + props = integration.get("adapter_properties", {}) + is_glue_link = props.get("catalog_linked_database_type") == "glue" + assert is_glue_link == is_glue_cataloged(binding) + if is_glue_link: + assert integration["catalog_type"] == "iceberg_rest" + + +class TestBigQueryCatalogKinds: + """``biglake_metastore`` is dbt-bigquery's only catalog type. + + It cannot redirect a model to an external Iceberg catalog, so emitting it + for ``catalog: lakekeeper`` had dbt create a BigLake table under the name + the streaming sink commits to in Lakekeeper. + """ + + @staticmethod + def _contract(catalog: Optional[str] = None) -> Dict[str, Any]: + return _contract(catalog=catalog, platform="gcp", bucket="lake", path="p") + + @pytest.mark.parametrize( + "catalog", ["lakekeeper", "Lakekeeper", "rest", "iceberg_rest", "glue", "not-a-catalog"] + ) + def test_non_biglake_catalog_writes_no_file(self, catalog): + assert generate_catalogs_yml(self._contract(catalog), _build("gcp")) is None + + @pytest.mark.parametrize("catalog", ["bigquery", "BigQuery", "BIGQUERY"]) + def test_biglake_spellings_are_byte_identical_to_absent(self, catalog): + absent = generate_catalogs_yml(self._contract(), _build("gcp")) + assert generate_catalogs_yml(self._contract(catalog), _build("gcp")) == absent + assert "catalog_type: biglake_metastore" in absent + + def test_lakekeeper_expose_is_omitted_beside_a_biglake_one(self): + contract = self._contract("lakekeeper") + biglake = {**contract["exposes"][0], "exposeId": "biglake"} + biglake["binding"] = { + **biglake["binding"], + "location": {"dataset": "d", "table": "T", "bucket": "lake"}, + } + contract["exposes"].append(biglake) + doc = _parse(generate_catalogs_yml(contract, _build("gcp"))) + assert [c["name"] for c in doc["catalogs"]] == ["biglake_catalog"] + + def test_omission_is_logged_with_the_way_out(self, caplog): + with caplog.at_level(logging.WARNING, logger="fluid_build.engines.dbt.catalogs_yml"): + generate_catalogs_yml(self._contract("lakekeeper"), _build("gcp")) + assert "dbt-bigquery" in caplog.text and "'lakekeeper'" in caplog.text + + +class TestByteIdenticalShapes: + """Golden pins for the shapes the catalog-kind table must not move.""" + + _PREFIX = ( + "# Generated by fluid generate. Do not edit manually.\n" + "# Regenerate with: fluid generate\n\n" + "catalogs:\n" + "- name: orders_catalog\n" + " active_write_integration: orders_write_integration\n" + " write_integrations:\n" + " - name: orders_write_integration\n" + ) + + def test_snowflake_absent_catalog(self): + assert generate_catalogs_yml(_contract(), _build()) == self._PREFIX + ( + " table_format: iceberg\n" + " catalog_type: built_in\n" + " external_volume: FLUID_GOLD_ORDERS_VOL\n" + ) + + def test_snowflake_glue_catalog(self): + assert generate_catalogs_yml(_contract(catalog="glue"), _build()) == self._PREFIX + ( + " table_format: iceberg\n" + " catalog_type: iceberg_rest\n" + " adapter_properties:\n" + " catalog_linked_database: ANALYTICS\n" + " catalog_linked_database_type: glue\n" + ) + + def test_bigquery_absent_catalog(self): + contract = _contract(platform="gcp", bucket="lake", path="p") + assert generate_catalogs_yml(contract, _build("gcp")) == self._PREFIX + ( + " external_volume: gs://lake/p\n" + " table_format: iceberg\n" + " file_format: parquet\n" + " catalog_type: biglake_metastore\n" + ) + + class TestEmitsNothingWhenItShould: def test_no_iceberg_expose_writes_no_file(self): assert generate_catalogs_yml(_contract(fmt="snowflake_table"), _build()) is None diff --git a/tests/iac/test_iac_aws_iceberg_catalog.py b/tests/iac/test_iac_aws_iceberg_catalog.py new file mode 100644 index 00000000..3980bd3f --- /dev/null +++ b/tests/iac/test_iac_aws_iceberg_catalog.py @@ -0,0 +1,419 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The AWS emitter reads an Iceberg expose's ``location.catalog``. + +``catalog: lakekeeper`` used to stream over REST while the AWS IaC created a +Glue database and table for the same name: a second, metadata-less claim on a +table that lived in Lakekeeper. Now an Iceberg table in a non-Glue catalog gets +its S3 bucket (the catalog's storage profile still needs it) and no Glue +resource, every Lake Formation gate agrees (an LF resource naming an undeclared +``aws_glue_catalog_table`` fails ``tofu validate``), and Lake Formation on it is +refused at emit and at validate rather than dropped. Absent or ``glue`` on AWS +is the Glue catalog, and its emit is unchanged byte for byte. + +Pure-function tests: no credentials, no network. +""" + +from __future__ import annotations + +import copy +import json +import re +import shutil +import subprocess +from typing import Any, Dict, List, Optional + +import pytest + +from fluid_build.iac import build_module, get_iac_plugin +from fluid_build.iac.base import UnsupportedBindingError +from fluid_build.iac.governance_validation import validate_governance +from fluid_build.iac.providers import aws as aws_mod + +pytestmark = [pytest.mark.unit, pytest.mark.provider] + +#: Every non-Glue spelling the emitter must treat as "not in Glue". +NON_GLUE_CATALOGS = ["lakekeeper", "rest", "iceberg-rest", "iceberg_rest", "polaris", "nessie"] +GLUE_TYPES = ("aws_glue_catalog_database", "aws_glue_catalog_table") +ANALYST = "arn:aws:iam::111111111111:role/analyst" +SCHEMA = [{"name": "order_id", "type": "string"}, {"name": "qty", "type": "integer"}] + + +def _aws(): + return get_iac_plugin("aws") + + +def _expose( + fmt: str = "iceberg", + catalog: Optional[str] = None, + *, + expose_id: str = "orders", + table: str = "orders", + governance: Optional[Dict[str, Any]] = None, + policy: Optional[Dict[str, Any]] = None, +) -> Dict[str, Any]: + location: Dict[str, Any] = { + "database": "sales", + "table": table, + "bucket": "lake", + "path": f"{table}/", + } + if catalog is not None: + location["catalog"] = catalog + binding: Dict[str, Any] = {"platform": "aws", "format": fmt, "location": location} + if governance is not None: + binding["governance"] = governance + exposure: Dict[str, Any] = { + "exposeId": expose_id, + "binding": binding, + "contract": {"schema": copy.deepcopy(SCHEMA)}, + } + if policy is not None: + exposure["policy"] = policy + return exposure + + +def _contract(*exposes: Dict[str, Any]) -> Dict[str, Any]: + return {"id": "analytics.lake", "name": "Lake", "exposes": list(exposes)} + + +def _lake_formation() -> Dict[str, Any]: + return { + "lakeFormation": { + "registerLocation": True, + "grants": [{"principal": ANALYST, "permissions": ["SELECT", "DESCRIBE"]}], + "tags": {"tier": "gold"}, + "rowFilter": {"name": "eu", "rowExpression": "qty > 0"}, + } + } + + +def _without_contract_yaml(resources: Dict[str, Any]) -> Dict[str, Any]: + """The emit minus ``parameters.fluid_contract``: the contract's own YAML, which + differs between two contracts by exactly the ``catalog`` line under test.""" + out = copy.deepcopy(resources) + for table in (out.get("aws_glue_catalog_table") or {}).values(): + table["parameters"].pop("fluid_contract", None) + return out + + +def _dangling_glue_refs(resources: Dict[str, Any]) -> List[str]: + """``${aws_glue_catalog_*....}`` references whose resource is not emitted.""" + text = json.dumps(resources, default=str) + refs = set(re.findall(r"\$\{(aws_glue_catalog_(?:table|database))\.([A-Za-z0-9_]+)\.", text)) + return sorted(f"{t}.{k}" for t, k in refs if k not in (resources.get(t) or {})) + + +# --------------------------------------------------------------------------- +# A non-Glue Iceberg catalog: the bucket, no Glue +# --------------------------------------------------------------------------- + + +class TestNonGlueIcebergCatalog: + @pytest.mark.parametrize("catalog", NON_GLUE_CATALOGS) + def test_emits_the_bucket_and_no_glue_database_or_table(self, catalog): + resources = _aws().emit(_contract(_expose("iceberg", catalog))) + + for resource_type in GLUE_TYPES: + assert resource_type not in resources, (catalog, resource_type) + # The catalog's storage profile still writes into the bucket. + assert resources["aws_s3_bucket"]["analytics_lake_lake"]["bucket"] == "lake" + + @pytest.mark.parametrize("catalog", NON_GLUE_CATALOGS) + def test_imports_no_glue_resource_and_resolves_no_account(self, catalog, monkeypatch): + """No ``tofu import`` of a Glue table of the same name (it would adopt + someone else's table into this state), and no STS call to resolve a + catalog id nothing needs.""" + + def _no_sts() -> str: + raise AssertionError("discover_imports resolved a Glue catalog id") + + monkeypatch.setattr(aws_mod, "_resolve_catalog_id", _no_sts) + blocks = _aws().discover_imports(_contract(_expose("iceberg", catalog))) + + addresses = [block.to for block in blocks] + assert not [a for a in addresses if a.startswith("aws_glue_")] + assert "aws_s3_bucket.analytics_lake_lake" in addresses + + def test_a_glue_database_another_expose_uses_is_kept(self, monkeypatch): + monkeypatch.setenv("AWS_ACCOUNT_ID", "123456789012") + contract = _contract( + _expose("parquet", expose_id="facts", table="facts"), + _expose("iceberg", "lakekeeper"), + ) + + resources = _aws().emit(contract) + addresses = {block.to for block in _aws().discover_imports(contract)} + + assert list(resources["aws_glue_catalog_database"]) == ["analytics_lake_sales"] + assert list(resources["aws_glue_catalog_table"]) == ["analytics_lake_sales_facts"] + assert "aws_glue_catalog_table.analytics_lake_sales_orders" not in addresses + assert "aws_glue_catalog_table.analytics_lake_sales_facts" in addresses + assert "aws_glue_catalog_database.analytics_lake_sales" in addresses + + def test_iceberg_table_format_spelling_is_the_same_table(self): + resources = _aws().emit(_contract(_expose("ICEBERG", "Lakekeeper"))) + assert not set(GLUE_TYPES) & set(resources) + + +# --------------------------------------------------------------------------- +# Glue, absent or explicit: unchanged +# --------------------------------------------------------------------------- + +#: The emit of an Iceberg expose with no ``catalog`` (minus the contract YAML +#: parameter), as forge-cli emitted it before ``location.catalog`` was read. +ABSENT_CATALOG_GOLDEN: Dict[str, Any] = { + "aws_glue_catalog_database": { + "analytics_lake_sales": {"name": "sales", "lifecycle": {"ignore_changes": ["parameters"]}} + }, + "aws_glue_catalog_table": { + "analytics_lake_sales_orders": { + "name": "orders", + "database_name": "${aws_glue_catalog_database.analytics_lake_sales.name}", + "table_type": "EXTERNAL_TABLE", + "parameters": { + "classification": "iceberg", + "managed_by": "fluid", + "table_type": "ICEBERG", + }, + "storage_descriptor": { + "columns": [{"name": "order_id", "type": "string"}, {"name": "qty", "type": "int"}], + "location": "s3://lake/orders/", + }, + "lifecycle": {"ignore_changes": ["parameters"]}, + } + }, + "aws_s3_bucket": { + "analytics_lake_lake": { + "bucket": "lake", + "force_destroy": True, + "tags": {"managed_by": "fluid", "fluid_contract": "analytics_lake"}, + } + }, +} + + +class TestGlueCatalogUnchanged: + def test_absent_catalog_emit_is_the_golden(self): + resources = _aws().emit(_contract(_expose("iceberg"))) + assert json.dumps(_without_contract_yaml(resources), sort_keys=True) == json.dumps( + ABSENT_CATALOG_GOLDEN, sort_keys=True + ) + + @pytest.mark.parametrize("spelling", ["glue", "GLUE", " Glue "]) + def test_explicit_glue_emits_what_absent_does(self, spelling, monkeypatch): + monkeypatch.setenv("AWS_ACCOUNT_ID", "123456789012") + absent = _contract(_expose("iceberg", governance=_lake_formation())) + explicit = _contract(_expose("iceberg", spelling, governance=_lake_formation())) + plugin = _aws() + + assert json.dumps(_without_contract_yaml(plugin.emit(absent)), sort_keys=True) == ( + json.dumps(_without_contract_yaml(plugin.emit(explicit)), sort_keys=True) + ) + assert json.dumps(plugin.emit_data(absent), sort_keys=True, default=str) == json.dumps( + plugin.emit_data(explicit), sort_keys=True, default=str + ) + assert plugin.discover_imports(absent) == plugin.discover_imports(explicit) + + def test_glue_lake_formation_still_emits_against_the_glue_table(self): + resources = _aws().emit(_contract(_expose("iceberg", "glue", governance=_lake_formation()))) + assert resources["aws_lakeformation_permissions"] + assert resources["aws_lakeformation_resource_lf_tags"] + assert resources["aws_lakeformation_data_cells_filter"] + assert _dangling_glue_refs(resources) == [] + + @pytest.mark.parametrize("catalog", ["lakekeeper", "rest"]) + def test_a_parquet_binding_keeps_its_glue_table_whatever_catalog_says( + self, catalog, monkeypatch + ): + """The catalog kind is an Iceberg property: Glue catalogs a parquet table.""" + monkeypatch.setenv("AWS_ACCOUNT_ID", "123456789012") + contract = _contract(_expose("parquet", catalog)) + + resources = _aws().emit(contract) + addresses = {block.to for block in _aws().discover_imports(contract)} + + assert "analytics_lake_sales_orders" in resources["aws_glue_catalog_table"] + assert "aws_glue_catalog_table.analytics_lake_sales_orders" in addresses + + +# --------------------------------------------------------------------------- +# Lake Formation and column restrictions on a non-Glue table: refused +# --------------------------------------------------------------------------- + + +class TestLakeFormationNeedsGlue: + def test_emit_refuses_lake_formation_on_a_lakekeeper_table(self): + with pytest.raises(UnsupportedBindingError) as excinfo: + _aws().emit(_contract(_expose("iceberg", "lakekeeper", governance=_lake_formation()))) + + assert excinfo.value.kind == "lake-formation-needs-glue-catalog" + assert "exposes[orders]" in str(excinfo.value) + assert "'lakekeeper' catalog" in str(excinfo.value) + assert any("lakekeeper catalog itself" in r for r in excinfo.value.remediation) + + def test_build_module_refuses_it_too(self): + contract = _contract(_expose("iceberg", "polaris", governance=_lake_formation())) + with pytest.raises(UnsupportedBindingError) as excinfo: + build_module(_aws(), contract) + assert excinfo.value.kind == "lake-formation-needs-glue-catalog" + + def test_validate_reports_it_with_the_emit_message(self): + contract = _contract(_expose("iceberg", "lakekeeper", governance=_lake_formation())) + + errors, _ = validate_governance(contract) + + assert len(errors) == 1 + assert "governance.lakeFormation" in errors[0] + assert "'lakekeeper' catalog" in errors[0] + + def test_validate_passes_a_lakekeeper_table_without_lake_formation(self): + errors, _ = validate_governance(_contract(_expose("iceberg", "lakekeeper"))) + assert errors == [] + + def test_unenforced_access_points_at_the_catalog_not_at_lake_formation(self): + """The Glue binding's warning says "add governance.lakeFormation.grants"; + for a Lakekeeper table that advice would be refused, so it names the catalog.""" + contract = _contract( + _expose("parquet", expose_id="facts", table="facts"), + _expose("iceberg", "lakekeeper"), + ) + contract["accessPolicy"] = {"grants": [{"principal": ANALYST, "permissions": ["read"]}]} + + errors, warnings = validate_governance(contract) + + assert errors == [] + glue_warning, lakekeeper_warning = warnings + assert "aws binding(s) facts:" in glue_warning + assert "governance.lakeFormation.grants" in glue_warning + assert "aws binding(s) orders (lakekeeper):" in lakekeeper_warning + assert "Grant access in that catalog" in lakekeeper_warning + assert "governance.lakeFormation" not in lakekeeper_warning + + def test_column_restrictions_name_the_catalog_not_lake_formation(self): + """``column_access`` alone would say "add Lake Formation grants", which the + refusal above then refuses: the author would be sent in a circle.""" + policy = { + "authz": { + "readers": [ANALYST], + "columnRestrictions": [ + {"principal": ANALYST, "columns": ["qty"], "access": "deny"} + ], + } + } + contract = _contract(_expose("iceberg", "lakekeeper", policy=policy)) + + with pytest.raises(UnsupportedBindingError) as excinfo: + _aws().emit(contract) + errors, _ = validate_governance(contract) + + assert excinfo.value.kind == "column-restriction-unenforceable" + assert "'lakekeeper' catalog" in str(excinfo.value) + assert "add governance.lakeFormation" not in str(excinfo.value).lower() + assert len(errors) == 1 and "'lakekeeper' catalog" in errors[0] + + +# --------------------------------------------------------------------------- +# Every gate shares one predicate +# --------------------------------------------------------------------------- + + +class TestRowFilterNeedsGlue: + """``policy.authz.rowFilters`` is enforced only as a Lake Formation data cells + filter on a Glue table. On a Lakekeeper table it is refused by catalog name, + not with the generic "add Lake Formation grants" advice, which + :func:`refuse_lake_formation_on_external_catalog` would then refuse.""" + + ANALYSTS = "group:analysts" + + def _row_filtered(self, catalog: Optional[str]) -> Dict[str, Any]: + exposure = _expose( + "iceberg", + catalog, + policy={ + "authz": { + "rowFilters": [ + {"principal": self.ANALYSTS, "name": "eu_only", "where": "qty > 0"} + ] + } + }, + ) + exposure["binding"]["principals"] = { + self.ANALYSTS: "arn:aws:iam::111122223333:role/analyst" + } + return _contract(exposure) + + @pytest.mark.parametrize("catalog", ["lakekeeper", "iceberg-rest", "polaris"]) + def test_emit_refuses_by_catalog(self, catalog): + with pytest.raises(UnsupportedBindingError) as info: + _aws().emit(self._row_filtered(catalog)) + assert info.value.kind == "row-filter-unenforceable" + assert "not in AWS Glue" in str(info.value) + assert "governance.lakeFormation grants" not in str(info.value) + + def test_validate_refuses_with_the_same_message(self): + errors, _ = validate_governance(self._row_filtered("lakekeeper")) + assert any( + "policy.authz.rowFilters" in e and "'lakekeeper' catalog" in e for e in errors + ), errors + + +class TestOnePredicate: + def test_no_lake_formation_resource_names_an_undeclared_glue_table(self): + """A mixed contract: a Glue parquet table under Lake Formation beside a + Lakekeeper table without it. Every LF reference resolves.""" + contract = _contract( + _expose("parquet", expose_id="facts", table="facts", governance=_lake_formation()), + _expose("iceberg", "lakekeeper"), + ) + resources = _aws().emit(contract) + assert resources["aws_lakeformation_permissions"] + assert _dangling_glue_refs(resources) == [] + + def test_every_glue_format_gate_goes_through_the_predicate(self): + """The format set is tested in ``_glue_cataloged`` alone: a gate that keys on + it directly would skip the catalog half, and an LF resource it lets through + would name an undeclared Glue table.""" + import inspect + + from fluid_build.cli import _diff_live + + assert len(re.findall(r"\bin _GLUE_CATALOG_FORMATS\b", inspect.getsource(aws_mod))) == 1 + assert "_GLUE_CATALOG_FORMATS" not in inspect.getsource(_diff_live) + + +_TOFU = shutil.which("tofu") + + +@pytest.mark.integration +@pytest.mark.skipif(_TOFU is None, reason="tofu not on PATH") +def test_a_mixed_module_passes_tofu_validate(tmp_path): + """The real check the shared predicate exists for: a Glue table under Lake + Formation beside a Lakekeeper table. An LF resource naming the Lakekeeper + table's (absent) Glue table fails ``Reference to undeclared resource``.""" + from tests.iac.test_iac_tofu_validate import _tofu_init_or_skip + + contract = _contract( + _expose("parquet", expose_id="facts", table="facts", governance=_lake_formation()), + _expose("iceberg", "lakekeeper"), + ) + # The LF-tag the parquet table is associated with must be defined. + contract["governance"] = {"lakeFormation": {"tagDefinitions": {"tier": ["gold"]}}} + (tmp_path / "main.tf.json").write_text(build_module(_aws(), contract)) + _tofu_init_or_skip(tmp_path) + done = subprocess.run( + [_TOFU, "validate", "-no-color"], cwd=tmp_path, capture_output=True, text=True + ) + assert done.returncode == 0, done.stderr or done.stdout diff --git a/tests/iac/test_iac_catalog_moves.py b/tests/iac/test_iac_catalog_moves.py new file mode 100644 index 00000000..2bc289d5 --- /dev/null +++ b/tests/iac/test_iac_catalog_moves.py @@ -0,0 +1,304 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The pre-plan guard for Glue resources of an Iceberg table that moved catalogs. + +A contract the previous release applied with ``catalog: lakekeeper`` (or rest, +polaris, nessie...) has a Glue database and table in its OpenTofu state, because +that release created them whatever the catalog said. The emitter no longer +does, so the next plan would DESTROY them, and destroying a Glue database +deletes every table in it. ``fluid apply`` fails closed before ``tofu plan`` +with the ``tofu state rm`` commands instead (``iac/catalog_moves.py``). +""" + +from __future__ import annotations + +import copy +import logging +from typing import Any, Dict, List, Optional + +import pytest + +from fluid_build.iac.catalog_moves import ( + CatalogMoveError, + detect_catalog_moves, + guard_catalog_moves, + moved_iceberg_exposes, +) +from fluid_build.iac.providers.aws import AwsIacPlugin + +pytestmark = [pytest.mark.unit, pytest.mark.provider] + +SCHEMA = [{"name": "order_id", "type": "string"}, {"name": "qty", "type": "integer"}] +DB = "aws_glue_catalog_database.analytics_lake_sales" +ORDERS = "aws_glue_catalog_table.analytics_lake_sales_orders" +FACTS = "aws_glue_catalog_table.analytics_lake_sales_facts" +BUCKET = "aws_s3_bucket.analytics_lake_lake" + + +def _expose( + fmt: str = "iceberg", + catalog: Optional[str] = None, + *, + expose_id: str = "orders", + table: str = "orders", +) -> Dict[str, Any]: + location: Dict[str, Any] = { + "database": "sales", + "table": table, + "bucket": "lake", + "path": f"{table}/", + } + if catalog is not None: + location["catalog"] = catalog + return { + "exposeId": expose_id, + "binding": {"platform": "aws", "format": fmt, "location": location}, + "contract": {"schema": copy.deepcopy(SCHEMA)}, + } + + +def _contract(*exposes: Dict[str, Any]) -> Dict[str, Any]: + return {"id": "analytics.lake", "name": "Lake", "exposes": list(exposes)} + + +class _SpyPlugin(AwsIacPlugin): + """The real AWS plugin, counting its emits (the fast path makes none).""" + + def __init__(self) -> None: + self.emits = 0 + + def emit(self, contract, actions=()): + self.emits += 1 + return super().emit(contract, actions) + + +# --------------------------------------------------------------------------- +# detection +# --------------------------------------------------------------------------- + + +class TestDetect: + @pytest.mark.parametrize( + "catalog", ["lakekeeper", "rest", "iceberg-rest", "iceberg_rest", "polaris", "nessie"] + ) + def test_the_old_glue_resources_in_state_are_found(self, catalog): + contract = _contract(_expose("iceberg", catalog)) + state = [BUCKET, DB, ORDERS] + + assert detect_catalog_moves(AwsIacPlugin(), contract, state) == (DB, ORDERS) + + def test_only_what_the_state_holds_is_reported(self): + contract = _contract(_expose("iceberg", "lakekeeper")) + assert detect_catalog_moves(AwsIacPlugin(), contract, [BUCKET, ORDERS]) == (ORDERS,) + + def test_a_state_without_them_is_a_no_op(self): + """Applied by this release from the start: nothing to release.""" + contract = _contract(_expose("iceberg", "lakekeeper")) + assert detect_catalog_moves(AwsIacPlugin(), contract, [BUCKET]) == () + + def test_an_empty_state_emits_nothing(self): + plugin = _SpyPlugin() + contract = _contract(_expose("iceberg", "lakekeeper")) + assert detect_catalog_moves(plugin, contract, []) == () + assert plugin.emits == 0 + + @pytest.mark.parametrize("catalog", [None, "glue", "GLUE"]) + def test_a_glue_contract_is_a_no_op_without_an_emit(self, catalog): + plugin = _SpyPlugin() + contract = _contract(_expose("iceberg", catalog), _expose("parquet", "lakekeeper")) + + assert moved_iceberg_exposes(contract) == () + assert detect_catalog_moves(plugin, contract, [BUCKET, DB, ORDERS]) == () + assert plugin.emits == 0 + + def test_a_glue_database_a_parquet_expose_still_uses_is_not_flagged(self): + contract = _contract( + _expose("parquet", expose_id="facts", table="facts"), + _expose("iceberg", "lakekeeper"), + ) + state = [BUCKET, DB, FACTS, ORDERS] + + assert detect_catalog_moves(AwsIacPlugin(), contract, state) == (ORDERS,) + + def test_a_module_prefixed_address_is_not_this_module_s(self): + """``fluid`` writes a root module: a child module's resource is not one + this emit change removes, so it is matched exactly, never by name.""" + contract = _contract(_expose("iceberg", "lakekeeper")) + state = [f"module.other.{DB}", f"module.other.{ORDERS}"] + assert detect_catalog_moves(AwsIacPlugin(), contract, state) == () + + def test_off_aws_and_without_a_database_nothing_moves(self): + gcp = _expose("iceberg", "lakekeeper") + gcp["binding"]["platform"] = "gcp" + no_db = _expose("iceberg", "lakekeeper") + del no_db["binding"]["location"]["database"] + assert moved_iceberg_exposes(_contract(gcp, no_db)) == () + + def test_the_caller_s_contract_is_not_mutated(self): + contract = _contract(_expose("iceberg", "lakekeeper")) + before = copy.deepcopy(contract) + detect_catalog_moves(AwsIacPlugin(), contract, [DB, ORDERS]) + assert contract == before + + +# --------------------------------------------------------------------------- +# the guard and its message +# --------------------------------------------------------------------------- + + +class TestGuard: + def test_raises_with_copy_pasteable_state_rm_commands(self): + contract = _contract(_expose("iceberg", "lakekeeper")) + + with pytest.raises(CatalogMoveError) as excinfo: + guard_catalog_moves( + AwsIacPlugin(), contract, [BUCKET, DB, ORDERS], workdir="/w/aws/analytics lake" + ) + + exc = excinfo.value + assert exc.kind == "iceberg-catalog-move" + assert exc.addresses == (DB, ORDERS) + assert exc.remediation == ( + f"tofu -chdir='/w/aws/analytics lake' state rm {DB}", + f"tofu -chdir='/w/aws/analytics lake' state rm {ORDERS}", + ) + message = str(exc) + assert "exposes[orders]: location.catalog lakekeeper" in message + assert "DESTROY" in message and "deletes every table in it" in message + assert "only this contract's claim on them is released" in message + assert exc.event_fields() == { + "kind": "iceberg-catalog-move", + "addresses": [DB, ORDERS], + "exposes": [{"expose": "orders", "catalog": "lakekeeper"}], + "remediation": list(exc.remediation), + } + + def test_without_a_workdir_the_commands_run_where_they_are(self): + with pytest.raises(CatalogMoveError) as excinfo: + guard_catalog_moves( + AwsIacPlugin(), _contract(_expose("iceberg", "rest")), [ORDERS], workdir=None + ) + assert excinfo.value.remediation == (f"tofu state rm {ORDERS}",) + + def test_passes_when_nothing_would_be_destroyed(self): + assert ( + guard_catalog_moves(AwsIacPlugin(), _contract(_expose("iceberg", "rest")), [BUCKET]) + is None + ) + + +# --------------------------------------------------------------------------- +# the apply engine adapter +# --------------------------------------------------------------------------- + + +class TestEngineAdapter: + @pytest.fixture + def engine(self): + from fluid_build.cli import _apply_opentofu_engine as engine + + return engine + + def _listing(self, monkeypatch, engine, state: List[str]) -> List[str]: + calls: List[str] = [] + + def _list(*_a, **_k): + calls.append("listed") + return list(state) + + monkeypatch.setattr(engine.runner, "tofu_state_list", _list) + return calls + + def test_translates_to_a_cli_error_with_remediation(self, monkeypatch, engine, caplog): + from fluid_build.cli._common import CLIError + + self._listing(monkeypatch, engine, [BUCKET, DB, ORDERS]) + logger = logging.getLogger("test.catalog_moves") + + with caplog.at_level(logging.DEBUG, logger="test.catalog_moves"): + with pytest.raises(CLIError) as excinfo: + engine._guard_catalog_moves( + AwsIacPlugin(), + _contract(_expose("iceberg", "lakekeeper")), + "aws", + "/w", + {}, + logger, + ) + + assert excinfo.value.event == "iceberg_catalog_move_blocked" + context = excinfo.value.context + assert context["kind"] == "iceberg-catalog-move" + assert context["remediation"] == [ + f"tofu -chdir=/w state rm {DB}", + f"tofu -chdir=/w state rm {ORDERS}", + ] + # The audit event is recorded before the raise. + assert "iceberg_catalog_move_blocked" in caplog.text + + def test_a_contract_with_nothing_moved_lists_no_state(self, monkeypatch, engine): + calls = self._listing(monkeypatch, engine, [BUCKET, DB, ORDERS]) + engine._guard_catalog_moves( + AwsIacPlugin(), _contract(_expose("iceberg")), "aws", "/w", {}, logging.getLogger("t") + ) + assert calls == [] + + def test_another_provider_is_untouched(self, monkeypatch, engine): + calls = self._listing(monkeypatch, engine, [DB, ORDERS]) + engine._guard_catalog_moves( + AwsIacPlugin(), + _contract(_expose("iceberg", "lakekeeper")), + "gcp", + "/w", + {}, + logging.getLogger("t"), + ) + assert calls == [] + + def test_an_empty_state_passes(self, monkeypatch, engine): + self._listing(monkeypatch, engine, []) + engine._guard_catalog_moves( + _SpyPlugin(), + _contract(_expose("iceberg", "lakekeeper")), + "aws", + "/w", + {}, + logging.getLogger("t"), + ) + + def test_a_failed_probe_does_not_fail_the_apply(self, monkeypatch, engine): + class _Broken(AwsIacPlugin): + def emit(self, contract, actions=()): + raise RuntimeError("probe exploded") + + self._listing(monkeypatch, engine, [DB, ORDERS]) + engine._guard_catalog_moves( + _Broken(), + _contract(_expose("iceberg", "lakekeeper")), + "aws", + "/w", + {}, + logging.getLogger("t"), + ) + + def test_the_engine_runs_it_after_the_packaging_guard_and_before_plan(self, engine): + import inspect + + source = inspect.getsource(engine.apply_via_opentofu) + packaging_at = source.index("_guard_packaging_transitions(") + moves_at = source.index("_guard_catalog_moves(") + adopt_at = source.index("_adopt_existing(") + plan_at = source.index("runner.tofu_plan(") + assert packaging_at < moves_at < adopt_at < plan_at diff --git a/tests/iac/test_iac_confluent.py b/tests/iac/test_iac_confluent.py index 73377927..083044ff 100644 --- a/tests/iac/test_iac_confluent.py +++ b/tests/iac/test_iac_confluent.py @@ -237,6 +237,46 @@ def test_validator_warns_on_missing_glue_database(): assert any("Glue database" in w for w in warnings) +# ── catalog kind: Tableflow publishes to Glue, and only Glue ──────────────── + + +@pytest.mark.parametrize("catalog", ["lakekeeper", "polaris", "unity", "rest", "horizon"]) +def test_validator_rejects_a_non_glue_catalog(catalog): + """A non-Glue ``catalog`` used to be ignored, so ``catalog: lakekeeper`` + published the table to Glue: a catalog the contract never named.""" + errors, warnings = validate_confluent_binding(_contract(_loc(catalog=catalog))) + assert len(errors) == 1 + assert f"binding.location.catalog is '{catalog}'" in errors[0] + assert "publishes only to AWS Glue" in errors[0] + # The Glue-database warning would only restate the error. + assert not any("Glue database" in w for w in warnings) + + +@pytest.mark.parametrize("catalog", ["lakekeeper", "polaris", "Unity"]) +def test_emitter_skips_the_catalog_integration_the_validator_refuses(catalog): + """Emit-when-derivable pairing: the Glue integration is the one resource a + non-Glue contract must not get; storage and the topic still derive.""" + res = _emit(_contract(_loc(catalog=catalog))) + assert "confluent_catalog_integration" not in res + assert set(res) == {"confluent_provider_integration", "confluent_tableflow_topic"} + + +@pytest.mark.parametrize("catalog", ["glue", "GLUE", None]) +def test_glue_or_absent_catalog_is_unchanged(catalog): + loc = _loc() if catalog is None else _loc(catalog=catalog) + assert validate_confluent_binding(_contract(loc)) == ([], []) + res = _emit(_contract(loc)) + glue = next(iter(res["confluent_catalog_integration"].values()))["aws_glue"] + assert glue["custom_database"] == "analytics" + + +def test_absent_catalog_module_is_byte_identical_to_explicit_glue(): + plugin = get_iac_plugin("confluent") + assert build_module(plugin, _contract()) == build_module( + plugin, _contract(_loc(catalog="glue")) + ) + + # ── schema: 0.7.5 accepts the confluent platform + location keys ──────────── diff --git a/tests/iac/test_iac_iceberg_validation.py b/tests/iac/test_iac_iceberg_validation.py index 10378f07..7434e8f5 100644 --- a/tests/iac/test_iac_iceberg_validation.py +++ b/tests/iac/test_iac_iceberg_validation.py @@ -22,6 +22,10 @@ (otherwise the gate blocks something that would have worked), and * every contract the validator accepts must genuinely emit one (otherwise the gate waves through a silent no-op, which is the bug it exists for). + +The one sanctioned third outcome is a WARNING: a catalog external to +Snowflake whose integration needs a secret, so nothing is emitted on purpose +and the gate says so instead of failing the contract. """ from __future__ import annotations @@ -105,13 +109,59 @@ def test_complete_glue_binding_is_clean(self): ) assert not errors and not warnings - @pytest.mark.parametrize("catalog", ["polaris", "unity", "rest", "nessie"]) + @pytest.mark.parametrize( + "catalog", + [ + "polaris", + "unity", + "rest", + "nessie", + "lakekeeper", + "bigquery", + # Spellings fold onto the same row. + "iceberg_rest", + "ICEBERG-REST", + "LakeKeeper", + ], + ) def test_deferred_catalogs_warn_rather_than_error(self, catalog): """Understood but not emitted, because their auth is secret-bearing.""" errors, warnings = validate_iceberg_bindings(_contract("snowflake", catalog=catalog)) assert not errors assert warnings and "credential-free" in warnings[0] + def test_lakekeeper_warehouse_name_is_not_a_false_error(self): + """A Lakekeeper warehouse is a catalog NAME. The gate used to send it + down the Snowflake-managed checks and demand an s3:// or gs:// + warehouse for a table Lakekeeper owns, while the emitter built an + EXTERNAL VOLUME nothing would ever write to.""" + contract = _contract("snowflake", catalog="lakekeeper", warehouse="demo") + errors, warnings = validate_iceberg_bindings(contract) + assert errors == [] + assert len(warnings) == 1 and "'lakekeeper'" in warnings[0] + assert not _emits_prereq(contract, "snowflake") + + @pytest.mark.parametrize("catalog", ["hive", "jdbc", "hadoop", "dynamodb"]) + def test_catalog_snowflake_cannot_integrate_is_an_error(self, catalog): + """No CATALOG_SOURCE exists for these, so there is nothing to defer. + Storage that would satisfy the managed path must not rescue it.""" + contract = _contract( + "snowflake", + catalog=catalog, + warehouse="s3://lake/p", + iam_role_arn="arn:aws:iam::1:role/r", + ) + errors, warnings = validate_iceberg_bindings(contract) + assert warnings == [] + assert len(errors) == 1 + assert f"no catalog integration for a {catalog} catalog" in errors[0] + assert not _emits_prereq(contract, "snowflake") + + def test_snowflake_alias_takes_the_managed_checks(self): + """``catalog: snowflake`` is Snowflake-managed, so it needs storage.""" + errors, _ = validate_iceberg_bindings(_contract("snowflake", catalog="snowflake")) + assert errors and "Snowflake-managed Iceberg table" in errors[0] + def test_explicit_volume_override_is_clean(self): contract = _contract("snowflake") contract["exposes"][0]["binding"]["icebergConfig"] = { @@ -138,42 +188,177 @@ def test_gs_warehouse_is_clean(self): errors, _ = validate_iceberg_bindings(_contract("gcp", warehouse="gs://lake/p")) assert not errors + @pytest.mark.parametrize("catalog", ["lakekeeper", "rest", "polaris", "nessie"]) + def test_external_catalog_name_warehouse_is_clean(self, catalog): + """An external catalog owns its storage and names a warehouse + (``demo``). The BigLake "backed by GCS" error used to fire for it.""" + errors, warnings = validate_iceberg_bindings( + _contract("gcp", catalog=catalog, warehouse="demo") + ) + assert errors == [] and warnings == [] + + @pytest.mark.parametrize("warehouse", ["s3://aws-bucket/p", "abfss://c@acct.dfs/p"]) + def test_external_catalog_on_foreign_storage_is_one_error(self, warehouse): + errors, _ = validate_iceberg_bindings( + _contract("gcp", catalog="lakekeeper", warehouse=warehouse) + ) + assert len(errors) == 1 + assert "not Google Cloud Storage" in errors[0] + assert "lakekeeper catalog" in errors[0] + assert "BigQuery" not in errors[0] + + def test_external_catalog_on_gcs_still_gets_its_bucket(self): + contract = _contract("gcp", catalog="lakekeeper", warehouse="gs://lake/p") + assert validate_iceberg_bindings(contract) == ([], []) + assert _emits_prereq(contract, "gcp") + + @pytest.mark.parametrize("location", [{}, {"catalog": "bigquery"}, {"catalog": "BigQuery"}]) + def test_biglake_messages_are_unchanged(self, location): + """Absent and ``bigquery`` keep the BigLake checks byte for byte.""" + errors, _ = validate_iceberg_bindings( + _contract("gcp", warehouse="s3://aws-bucket/p", **location) + ) + assert errors == [ + "expose 'events': binding.location.warehouse is 's3://aws-bucket/p', but a " + "BigQuery Iceberg table is backed by GCS. Use a gs:// warehouse or " + "binding.location.bucket so FLUID can create the bucket dbt's " + "catalogs.yml points at." + ] + + +class TestCatalogKindGate: + """An unknown kind is refused once, on every platform. + + Every emitter keeps a fallback for a value it does not know and the + fallbacks disagree (REST for the streaming sink, Snowflake-managed for + dbt and the Snowflake IaC), so the gate refuses the value itself. + """ + + @pytest.mark.parametrize("platform", ["snowflake", "gcp", "aws", "local"]) + def test_unknown_kind_is_one_error_listing_the_known_ones(self, platform): + errors, warnings = validate_iceberg_bindings(_contract(platform, catalog="lakekeper")) + assert warnings == [] + assert len(errors) == 1, errors + assert "'lakekeper' is not a catalog kind" in errors[0] + for known in ("lakekeeper", "glue", "rest", "snowflake-managed"): + assert known in errors[0] + + def test_unknown_kind_on_snowflake_skips_the_managed_noise(self): + """A mistyped external catalog needs no EXTERNAL VOLUME storage, so + the managed "needs a warehouse" error would read as a second problem.""" + errors, _ = validate_iceberg_bindings(_contract("snowflake", catalog="horizon")) + assert len(errors) == 1 and "not a catalog kind" in errors[0] + + def test_unknown_kind_on_gcp_skips_the_storage_noise(self): + errors, _ = validate_iceberg_bindings( + _contract("gcp", catalog="horizon", warehouse="s3://aws/p") + ) + assert len(errors) == 1 and "not a catalog kind" in errors[0] + + @pytest.mark.parametrize( + "catalog", ["glue", "rest", "iceberg_rest", "lakekeeper", "snowflake", "Snowflake_Managed"] + ) + def test_known_kinds_and_aliases_pass(self, catalog): + errors, _ = validate_iceberg_bindings(_contract("aws", catalog=catalog, bucket="lake")) + assert errors == [] + + def test_the_sinks_format_spelling_is_checked(self): + """``iceberg-table`` is Iceberg to the streaming sink, which reads the + catalog, so the gate reads it too.""" + contract = _contract("aws", catalog="horizon") + contract["exposes"][0]["binding"]["format"] = "iceberg-table" + errors, _ = validate_iceberg_bindings(contract) + assert errors and "not a catalog kind" in errors[0] + + def test_non_iceberg_expose_is_not_checked(self): + contract = _contract("aws", catalog="horizon") + contract["exposes"][0]["binding"]["format"] = "parquet" + assert validate_iceberg_bindings(contract) == ([], []) + + def test_confluent_is_left_to_its_own_gate(self): + """validate_confluent_binding refuses every non-Glue catalog; listing + the REST kinds here would suggest values that gate rejects too.""" + from fluid_build.iac.providers.confluent import validate_confluent_binding + + contract = _contract("confluent", catalog="horizon") + assert validate_iceberg_bindings(contract) == ([], []) + cf_errors, _ = validate_confluent_binding(contract) + assert any("publishes only to AWS Glue" in e for e in cf_errors) + class TestGateMatchesEmitterBothWays: - """The pairing invariant, asserted in both directions.""" + """The pairing invariant, asserted in both directions. + + Restated for the kind table: every KNOWN catalog kind lands in exactly + one outcome. An error means nothing was emitted and the user can fix it; + a warning means nothing was emitted on purpose (a catalog external to + Snowflake whose integration needs a secret); clean means the + prerequisite was emitted. Before the table, ``catalog: lakekeeper`` with + a name warehouse was BOTH rejected by the gate and given a volume by the + emitter once storage was added, and ``catalog: hive`` was waved through + to a volume Snowflake could never use. An unknown kind is refused + outright (see TestCatalogKindGate) and is not part of this pairing. + """ + + _ROLE = "arn:aws:iam::1:role/r" SNOWFLAKE_CASES = [ {}, {"warehouse": "s3://lake/p"}, - {"warehouse": "s3://lake/p", "iam_role_arn": "arn:aws:iam::1:role/r"}, + {"warehouse": "s3://lake/p", "iam_role_arn": _ROLE}, {"warehouse": "gs://lake/p"}, # F1: a gs:// warehouse ALONGSIDE a bucket. The emitter resolves # scheme-first so this is a GCS volume needing no role; the gate # used to OR the two and demand one. {"warehouse": "gs://lake/p", "bucket": "lake"}, {"warehouse": "s3://lake/p", "bucket": "other"}, - {"bucket": "lake", "iam_role_arn": "arn:aws:iam::1:role/r"}, + {"bucket": "lake", "iam_role_arn": _ROLE}, {"catalog": "glue"}, - {"catalog": "glue", "account": "1", "iam_role_arn": "arn:aws:iam::1:role/r"}, + {"catalog": "glue", "account": "1", "iam_role_arn": _ROLE}, + {"catalog": "GLUE", "account": "1", "iam_role_arn": _ROLE}, + # Deferred external kinds: warned, never emitted, storage or not. + {"catalog": "lakekeeper", "warehouse": "demo"}, + {"catalog": "lakekeeper", "warehouse": "s3://lake/p", "iam_role_arn": _ROLE}, + {"catalog": "iceberg_rest", "warehouse": "s3://lake/p", "iam_role_arn": _ROLE}, + {"catalog": "polaris"}, + {"catalog": "bigquery", "warehouse": "gs://lake/p"}, + # No Snowflake integration exists: refused, never emitted. + {"catalog": "hive", "warehouse": "s3://lake/p", "iam_role_arn": _ROLE}, + {"catalog": "dynamodb"}, + # ``snowflake`` is the Snowflake-managed alias: the volume path. + {"catalog": "snowflake"}, + {"catalog": "snowflake", "warehouse": "s3://lake/p", "iam_role_arn": _ROLE}, ] - GCP_CASES = [ + #: Absent and ``bigquery``: the BigLake path, error iff no bucket. + GCP_BIGLAKE_CASES = [ {}, {"bucket": "lake"}, {"warehouse": "gs://lake/p"}, {"warehouse": "s3://aws/p"}, + {"catalog": "bigquery"}, + {"catalog": "bigquery", "warehouse": "gs://lake/p"}, + ] + + #: An external catalog owns its storage: ``(location, error, bucket)``. + GCP_EXTERNAL_CASES = [ + ({"catalog": "lakekeeper", "warehouse": "demo"}, False, False), + ({"catalog": "lakekeeper"}, False, False), + ({"catalog": "lakekeeper", "warehouse": "gs://lake/p"}, False, True), + ({"catalog": "rest", "bucket": "lake"}, False, True), + ({"catalog": "lakekeeper", "warehouse": "s3://aws/p"}, True, False), + ({"catalog": "nessie", "warehouse": "abfss://c@a.dfs/p"}, True, False), ] @pytest.mark.parametrize("location", SNOWFLAKE_CASES) - def test_snowflake_error_iff_no_prereq_emitted(self, location): + def test_snowflake_exactly_one_outcome(self, location): contract = _contract("snowflake", **location) - errors, _ = validate_iceberg_bindings(contract) + errors, warnings = validate_iceberg_bindings(contract) emitted = _emits_prereq(contract, "snowflake") - assert bool(errors) != emitted, ( - f"gate and emitter disagree for {location}: " f"errors={bool(errors)} emitted={emitted}" - ) + outcomes = {"error": bool(errors), "warning": bool(warnings), "emitted": emitted} + assert sum(outcomes.values()) == 1, f"gate and emitter disagree for {location}: {outcomes}" - @pytest.mark.parametrize("location", GCP_CASES) + @pytest.mark.parametrize("location", GCP_BIGLAKE_CASES) def test_gcp_error_iff_no_bucket_emitted(self, location): contract = _contract("gcp", **location) errors, _ = validate_iceberg_bindings(contract) @@ -182,6 +367,26 @@ def test_gcp_error_iff_no_bucket_emitted(self, location): f"gate and emitter disagree for {location}: " f"errors={bool(errors)} emitted={emitted}" ) + @pytest.mark.parametrize("location,error,bucket", GCP_EXTERNAL_CASES) + def test_gcp_external_catalog_never_errors_and_emits(self, location, error, bucket): + contract = _contract("gcp", **location) + errors, _ = validate_iceberg_bindings(contract) + emitted = _emits_prereq(contract, "gcp") + assert (bool(errors), emitted) == (error, bucket), location + assert not (errors and emitted) + + def test_unknown_kind_is_refused_while_the_emitter_keeps_its_fallback(self): + """The one case outside the pairing, by design. The Snowflake IaC + keeps dbt's ``built_in`` fallback (so dbt never references a volume + that does not exist), and the gate refuses the value, so a validated + contract never reaches that fallback.""" + contract = _contract( + "snowflake", catalog="horizon", warehouse="s3://lake/p", iam_role_arn=self._ROLE + ) + errors, _ = validate_iceberg_bindings(contract) + assert errors and "not a catalog kind" in errors[0] + assert _emits_prereq(contract, "snowflake") + class TestScope: def test_non_iceberg_exposes_are_ignored(self): @@ -267,6 +472,53 @@ def test_colliding_volumes_are_caught_at_validate(self): errors, _ = validate_iceberg_bindings(contract) assert errors and "different storage" in errors[0] + @staticmethod + def _two_exposes(catalog: str) -> Dict[str, Any]: + """Two exposes, one product id, so one derived volume name, on two buckets.""" + contract = _contract( + "snowflake", + catalog=catalog, + warehouse="s3://lake-a/p", + iam_role_arn="arn:aws:iam::1:role/r", + ) + second = { + "exposeId": "events2", + "kind": "table", + "binding": { + "platform": "snowflake", + "format": "iceberg", + "location": { + "database": "DB", + "schema": "PUBLIC", + "table": "T2", + "catalog": catalog, + "warehouse": "s3://lake-b/p", + "iam_role_arn": "arn:aws:iam::1:role/r", + }, + }, + "contract": {"schema": [{"name": "id", "type": "string"}]}, + } + contract["exposes"].append(second) + return contract + + def test_snowflake_alias_collision_is_caught_at_validate(self): + """The collision check skipped ANY non-empty catalog, but the emitter + builds volumes for ``catalog: snowflake``: validate passed and apply + raised mid-emit. Both sides now read the same ``built_in`` row.""" + contract = self._two_exposes("snowflake") + errors, _ = validate_iceberg_bindings(contract) + assert any("different storage" in e for e in errors) + with pytest.raises(ValueError, match="different storage locations"): + get_iac_plugin("snowflake").emit(contract) + + def test_external_catalog_exposes_never_collide(self): + """No volume is built for a Lakekeeper table, so there is nothing to + collide: the gate must not invent the error the emitter cannot hit.""" + contract = self._two_exposes("lakekeeper") + errors, _ = validate_iceberg_bindings(contract) + assert not any("different storage" in e for e in errors) + assert "snowflake_external_volume" not in get_iac_plugin("snowflake").emit(contract) + def test_same_volume_same_storage_is_fine(self): contract = _contract( "snowflake", warehouse="s3://lake/p", iam_role_arn="arn:aws:iam::1:role/r" diff --git a/tests/iac/test_iac_shadow.py b/tests/iac/test_iac_shadow.py index c5cad894..deaa4a6e 100644 --- a/tests/iac/test_iac_shadow.py +++ b/tests/iac/test_iac_shadow.py @@ -27,6 +27,7 @@ native_logical_resources, opentofu_logical_resources, ) +from fluid_build.providers.aws.plan.planner import plan_actions pytestmark = [pytest.mark.unit, pytest.mark.provider] @@ -127,3 +128,65 @@ def test_summary_names_provider_and_verdict(self): ) assert "aws" in report.summary() assert "cutover-safe" in report.summary() + + +def _iceberg_contract(**catalog): + return { + "id": "analytics.lake", + "name": "Lake", + "exposes": [ + { + "exposeId": "orders", + "kind": "table", + "binding": { + "platform": "aws", + "format": "iceberg", + "location": { + "database": "sales", + "table": "orders", + "bucket": "lake", + **catalog, + }, + }, + "contract": {"schema": [{"name": "order_id", "type": "string"}]}, + } + ], + } + + +def _glue_resources(resources): + return {r for r in resources if r.kind in ("database", "table")} + + +class TestIcebergCatalogParity: + """Both engines agree on WHO owns an Iceberg table: Glue, or its catalog. + + ``catalog: lakekeeper`` used to stream over REST while both engines + provisioned a Glue table for the same name. Each engine now reads the + shared ``is_glue_cataloged`` classification; this pins that they read it + the same way, from the real native planner output (not a hand-written + action list), so one engine cannot drift back to Glue alone. + """ + + def test_absent_catalog_both_engines_plan_the_glue_table(self): + contract = _iceberg_contract() + native = native_logical_resources(plan_actions(contract, "123456789012", "us-east-1")) + tf = opentofu_logical_resources(contract, get_iac_plugin("aws")) + expected = {LogicalResource("database", "sales"), LogicalResource("table", "orders")} + assert _glue_resources(native) == _glue_resources(tf) == expected + + @pytest.mark.parametrize("catalog", ["lakekeeper", "LakeKeeper", "iceberg_rest", "polaris"]) + def test_non_glue_catalog_neither_engine_plans_glue(self, catalog): + contract = _iceberg_contract( + catalog=catalog, uri="http://lakekeeper:8181/catalog", warehouse="demo" + ) + native_actions = plan_actions(contract, "123456789012", "us-east-1") + report = shadow_compare( + contract, plugin=get_iac_plugin("aws"), native_actions=native_actions + ) + native = native_logical_resources(native_actions) + tf = opentofu_logical_resources(contract, get_iac_plugin("aws"), native_actions) + assert _glue_resources(native) == set() + assert _glue_resources(tf) == set(), f"OpenTofu still plans Glue for {catalog!r}" + # The declared bucket is the table's storage: both engines keep it. + assert LogicalResource("bucket", "lake") in report.matched diff --git a/tests/iac/test_iac_snowflake_iceberg_prereqs.py b/tests/iac/test_iac_snowflake_iceberg_prereqs.py index d82750b2..2bd8c3c7 100644 --- a/tests/iac/test_iac_snowflake_iceberg_prereqs.py +++ b/tests/iac/test_iac_snowflake_iceberg_prereqs.py @@ -183,6 +183,22 @@ def test_glue_catalog_emits_no_external_volume(self): ) assert "snowflake_external_volume" not in res + def test_glue_spelling_folds_onto_the_glue_row(self): + res = _sf().emit( + _contract( + [ + _iceberg_exposure( + catalog="GLUE", + account="123456789012", + iam_role_arn="arn:aws:iam::123456789012:role/r", + warehouse="s3://lake/p/", + ) + ] + ) + ) + assert "snowflake_catalog_integration_aws_glue" in res + assert "snowflake_external_volume" not in res + def test_missing_role_or_account_emits_nothing(self): for location in ( {"catalog": "glue", "account": "123456789012"}, @@ -228,14 +244,24 @@ def test_explicit_override_emits_no_create(self): res = _sf().emit(_contract([exposure])) assert "snowflake_external_volume" not in res - def test_unlisted_catalog_value_is_managed_to_both_emitters(self): - """Predicate alignment: `catalog: snowflake` is not in the shared - external set, so dbt emits built_in AND the IaC emits the volume. - Before the fix the IaC skipped on any truthy catalog value.""" + @pytest.mark.parametrize( + "catalog", + [ + # The kind table's alias for snowflake-managed. + "snowflake", + # A value the table does not know: both emitters keep the + # built_in fallback, and ``fluid validate`` refuses the value. + "horizon", + ], + ) + def test_managed_catalog_values_are_managed_to_both_emitters(self, catalog): + """Predicate alignment: dbt emits built_in AND the IaC emits the + volume dbt references. Before the first fix the IaC skipped on any + truthy catalog value, so dbt referenced a volume nobody created.""" from fluid_build.engines.dbt.catalogs_yml import generate_catalogs_yml loc = dict( - catalog="snowflake", + catalog=catalog, warehouse="s3://lake/p/", iam_role_arn="arn:aws:iam::123456789012:role/r", ) @@ -249,6 +275,30 @@ def test_unlisted_catalog_value_is_managed_to_both_emitters(self): volume = next(iter(res["snowflake_external_volume"].values())) assert volume["name"] in content + def test_lakekeeper_is_external_to_both_emitters(self): + """The bug the kind table exists for. ``catalog: lakekeeper`` fell + outside the hand-kept external set, so the IaC built an EXTERNAL + VOLUME and dbt wrote a Snowflake-managed table into it, while the + streaming sink wrote the same table to Lakekeeper over REST.""" + from fluid_build.engines.dbt.catalogs_yml import generate_catalogs_yml + + contract = _contract( + [ + _iceberg_exposure( + catalog="lakekeeper", + warehouse="s3://lake/p/", + iam_role_arn="arn:aws:iam::123456789012:role/r", + ) + ] + ) + res = _sf().emit(contract) + assert "snowflake_external_volume" not in res + + build = {"engine": "dbt", "execution": {"runtime": {"platform": "snowflake"}}} + content = generate_catalogs_yml(contract, build) + assert "catalog_type: iceberg_rest" in content + assert "external_volume" not in content + class TestScopeBoundaries: def test_non_iceberg_exposes_emit_no_prereqs(self): @@ -260,12 +310,59 @@ def test_non_iceberg_exposes_emit_no_prereqs(self): assert "snowflake_external_volume" not in res assert "snowflake_catalog_integration_aws_glue" not in res - @pytest.mark.parametrize("catalog", ["rest", "polaris", "unity", "nessie"]) + @pytest.mark.parametrize( + "catalog", + [ + "rest", + "polaris", + "unity", + "nessie", + "lakekeeper", + "bigquery", + "iceberg_rest", + "ICEBERG-REST", + ], + ) def test_secret_bearing_catalogs_are_a_documented_follow_up(self, catalog): """Their integrations authenticate with OAuth secrets or bearer - tokens, and the emitted .tf.json is credential-free by invariant.""" - res = _sf().emit(_contract([_iceberg_exposure(catalog=catalog)])) + tokens, and the emitted .tf.json is credential-free by invariant. + + Storage that would satisfy the managed path is present on purpose: + a kind misclassified as Snowflake-managed would get a volume here, + which is how ``lakekeeper`` used to fail.""" + res = _sf().emit( + _contract( + [ + _iceberg_exposure( + catalog=catalog, + warehouse="s3://lake/p/", + iam_role_arn="arn:aws:iam::123456789012:role/r", + ) + ] + ) + ) assert "snowflake_catalog_integration_iceberg_rest" not in res + assert "snowflake_external_volume" not in res + assert "snowflake_catalog_integration_aws_glue" not in res + + @pytest.mark.parametrize("catalog", ["hive", "jdbc", "hadoop", "dynamodb"]) + def test_catalogs_snowflake_cannot_integrate_emit_nothing(self, catalog): + """No Snowflake CATALOG_SOURCE exists for these. They used to fall + through to the managed path and get a volume no table could use; + ``fluid validate`` now refuses them instead.""" + res = _sf().emit( + _contract( + [ + _iceberg_exposure( + catalog=catalog, + warehouse="s3://lake/p/", + iam_role_arn="arn:aws:iam::123456789012:role/r", + ) + ] + ) + ) + assert "snowflake_external_volume" not in res + assert "snowflake_catalog_integration_aws_glue" not in res def test_emitted_module_stays_credential_free(self): import json diff --git a/tests/integration/test_iceberg_sink_gcs_live.py b/tests/integration/test_iceberg_sink_gcs_live.py index 14d259c1..08d055ed 100644 --- a/tests/integration/test_iceberg_sink_gcs_live.py +++ b/tests/integration/test_iceberg_sink_gcs_live.py @@ -178,8 +178,10 @@ def test_iceberg_sink_writes_to_gcs(tmp_path: Path): if not wait_for_http("http://localhost:8181/v1/config"): pytest.skip("iceberg-rest catalog did not come up within timeout") - # create the namespace over REST before the sink writes (RFC §14: the - # connector auto-creates the TABLE but not the NAMESPACE). + # create the namespace over REST before writing: the writer here is + # pyiceberg (no Kafka Connect sink runs in this test), and its + # create_table needs an existing namespace. The Apache sink creates the + # namespace itself on auto-create since Iceberg 1.6.0 (RFC §6.8 #6). urllib.request.urlopen( # noqa: S310 — localhost emulator urllib.request.Request( "http://localhost:8181/v1/namespaces", diff --git a/tests/integration/test_lakekeeper_kafka_connect_live.py b/tests/integration/test_lakekeeper_kafka_connect_live.py new file mode 100644 index 00000000..75a94757 --- /dev/null +++ b/tests/integration/test_lakekeeper_kafka_connect_live.py @@ -0,0 +1,825 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""LIVE proof: forge-derived Iceberg sink configs on a real Kafka Connect worker. + +The Kafka Connect Iceberg-sink deriver once emitted BOTH ``iceberg.catalog.type`` +and ``iceberg.catalog.catalog-impl``. Apache Iceberg's ``CatalogUtil`` refuses +that ("both type and catalog-impl are set"), so every Glue sink crashed at +startup, and no unit test could see it: the refusal lives in the worker's +runtime. This module runs forge's REAL derivation chain +(``find_iceberg_expose_binding`` -> ``resolve_iceberg_catalog`` -> +``emit_iceberg_sink_config``) and hands the result to a real worker: + +* ``location.catalog: Lakekeeper`` streams records into a real Lakekeeper REST + catalog: ``type=rest``, the warehouse addressed by NAME, no FileIO and no + storage keys, because Lakekeeper vends the table's storage credentials (STS + against Silo, an S3-compatible store). The Apache sink creates the namespace + itself (apache/iceberg#10186). +* The same worker refuses a config carrying both selector keys, which is the + failure every Glue sink used to hit. +* The forge-derived Glue config (``catalog-impl`` only) gets past that gate on + the same worker. It never writes: the worker has no AWS account, and no record + is produced for it. + +Stack (every image pinned by digest, all multi-arch): Postgres + Lakekeeper, +Silo for S3 + STS (``minio/minio`` and ``minio/mc`` left Docker Hub on +2026-09-11), a single-node KRaft Kafka, and cp-kafka-connect with the ASF-owned +Apache Iceberg sink ZIP, which a one-shot container downloads and verifies by +sha1 (no confluent-hub CLI: that is under the Confluent Enterprise License). + +Gating: ``integration`` + ``emulated_heavy``; self-skips unless +``FLUID_TEST_LAKEKEEPER=1`` and Docker is reachable. Docker and heavy emulator +tests run only in the CI integration stage (the ``lakekeeper-integration`` job +of ``.github/workflows/integration-emulated-heavy.yml``), never in the light +suite. Locally: + + FLUID_TEST_LAKEKEEPER=1 \\ + FLUID_LK_PLUGIN_CACHE=~/.cache/fluid/iceberg-kafka-connect \\ + python -m pytest -v -m emulated_heavy \\ + tests/integration/test_lakekeeper_kafka_connect_live.py + +Optional environment: + +* ``FLUID_LK_PLUGIN_CACHE``: host directory for the unzipped sink plugin, kept + across runs (CI caches it). Unset, a compose volume holds it and teardown + removes it, so every run downloads it again. +* ``FLUID_LK_PROJECT``: compose project name (default ``fluid-lk-``), so + CI can collect logs from, and tear down, a run the test could not clean up. +* ``FLUID_LK_LOG_DIR``: on failure, the compose logs are also written here. +""" + +from __future__ import annotations + +import contextlib +import json +import os +import socket +import subprocess +import time +import urllib.error +import urllib.request +import uuid +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Dict, Iterator, List, Mapping, Optional, Tuple + +import pytest + +from fluid_build.build_runners.kafka_connect.iceberg_sink import emit_iceberg_sink_config +from fluid_build.build_runners.kafka_connect.iceberg_sink_validation import ( + iceberg_sink_preflight, +) +from fluid_build.providers._iceberg_catalog import ( + GLUE_CATALOG_IMPL, + find_iceberg_expose_binding, + resolve_iceberg_catalog, +) +from tests.integration._iceberg_objectstore_live import _docker_available + +pytestmark = [ + pytest.mark.integration, + pytest.mark.emulated_heavy, + pytest.mark.slow, + # The first test also pays for the module stack: image pulls, the plugin + # download and the worker's warm-up commit rounds. The 600 s default from + # pyproject.toml covers setup + call + teardown, which is not enough. + pytest.mark.timeout(1800), + pytest.mark.skipif( + os.environ.get("FLUID_TEST_LAKEKEEPER") != "1" or not _docker_available(), + reason=( + "set FLUID_TEST_LAKEKEEPER=1 and have Docker running to run the " + "Lakekeeper + Kafka Connect live test" + ), + ), +] + +# Throwaway credentials for containers that live only as long as this module. +_PG_PASSWORD = "postgres" # pragma: allowlist secret +_LK_ENCRYPTION_KEY = "fluid-lakekeeper-live-test-only" # pragma: allowlist secret +_S3_USER = "fluid-silo" +_S3_PASSWORD = "fluid-silo-live-test-only" # pragma: allowlist secret + +_BUCKET = "forge" +_WAREHOUSE = "forge" +_S3_REGION = "local-01" +_NAMESPACE = "streaming" +_RECORDS = 5 + +#: Connect-side address of the catalog (the compose network, not the host). +_CATALOG_URI = "http://lakekeeper:8181/catalog" +_KAFKA_BIN = "/opt/kafka/bin" +_BOTH_KEYS_ERROR = "both type and catalog-impl are set" + +_COMPOSE = """\ +services: + db: + image: postgres:17.11@sha256:d74eeac9a635390a49bc21bd49fccd973de707e2a53a76ac49b552b8712ec46f + environment: + POSTGRES_PASSWORD: ${LK_PG_PASSWORD} + healthcheck: + test: ["CMD-SHELL", "pg_isready -U postgres -d postgres"] + interval: 2s + timeout: 5s + retries: 60 + + migrate: + image: quay.io/lakekeeper/catalog:v0.13.6@sha256:d6829722cac0d00dfc5665b0955766387b93ccadd8b1e70e679b49619bc30ea5 + command: ["migrate"] + restart: "no" + environment: &lakekeeper-env + LAKEKEEPER__PG_DATABASE_URL_READ: postgresql://postgres:${LK_PG_PASSWORD}@db:5432/postgres + LAKEKEEPER__PG_DATABASE_URL_WRITE: postgresql://postgres:${LK_PG_PASSWORD}@db:5432/postgres + LAKEKEEPER__PG_ENCRYPTION_KEY: ${LK_ENCRYPTION_KEY} + LAKEKEEPER__AUTHZ_BACKEND: allowall + depends_on: + db: + condition: service_healthy + + silo: + image: pgsty/silo:RELEASE.2026-09-03T13-18-01Z@sha256:b616a0cf8cb281e7e6bb3c9b1fb53875b4016a2878223925541c18f82d6c5ca3 + command: ["server", "/data"] + environment: + MINIO_ROOT_USER: ${LK_S3_USER} + MINIO_ROOT_PASSWORD: ${LK_S3_PASSWORD} + healthcheck: + test: ["CMD", "mc", "ready", "local"] + interval: 2s + timeout: 5s + retries: 60 + + create-bucket: + image: pgsty/silo:RELEASE.2026-09-03T13-18-01Z@sha256:b616a0cf8cb281e7e6bb3c9b1fb53875b4016a2878223925541c18f82d6c5ca3 + restart: "no" + entrypoint: ["/bin/sh", "-ec"] + command: + - mc alias set local http://silo:9000 "$$S3_USER" "$$S3_PASSWORD" && mc mb --ignore-existing local/forge + environment: + S3_USER: ${LK_S3_USER} + S3_PASSWORD: ${LK_S3_PASSWORD} + depends_on: + silo: + condition: service_healthy + + lakekeeper: + image: quay.io/lakekeeper/catalog:v0.13.6@sha256:d6829722cac0d00dfc5665b0955766387b93ccadd8b1e70e679b49619bc30ea5 + command: ["serve"] + environment: *lakekeeper-env + depends_on: + migrate: + condition: service_completed_successfully + create-bucket: + condition: service_completed_successfully + ports: + - "127.0.0.1:${LK_CATALOG_PORT}:8181" + healthcheck: + # Distroless image: no shell, so exec form only. + test: ["CMD", "/home/nonroot/lakekeeper", "healthcheck"] + interval: 2s + timeout: 5s + retries: 60 + + kafka: + image: apache/kafka:3.9.1@sha256:4ceccc577f03f51f6af8dbfda55194d0d892f4fa7913ffbded567ce3895622ed + environment: + KAFKA_NODE_ID: 1 + KAFKA_PROCESS_ROLES: broker,controller + KAFKA_LISTENERS: PLAINTEXT://:19092,CONTROLLER://:9093 + KAFKA_ADVERTISED_LISTENERS: PLAINTEXT://kafka:19092 + KAFKA_CONTROLLER_LISTENER_NAMES: CONTROLLER + KAFKA_LISTENER_SECURITY_PROTOCOL_MAP: CONTROLLER:PLAINTEXT,PLAINTEXT:PLAINTEXT + KAFKA_CONTROLLER_QUORUM_VOTERS: 1@localhost:9093 + KAFKA_OFFSETS_TOPIC_REPLICATION_FACTOR: 1 + # The sink's commit coordinator writes through a transactional producer; + # with the default RF of 3 a single broker can never commit. + KAFKA_TRANSACTION_STATE_LOG_REPLICATION_FACTOR: 1 + KAFKA_TRANSACTION_STATE_LOG_MIN_ISR: 1 + KAFKA_GROUP_INITIAL_REBALANCE_DELAY_MS: 0 + KAFKA_HEAP_OPTS: -Xmx512m + healthcheck: + test: ["CMD-SHELL", "/opt/kafka/bin/kafka-broker-api-versions.sh --bootstrap-server localhost:19092 >/dev/null 2>&1"] + interval: 5s + timeout: 20s + retries: 40 + start_period: 10s + + connect-plugins: + image: alpine:3.22@sha256:5291449c3df73caf6ed85e649dec1b9e818b39a5d8c871e97afc13e9cd5e8fa8 + restart: "no" + volumes: + - ${LK_PLUGIN_SOURCE}:/plugins + entrypoint: ["/bin/sh", "-euc"] + command: + - | + sha1=52dab2ff2b9659008deec803d1c1e92813c5ffa7 + if [ "$$(cat /plugins/.sha1 2>/dev/null || true)" = "$$sha1" ]; then + echo "iceberg sink plugin cached"; exit 0 + fi + wget -q -O /tmp/sink.zip https://hub-downloads.confluent.io/api/plugins/iceberg/iceberg-kafka-connect/versions/1.9.2/iceberg-iceberg-kafka-connect-1.9.2.zip + echo "$$sha1 /tmp/sink.zip" | sha1sum -c - + rm -rf /plugins/iceberg-iceberg-kafka-connect-* + unzip -q /tmp/sink.zip -d /plugins + rm -rf /plugins/__MACOSX + chmod -R a+rX /plugins + echo "$$sha1" > /plugins/.sha1 + echo "iceberg sink plugin installed" + + connect: + image: confluentinc/cp-kafka-connect:7.9.10@sha256:b174fd9317b1864a22b01da266407a802a56c5d1ab9ee77aef03d2df9854fdc9 + depends_on: + kafka: + condition: service_healthy + connect-plugins: + condition: service_completed_successfully + lakekeeper: + condition: service_healthy + ports: + - "127.0.0.1:${LK_CONNECT_PORT}:8083" + volumes: + - ${LK_PLUGIN_SOURCE}:/plugins:ro + environment: + CONNECT_BOOTSTRAP_SERVERS: kafka:19092 + CONNECT_REST_ADVERTISED_HOST_NAME: connect + CONNECT_GROUP_ID: fluid-lakekeeper-live + CONNECT_CONFIG_STORAGE_TOPIC: _connect-configs + CONNECT_OFFSET_STORAGE_TOPIC: _connect-offsets + CONNECT_STATUS_STORAGE_TOPIC: _connect-status + CONNECT_CONFIG_STORAGE_REPLICATION_FACTOR: 1 + CONNECT_OFFSET_STORAGE_REPLICATION_FACTOR: 1 + CONNECT_STATUS_STORAGE_REPLICATION_FACTOR: 1 + CONNECT_KEY_CONVERTER: org.apache.kafka.connect.storage.StringConverter + CONNECT_VALUE_CONVERTER: org.apache.kafka.connect.json.JsonConverter + CONNECT_VALUE_CONVERTER_SCHEMAS_ENABLE: "false" + CONNECT_PLUGIN_PATH: /usr/share/java,/plugins + CONNECT_CONFIG_PROVIDERS: env + CONNECT_CONFIG_PROVIDERS_ENV_CLASS: org.apache.kafka.common.config.provider.EnvVarConfigProvider + KAFKA_HEAP_OPTS: -Xms256m -Xmx768m + healthcheck: + test: ["CMD-SHELL", "curl -sf localhost:8083/connector-plugins | grep -q IcebergSinkConnector"] + interval: 5s + timeout: 10s + retries: 60 + start_period: 30s + +volumes: + plugins: {} +""" + + +# --------------------------------------------------------------------------- +# Plumbing +# --------------------------------------------------------------------------- + + +def _free_ports(count: int) -> List[int]: + """``count`` distinct free loopback ports (all held open while picking).""" + socks = [] + try: + for _ in range(count): + s = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + s.bind(("127.0.0.1", 0)) + socks.append(s) + return [s.getsockname()[1] for s in socks] + finally: + for s in socks: + s.close() + + +def _parse(body: str) -> Any: + """JSON from a Connect or Lakekeeper response. ``strict=False``: a Connect + task trace is a Java stack trace and may carry raw control characters.""" + return json.loads(body, strict=False) if body.strip() else {} + + +def _http( + method: str, + url: str, + body: Optional[Mapping[str, Any]] = None, + *, + headers: Optional[Mapping[str, str]] = None, + timeout: float = 30.0, +) -> Tuple[int, str]: + """One request against a loopback service; returns ``(status, body)``, + including for 4xx/5xx, so callers can assert on the error body.""" + data = json.dumps(body).encode() if body is not None else None + request = urllib.request.Request( + url, + data=data, + method=method, + headers={ + "Content-Type": "application/json", + "Accept": "application/json", + **(headers or {}), + }, + ) + try: + with urllib.request.urlopen(request, timeout=timeout) as resp: # noqa: S310 — loopback only + return resp.status, resp.read().decode("utf-8", "replace") + except urllib.error.HTTPError as exc: + return exc.code, exc.read().decode("utf-8", "replace") + + +@dataclass +class _Stack: + project: str + workdir: Path + env: Dict[str, str] + catalog_url: str + connect_url: str + prefix: str = "" + + def compose( + self, *args: str, stdin: Optional[str] = None, timeout: int = 120 + ) -> "subprocess.CompletedProcess[str]": + return subprocess.run( + ["docker", "compose", "-p", self.project, *args], + cwd=self.workdir, + env={**os.environ, **self.env}, + input=stdin, + capture_output=True, + text=True, + timeout=timeout, + ) + + def kafka(self, script: str, *args: str, stdin: Optional[str] = None) -> str: + """Run a Kafka CLI inside the broker container; no host Kafka client.""" + done = self.compose("exec", "-T", "kafka", f"{_KAFKA_BIN}/{script}", *args, stdin=stdin) + assert done.returncode == 0, f"{script} {args}: {done.stdout}\n{done.stderr}" + return done.stdout + + def worker_log(self) -> str: + return self.compose("logs", "--no-color", "connect", timeout=60).stdout + + def dump_logs(self, label: str) -> str: + logs = self.compose("logs", "--no-color", "--tail=200", timeout=60) + text = logs.stdout + logs.stderr + print(f"\n===== docker compose -p {self.project} logs ({label}) =====\n{text}") + log_dir = os.environ.get("FLUID_LK_LOG_DIR") + if log_dir: + out = Path(log_dir) + out.mkdir(parents=True, exist_ok=True) + (out / f"{self.project}-{label}.log").write_text(text) + return text + + @contextlib.contextmanager + def logs_on_failure(self, label: str) -> Iterator[None]: + try: + yield + except pytest.skip.Exception: + raise + except BaseException: + self.dump_logs(label) + raise + + +def _bootstrap_lakekeeper(stack: _Stack) -> str: + """Bootstrap Lakekeeper, create the ``forge`` warehouse on Silo, and return + the REST path prefix the catalog assigned it (never hard-coded).""" + base = stack.catalog_url + status, body = _http("GET", f"{base}/health") + assert status == 200, f"GET /health -> {status}: {body}" + + status, body = _http("POST", f"{base}/management/v1/bootstrap", {"accept-terms-of-use": True}) + assert status == 204 or ( + status == 400 and "CatalogAlreadyBootstrapped" in body + ), f"bootstrap -> {status}: {body}" + + warehouse = { + "warehouse-name": _WAREHOUSE, + "storage-profile": { + "type": "s3", + "bucket": _BUCKET, + "region": _S3_REGION, + "endpoint": "http://silo:9000", + "sts-endpoint": "http://silo:9000", + "path-style-access": True, + "flavor": "s3-compat", + "sts-enabled": True, + }, + "storage-credential": { + "type": "s3", + "credential-type": "access-key", + "aws-access-key-id": _S3_USER, + "aws-secret-access-key": _S3_PASSWORD, + }, + } + status, body = _http("POST", f"{base}/management/v1/warehouse", warehouse) + assert status in (200, 201, 409), f"create warehouse -> {status}: {body}" + + status, body = _http("GET", f"{base}/catalog/v1/config?warehouse={_WAREHOUSE}") + assert status == 200, f"GET /catalog/v1/config -> {status}: {body}" + config = _parse(body) + # The order an Iceberg REST client applies them: defaults, then overrides. + merged = {**(config.get("defaults") or {}), **(config.get("overrides") or {})} + prefix = merged.get("prefix") + assert prefix, f"Lakekeeper /v1/config carries no prefix: {config}" + return str(prefix) + + +@pytest.fixture(scope="module") +def lakekeeper_stack(tmp_path_factory: pytest.TempPathFactory) -> Iterator[_Stack]: + workdir = tmp_path_factory.mktemp("lakekeeper-live") + (workdir / "docker-compose.yml").write_text(_COMPOSE) + project = os.environ.get("FLUID_LK_PROJECT") or f"fluid-lk-{uuid.uuid4().hex[:8]}" + + cache = os.environ.get("FLUID_LK_PLUGIN_CACHE") + if cache: + cache_dir = Path(cache).expanduser().resolve() + # Created here, not by the Docker daemon, so the directory is ours. + cache_dir.mkdir(parents=True, exist_ok=True) + plugin_source = str(cache_dir) + else: + plugin_source = "plugins" # the compose volume; removed by `down -v` + + catalog_port, connect_port = _free_ports(2) + stack = _Stack( + project=project, + workdir=workdir, + env={ + "LK_PG_PASSWORD": _PG_PASSWORD, + "LK_ENCRYPTION_KEY": _LK_ENCRYPTION_KEY, + "LK_S3_USER": _S3_USER, + "LK_S3_PASSWORD": _S3_PASSWORD, + "LK_CATALOG_PORT": str(catalog_port), + "LK_CONNECT_PORT": str(connect_port), + "LK_PLUGIN_SOURCE": plugin_source, + }, + catalog_url=f"http://127.0.0.1:{catalog_port}", + connect_url=f"http://127.0.0.1:{connect_port}", + ) + try: + pull = stack.compose("pull", "--quiet", "--policy", "missing", timeout=600) + if pull.returncode != 0: + pytest.fail(f"docker compose pull failed:\n{pull.stderr[-3000:]}") + # Name only the long-running leaves: `--wait` on a one-shot fails the + # moment it exits, even with status 0. Their dependencies come up too. + up = stack.compose( + "up", "-d", "--wait", "--wait-timeout", "420", "lakekeeper", "connect", timeout=600 + ) + if up.returncode != 0: + stack.dump_logs("bring-up") + pytest.fail( + f"the Lakekeeper + Kafka Connect stack did not come up:\n{up.stderr[-3000:]}" + ) + with stack.logs_on_failure("bootstrap"): + stack.prefix = _bootstrap_lakekeeper(stack) + yield stack + finally: + stack.compose("down", "-v", "--remove-orphans", timeout=300) + + +# --------------------------------------------------------------------------- +# Contract + derivation (forge's real code only) +# --------------------------------------------------------------------------- + + +def _contract(product_id: str, location: Mapping[str, Any]) -> Dict[str, Any]: + """A contract with one Iceberg expose and one Kafka Connect build writing it.""" + return { + "fluidVersion": "0.7.6", + "kind": "DataProduct", + "id": product_id, + "name": product_id.rsplit(".", 1)[-1], + "metadata": {"layer": "Bronze", "owner": {"team": "data-platform"}}, + "exposes": [ + { + "exposeId": "orders", + "kind": "table", + "binding": {"platform": "aws", "format": "iceberg", "location": dict(location)}, + } + ], + "builds": [ + { + "id": "stream_orders", + "pattern": "acquisition", + "engine": "kafka-connect", + "outputs": ["orders"], + "properties": { + "sink": {"format": "iceberg"}, + "kafka-connect": { + "streamingSink": {"autoCreate": True, "commitIntervalMs": 5000} + }, + }, + } + ], + } + + +def _derive(contract: Mapping[str, Any], topic: str) -> Dict[str, str]: + """The sink config exactly as the Kafka Connect runner derives it.""" + assert iceberg_sink_preflight(contract, "stream_orders") is None + binding = find_iceberg_expose_binding(contract) + assert binding is not None, "the contract's Iceberg expose was not found" + resolved = resolve_iceberg_catalog(binding, contract=contract) + kc_props = contract["builds"][0]["properties"]["kafka-connect"] + return emit_iceberg_sink_config( + resolved, product_id=contract["id"], topics=[topic], kc_props=kc_props + ) + + +def _lakekeeper_location(table: str) -> Dict[str, Any]: + return { + "catalog": "Lakekeeper", # any spelling classifies the same way + "uri": _CATALOG_URI, + "warehouse": _WAREHOUSE, + "database": _NAMESPACE, + "table": table, + "region": _S3_REGION, + } + + +# --------------------------------------------------------------------------- +# Kafka Connect + catalog probes +# --------------------------------------------------------------------------- + + +def _put_connector(stack: _Stack, name: str, config: Mapping[str, str]) -> None: + status, body = _http("PUT", f"{stack.connect_url}/connectors/{name}/config", config) + assert status in (200, 201), f"PUT /connectors/{name}/config -> {status}: {body}" + + +def _delete_connector(stack: _Stack, name: str) -> None: + _http("DELETE", f"{stack.connect_url}/connectors/{name}") + + +def _status(stack: _Stack, name: str) -> Optional[Dict[str, Any]]: + status, body = _http("GET", f"{stack.connect_url}/connectors/{name}/status") + if status == 404: # Connect's status store lags a fresh PUT + return None + assert status == 200, f"GET /connectors/{name}/status -> {status}: {body}" + return _parse(body) + + +def _failed_task(status: Mapping[str, Any]) -> Optional[Mapping[str, Any]]: + for task in status.get("tasks") or []: + if task.get("state") == "FAILED": + return task + return None + + +def _wait_running(stack: _Stack, name: str, *, deadline_s: float = 180) -> None: + """Connector and every task RUNNING; a FAILED task fails with its trace.""" + deadline = time.monotonic() + deadline_s + last: Optional[Dict[str, Any]] = None + while time.monotonic() < deadline: + last = _status(stack, name) + if last: + failed = _failed_task(last) + if failed or last.get("connector", {}).get("state") == "FAILED": + trace = (failed or last.get("connector") or {}).get("trace", "") + pytest.fail(f"connector {name} FAILED:\n{trace}") + tasks = last.get("tasks") or [] + if ( + last.get("connector", {}).get("state") == "RUNNING" + and tasks + and all(t.get("state") == "RUNNING" for t in tasks) + ): + return + time.sleep(2) + pytest.fail(f"connector {name} not RUNNING after {deadline_s:.0f}s: {last}") + + +def _create_topic(stack: _Stack, topic: str) -> None: + stack.kafka( + "kafka-topics.sh", + "--bootstrap-server", + "localhost:19092", + "--create", + "--if-not-exists", + "--topic", + topic, + "--partitions", + "1", + "--replication-factor", + "1", + ) + + +def _table_path(stack: _Stack, table: str) -> str: + return f"{stack.catalog_url}/catalog/v1/{stack.prefix}/namespaces/{_NAMESPACE}/tables/{table}" + + +def _load_table(stack: _Stack, table: str) -> Optional[Dict[str, Any]]: + status, body = _http( + "GET", + _table_path(stack, table), + headers={"X-Iceberg-Access-Delegation": "client-managed"}, + ) + if status == 404: + return None + assert status == 200, f"loadTable {_NAMESPACE}.{table} -> {status}: {body}" + return _parse(body) + + +def _total_records(loaded: Mapping[str, Any]) -> int: + """``total-records`` of the CURRENT snapshot (0 before the first commit).""" + metadata = loaded.get("metadata") or {} + current = metadata.get("current-snapshot-id") + for snapshot in metadata.get("snapshots") or []: + if snapshot.get("snapshot-id") == current: + return int((snapshot.get("summary") or {}).get("total-records", 0)) + return 0 + + +# --------------------------------------------------------------------------- +# Tests +# --------------------------------------------------------------------------- + + +def test_derived_lakekeeper_config_streams_into_lakekeeper(request: pytest.FixtureRequest): + """forge's derived ``catalog: Lakekeeper`` config commits every record into a + real Lakekeeper through a real worker, with the namespace and table created + by the sink and the storage credentials vended by the catalog.""" + suffix = uuid.uuid4().hex[:8] + table = f"orders_{suffix}" + topic = f"orders-{suffix}" + connector = f"fluid-lk-sink-{suffix}" + config = _derive( + _contract(f"live.lakekeeper.orders_{suffix}", _lakekeeper_location(table)), topic + ) + + # Checked before any container starts: a deriver regression fails here in + # well under a second. + assert config["iceberg.catalog.type"] == "rest" + assert "iceberg.catalog.catalog-impl" not in config + assert "iceberg.catalog.io-impl" not in config # Lakekeeper vends the FileIO config + assert not [k for k in config if k.startswith("iceberg.catalog.s3.")] + assert config["iceberg.catalog.uri"] == _CATALOG_URI + assert config["iceberg.catalog.warehouse"] == _WAREHOUSE # a NAME, not s3:// + assert config["iceberg.tables"] == f"{_NAMESPACE}.{table}" + + stack: _Stack = request.getfixturevalue("lakekeeper_stack") + with stack.logs_on_failure(request.node.name): + try: + _create_topic(stack, topic) + _put_connector(stack, connector, config) + _wait_running(stack, connector) + + records = "".join( + json.dumps({"id": i, "customer": f"c{i}", "amount_cents": 100 * i}) + "\n" + for i in range(1, _RECORDS + 1) + ) + stack.kafka( + "kafka-console-producer.sh", + "--bootstrap-server", + "localhost:19092", + "--topic", + topic, + stdin=records, + ) + + # The worker's first commit rounds time out while it warms up (up + # to about 2.5 minutes observed), hence the generous deadline. + deadline = time.monotonic() + 360 + loaded: Optional[Dict[str, Any]] = None + total = -1 + while time.monotonic() < deadline: + status = _status(stack, connector) or {} + failed = _failed_task(status) + if failed: + pytest.fail(f"sink task FAILED while committing:\n{failed.get('trace', '')}") + loaded = _load_table(stack, table) + total = _total_records(loaded) if loaded else -1 + if total >= _RECORDS: + break + time.sleep(5) + assert loaded is not None, f"the sink never created {_NAMESPACE}.{table} in Lakekeeper" + assert total == _RECORDS, f"current snapshot holds {total} records, want {_RECORDS}" + + # Lakekeeper placed the table in the warehouse's bucket: the + # warehouse NAME resolved to its storage profile. + location = (loaded.get("metadata") or {}).get("location", "") + assert location.startswith(f"s3://{_BUCKET}/"), location + + # The sink created the namespace on its own (apache/iceberg#10186). + status_code, body = _http( + "GET", f"{stack.catalog_url}/catalog/v1/{stack.prefix}/namespaces/{_NAMESPACE}" + ) + assert status_code == 200, f"namespace {_NAMESPACE}: {status_code} {body}" + finally: + _delete_connector(stack, connector) + + +def test_both_type_and_catalog_impl_fail_on_a_real_worker( + lakekeeper_stack: _Stack, request: pytest.FixtureRequest +): + """The negative control for the bug class: the same config, plus the second + selector key the old deriver added, never starts on a real worker.""" + stack = lakekeeper_stack + suffix = uuid.uuid4().hex[:8] + table = f"orders_{suffix}" + connector = f"fluid-lk-both-{suffix}" + config = _derive( + _contract(f"live.lakekeeper.both_{suffix}", _lakekeeper_location(table)), + f"orders-{suffix}", + ) + config["iceberg.catalog.catalog-impl"] = "org.apache.iceberg.rest.RESTCatalog" + assert config["iceberg.catalog.type"] == "rest" + + with stack.logs_on_failure(request.node.name): + try: + # No topic is needed: sink 1.9.2 loads the catalog in + # IcebergSinkTask.start() (IcebergSinkTask.java:48), so the task + # dies before it consumes anything. + _put_connector(stack, connector, config) + deadline = time.monotonic() + 180 + status: Dict[str, Any] = {} + failed: Optional[Mapping[str, Any]] = None + while time.monotonic() < deadline and failed is None: + status = _status(stack, connector) or {} + failed = _failed_task(status) + if failed is None: + time.sleep(2) + assert failed is not None, f"a sink with both selector keys did not fail: {status}" + + # Both surfaces carry the refusal (seen on cp-kafka-connect 7.9.10 + # with sink 1.9.2). The task status trace, which forge's runner + # reads, starts with the IllegalArgumentException itself; the + # worker logs "Task threw an uncaught and unrecoverable exception" + # for the task, then the same exception. The connector stays + # RUNNING: only the task fails, which is why the runner polls the + # task states and not just the connector's. + assert status.get("connector", {}).get("state") == "RUNNING", status + trace = str(failed.get("trace", "")) + head = trace.splitlines()[0] if trace else "" + assert head.startswith("java.lang.IllegalArgumentException: "), trace + assert f"Cannot create catalog iceberg, {_BOTH_KEYS_ERROR}" in head, trace + assert "type=rest, catalog-impl=org.apache.iceberg.rest.RESTCatalog" in head, trace + + log = stack.worker_log() + killed = f"WorkerSinkTask{{id={connector}-0}} Task threw an uncaught and unrecoverable" + assert killed in log, f"no task-killed line for {connector} in the worker log" + assert any( + _BOTH_KEYS_ERROR in line and "RESTCatalog" in line for line in log.splitlines() + ), "the worker log does not carry the refusal" + + # And nothing reached the catalog. + assert _load_table(stack, table) is None + finally: + _delete_connector(stack, connector) + + +def test_derived_glue_config_gets_past_the_catalog_gate(request: pytest.FixtureRequest): + """The forge-derived Glue config (``catalog-impl`` only) gets past the + ``CatalogUtil`` check on the same worker that refuses both keys. + + No record is produced, so the sink never calls AWS: the GlueCatalog is + built (region from the binding, credentials resolved lazily) and the task + stays RUNNING with its commit coordinator started. + """ + suffix = uuid.uuid4().hex[:8] + topic = f"orders-glue-{suffix}" + connector = f"fluid-lk-glue-{suffix}" + location = { + "database": _NAMESPACE, + "table": f"orders_{suffix}", + "bucket": _BUCKET, + "region": "us-east-1", + } # no ``catalog``: AWS defaults to Glue + config = _derive(_contract(f"live.glue.orders_{suffix}", location), topic) + + # The startup crash, checked before any container starts. + assert config["iceberg.catalog.catalog-impl"] == GLUE_CATALOG_IMPL + assert "iceberg.catalog.type" not in config + + stack: _Stack = request.getfixturevalue("lakekeeper_stack") + with stack.logs_on_failure(request.node.name): + try: + _create_topic(stack, topic) + _put_connector(stack, connector, config) + # RUNNING is the proof: sink 1.9.2 builds the catalog in + # IcebergSinkTask.start(), and Connect reports a task RUNNING only + # after start() returns. The both-keys config never gets there. + _wait_running(stack, connector) + + # The leader task then starts the commit coordinator with that + # catalog, whose consumer group joins the control topic. + coordinator = f"groupId=connect-{connector}-coord] Adding newly assigned partitions" + deadline = time.monotonic() + 120 + while time.monotonic() < deadline and coordinator not in stack.worker_log(): + time.sleep(2) + assert coordinator in stack.worker_log(), f"{connector} never started a coordinator" + + status = _status(stack, connector) or {} + assert [t.get("state") for t in status.get("tasks") or []] == ["RUNNING"], status + # The worker is shared with the both-keys test, whose refusal names + # RESTCatalog; the old Glue crash named GlueCatalog. + refusals = [ + line + for line in stack.worker_log().splitlines() + if _BOTH_KEYS_ERROR in line and "GlueCatalog" in line + ] + assert not refusals, refusals + finally: + _delete_connector(stack, connector) diff --git a/tests/providers/aws/test_planner_iceberg_catalog.py b/tests/providers/aws/test_planner_iceberg_catalog.py new file mode 100644 index 00000000..96e80293 --- /dev/null +++ b/tests/providers/aws/test_planner_iceberg_catalog.py @@ -0,0 +1,166 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The native AWS planner provisions Glue only for a Glue-cataloged table. + +``catalog: lakekeeper`` on an AWS Iceberg expose used to stream over the +Iceberg REST protocol while this planner still emitted +``glue.ensure_database`` + ``glue.ensure_iceberg_table`` for the same name: a +second, metadata-less claim on a table the REST catalog owns. The planner now +reads the shared ``is_glue_cataloged`` classification. An absent catalog on +AWS is Glue, so those contracts must plan exactly as before. +""" + +from __future__ import annotations + +import logging + +import pytest + +from fluid_build.providers.aws.plan.planner import plan_actions + +pytestmark = [pytest.mark.unit, pytest.mark.provider] + +ACCT = "123456789012" +REGION = "us-east-1" +STAGING = f"{ACCT}-fluid-staging" + +_LAKEKEEPER = { + "catalog": "lakekeeper", + "uri": "http://lakekeeper:8181/catalog", + "warehouse": "demo", +} + + +def _contract(location, *, fmt="iceberg", platform="aws", policies=None, extra_exposes=()): + contract = { + "id": "analytics.lake", + "name": "Lake", + "exposes": [ + { + "exposeId": "orders", + "kind": "table", + "binding": {"platform": platform, "format": fmt, "location": location}, + "contract": {"schema": [{"name": "order_id", "type": "string"}]}, + }, + *extra_exposes, + ], + } + if policies: + contract["metadata"] = {"policies": policies} + return contract + + +def _ops(contract): + return [ + (a["op"], a.get("database") or a.get("bucket")) + for a in plan_actions(contract, ACCT, REGION, logging.getLogger("t")) + ] + + +def _glue_ops(contract): + return [op for op, _ in _ops(contract) if op.startswith(("glue.", "iam.bind_glue"))] + + +class TestGlueCatalogUnchanged: + def test_absent_catalog_plans_the_glue_database_and_iceberg_table(self): + loc = {"database": "sales", "table": "orders", "bucket": "lake"} + assert _ops(_contract(loc)) == [ + ("glue.ensure_database", "sales"), + ("s3.ensure_bucket", "lake"), + ("s3.ensure_bucket", STAGING), + ("glue.ensure_iceberg_table", "sales"), + ] + + @pytest.mark.parametrize("spelling", ["glue", "GLUE"]) + def test_explicit_glue_plans_identically_to_absent(self, spelling): + loc = {"database": "sales", "table": "orders", "bucket": "lake"} + absent = plan_actions(_contract(dict(loc)), ACCT, REGION) + explicit = plan_actions(_contract({**loc, "catalog": spelling}), ACCT, REGION) + assert explicit == absent + + def test_non_iceberg_format_ignores_the_catalog(self): + # ``catalog`` is an Iceberg concept: a parquet table is Glue-cataloged + # whatever the key says, so its plan must not change. + loc = {"database": "sales", "table": "orders", "bucket": "lake", "catalog": "lakekeeper"} + assert ("glue.ensure_table", "sales") in _ops(_contract(loc, fmt="parquet")) + + +class TestNonGlueIcebergCatalog: + @pytest.mark.parametrize( + "catalog", ["lakekeeper", "LakeKeeper", "rest", "iceberg_rest", "polaris", "nessie"] + ) + def test_no_glue_database_or_table(self, catalog): + loc = {"database": "sales", "table": "orders", "bucket": "lake", "catalog": catalog} + assert _glue_ops(_contract(loc)) == [] + + def test_declared_bucket_is_still_provisioned(self): + loc = {"database": "sales", "table": "orders", "bucket": "lake", **_LAKEKEEPER} + assert _ops(_contract(loc)) == [ + ("s3.ensure_bucket", "lake"), + ("s3.ensure_bucket", STAGING), + ] + + def test_no_fallback_data_bucket_without_a_declared_one(self): + # ``{account}-fluid-data`` only ever backed the Glue database's + # location, and there is no Glue database here. + loc = {"database": "sales", "table": "orders", **_LAKEKEEPER} + assert _ops(_contract(loc)) == [("s3.ensure_bucket", STAGING)] + + @pytest.mark.parametrize("platform", ["glue", "athena", "AWS"]) + def test_every_aws_platform_spelling(self, platform): + loc = {"database": "sales", "table": "orders", **_LAKEKEEPER} + assert _glue_ops(_contract(loc, platform=platform)) == [] + + def test_iceberg_table_alias_format(self): + loc = {"database": "sales", "table": "orders", **_LAKEKEEPER} + assert _glue_ops(_contract(loc, fmt="iceberg-table")) == [] + + def test_glue_expose_sharing_the_database_still_gets_it(self): + # The skipped Lakekeeper expose must not mark ``sales`` as created. + parquet = { + "exposeId": "raw", + "kind": "table", + "binding": { + "platform": "aws", + "format": "parquet", + "location": {"database": "sales", "table": "raw", "bucket": "lake"}, + }, + } + loc = {"database": "sales", "table": "orders", "bucket": "lake", **_LAKEKEEPER} + ops = _ops(_contract(loc, extra_exposes=[parquet])) + assert ops.count(("glue.ensure_database", "sales")) == 1 + assert ops.count(("s3.ensure_bucket", "lake")) == 1 + assert ("glue.ensure_table", "sales") in ops + assert ("glue.ensure_iceberg_table", "sales") not in ops + + +class TestIamPolicies: + _POLICIES = {"read": ["analyst"]} + + def test_glue_binding_still_binds_the_database(self): + contract = _contract({"table": "orders"}, policies=self._POLICIES) + contract["exposes"][0]["binding"]["database"] = "sales" + assert ("iam.bind_glue_database", "sales") in _ops(contract) + + def test_lakekeeper_binding_skips_the_glue_bind_and_says_so(self, caplog): + contract = _contract({"table": "orders", **_LAKEKEEPER}, policies=self._POLICIES) + contract["exposes"][0]["binding"]["database"] = "sales" + with caplog.at_level(logging.WARNING, logger="t"): + ops = _ops(contract) + assert not [op for op, _ in ops if op.startswith("iam.")] + assert any( + "glue_iam_binding_skipped" in r.getMessage() and "lakekeeper" in r.getMessage() + for r in caplog.records + ), [r.getMessage() for r in caplog.records] diff --git a/tests/providers/test_iceberg_catalog_kinds.py b/tests/providers/test_iceberg_catalog_kinds.py new file mode 100644 index 00000000..937b35dd --- /dev/null +++ b/tests/providers/test_iceberg_catalog_kinds.py @@ -0,0 +1,380 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The Iceberg catalog-kind table, and every emitter agreeing with it. + +``catalog: lakekeeper`` used to stream over REST while dbt's ``catalogs.yml`` +wrote a Snowflake-managed table and the AWS IaC created a Glue table for the +same name: four emitters each classified the free-string ``catalog`` by hand. +The unit tests pin the table; the agreement matrix runs EVERY emitter over +every kind and checks each answer against the kind's row, so a fifth +hand-rolled classification fails here rather than in a user's lake. +""" + +from __future__ import annotations + +from typing import Any, Dict, Optional + +import pytest +import yaml + +from fluid_build.build_runners.debezium.iceberg_sink import emit_debezium_iceberg_sink_config +from fluid_build.build_runners.kafka_connect.iceberg_sink import emit_iceberg_sink_config +from fluid_build.providers import _iceberg_catalog as ic +from fluid_build.providers._iceberg_catalog import ( + EXTERNAL_ICEBERG_CATALOGS, + FAMILY_GLUE, + FAMILY_UNKNOWN, + binding_catalog_kind, + canonical_catalog_kind, + catalog_kind_info, + default_catalog_kind, + find_iceberg_expose_binding, + iceberg_catalog_kind, + iceberg_sink_exposes, + is_glue_cataloged, + is_object_store_uri, + known_catalog_kinds, + resolve_iceberg_catalog, +) + +pytestmark = pytest.mark.unit + +#: Apache Iceberg ``CatalogUtil.buildIcebergCatalog`` accepts exactly these +#: ``type`` values (1.10.0; ``bigquery`` from 1.10) and throws on any other. +_CATALOG_UTIL_TYPES = {"hadoop", "hive", "rest", "glue", "nessie", "jdbc", "bigquery"} + +ALL_KINDS = sorted(ic._CATALOG_KINDS) + + +class TestCanonicalKind: + @pytest.mark.parametrize( + "raw,expected", + [ + ("lakekeeper", "lakekeeper"), + ("Lakekeeper", "lakekeeper"), + (" LAKEKEEPER ", "lakekeeper"), + ("iceberg_rest", "rest"), + ("iceberg-rest", "rest"), + ("ICEBERG_REST", "rest"), + ("snowflake", "snowflake-managed"), + ("Snowflake_Managed", "snowflake-managed"), + ("glue", "glue"), + (None, ""), + ("", ""), + ("gravitino", "gravitino"), + ], + ) + def test_folding_and_aliases(self, raw, expected): + assert canonical_catalog_kind(raw) == expected + + def test_unknown_value_gets_the_unknown_row_with_historic_fallbacks(self): + info = catalog_kind_info("gravitino") + assert info.family == FAMILY_UNKNOWN + # The sink keeps talking REST and dbt keeps built_in; fluid validate + # is what refuses the value. + assert info.runtime_type == "rest" + assert info.snowflake_catalog_type == "built_in" + + def test_known_kinds_lists_canonical_names_and_aliases(self): + known = known_catalog_kinds() + assert "lakekeeper" in known and "iceberg-rest" in known and "snowflake" in known + + +class TestTableIntegrity: + @pytest.mark.parametrize("kind", ALL_KINDS) + def test_wire_value_is_one_iceberg_accepts_and_never_both(self, kind): + info = catalog_kind_info(kind) + # catalog-impl XOR type: CatalogUtil throws when both are set. + assert bool(info.runtime_type) != bool(info.catalog_impl) + if info.runtime_type: + assert info.runtime_type in _CATALOG_UTIL_TYPES + + def test_no_vendor_name_ever_reaches_the_wire(self): + # No engine or SDK has a ``lakekeeper``/``polaris``/``unity`` type. + for vendor in ("lakekeeper", "polaris", "unity", "snowflake-managed"): + assert catalog_kind_info(vendor).runtime_type == "rest" + + def test_dynamodb_is_reached_by_impl_only(self): + info = catalog_kind_info("dynamodb") + assert info.runtime_type is None + assert info.catalog_impl == "org.apache.iceberg.aws.dynamodb.DynamoDbCatalog" + + def test_external_set_is_derived_from_the_table(self): + assert "lakekeeper" in EXTERNAL_ICEBERG_CATALOGS + assert "iceberg_rest" in EXTERNAL_ICEBERG_CATALOGS + for name in EXTERNAL_ICEBERG_CATALOGS: + assert catalog_kind_info(name).snowflake_catalog_type == "iceberg_rest" + for kind in ALL_KINDS: + external = catalog_kind_info(kind).snowflake_catalog_type == "iceberg_rest" + assert (kind in EXTERNAL_ICEBERG_CATALOGS) == external + + @pytest.mark.parametrize("kind", ["rest", "lakekeeper", "polaris", "unity"]) + def test_rest_catalogs_need_uri_and_warehouse(self, kind): + info = catalog_kind_info(kind) + assert info.speaks_rest + assert set(info.sink_requires) == {"uri", "warehouse"} + + +class TestKindPrecedence: + @pytest.mark.parametrize( + "platform,expected", + [ + ("aws", "glue"), + ("glue", "glue"), # a cloud alias of aws + ("athena", "glue"), + ("snowflake", "snowflake-managed"), + ("gcp", "rest"), + ("local", "rest"), + ("", "rest"), + ], + ) + def test_platform_default(self, platform, expected): + assert default_catalog_kind({"platform": platform}) == expected + + def test_location_catalog_beats_the_platform_default(self): + binding = {"platform": "aws", "location": {"catalog": "Lakekeeper"}} + assert binding_catalog_kind(binding) == "lakekeeper" + + def test_sink_catalog_beats_the_location(self): + binding = {"platform": "aws", "location": {"catalog": "glue"}} + assert iceberg_catalog_kind(binding, {"catalog": "lakekeeper"}) == "lakekeeper" + + def test_sink_may_be_a_spec_object(self): + class _Sink: + catalog = "iceberg_rest" + partition_by = None + + assert iceberg_catalog_kind({"platform": "aws"}, _Sink()) == "rest" + + def test_absent_everywhere_falls_to_the_platform(self): + assert iceberg_catalog_kind({"platform": "aws"}, {}) == "glue" + + +class TestGlueCataloged: + @pytest.mark.parametrize( + "fmt,catalog,platform,expected", + [ + ("iceberg", None, "aws", True), + ("iceberg", "glue", "aws", True), + ("iceberg", "lakekeeper", "aws", False), + ("iceberg", "iceberg-rest", "aws", False), + ("iceberg_table", "polaris", "aws", False), + # location.catalog is meaningless for a plain Glue table format. + ("parquet", "lakekeeper", "aws", True), + ("csv", None, "aws", True), + # an Iceberg expose off AWS defaults to a REST catalog + ("iceberg", None, "gcp", False), + ], + ) + def test_predicate(self, fmt, catalog, platform, expected): + location: Dict[str, Any] = {"database": "d", "table": "t"} + if catalog: + location["catalog"] = catalog + binding = {"platform": platform, "format": fmt, "location": location} + assert is_glue_cataloged(binding) is expected + + +class TestSinkExposeSelection: + def test_confluent_and_non_iceberg_exposes_are_not_sink_targets(self): + contract = { + "exposes": [ + {"exposeId": "a", "binding": {"platform": "confluent", "format": "iceberg"}}, + {"exposeId": "b", "binding": {"platform": "aws", "format": "parquet"}}, + {"exposeId": "c", "binding": {"platform": "aws", "format": "Iceberg_Table"}}, + ] + } + assert [e["exposeId"] for e in iceberg_sink_exposes(contract)] == ["c"] + assert find_iceberg_expose_binding(contract) == { + "platform": "aws", + "format": "Iceberg_Table", + } + + def test_no_target(self): + assert find_iceberg_expose_binding({"exposes": []}) is None + + +class TestObjectStoreSchemes: + @pytest.mark.parametrize( + "value", ["s3://b/w", "s3a://b", "s3n://b", "gs://b", "gcs://b", "abfss://c@a/x"] + ) + def test_object_store(self, value): + assert is_object_store_uri(value) + + @pytest.mark.parametrize("value", ["analytics", "0190-uuid/analytics", "", None]) + def test_names(self, value): + assert not is_object_store_uri(value) + + +def _binding(catalog: Optional[str], platform: str = "aws", **location) -> Dict[str, Any]: + loc: Dict[str, Any] = {"database": "sales", "table": "orders", **location} + if catalog is not None: + loc["catalog"] = catalog + return {"platform": platform, "format": "iceberg", "location": loc} + + +class TestResolver: + def test_lakekeeper_streams_over_rest_with_no_file_io(self): + r = resolve_iceberg_catalog( + _binding("lakekeeper", uri="http://lk:8181/catalog", warehouse="analytics") + ) + assert r.catalog_type == "rest" + assert r.catalog_impl is None + assert r.kind == "lakekeeper" + # A warehouse NAME: the catalog vends the table's FileIO config. + assert r.io_impl is None + assert r.uri == "http://lk:8181/catalog" + assert r.warehouse == "analytics" + + def test_glue_is_unchanged(self): + r = resolve_iceberg_catalog(_binding(None, bucket="lake", region="eu-west-1")) + assert r.catalog_type == "glue" + assert r.catalog_impl == ic.GLUE_CATALOG_IMPL + assert r.kind == "glue" + + def test_unknown_kind_keeps_the_rest_fallback(self): + assert resolve_iceberg_catalog(_binding("gravitino")).catalog_type == "rest" + + +@pytest.mark.parametrize("kind", ALL_KINDS + ["gravitino"]) +def test_both_sink_derivers_emit_impl_xor_type(kind): + """THE startup crash: a Glue sink once carried both keys and Iceberg's + CatalogUtil refused it before writing a record.""" + resolved = resolve_iceberg_catalog( + _binding(kind, uri="http://c/catalog", warehouse="w", bucket="lake", region="us-east-1") + ) + kc = emit_iceberg_sink_config(resolved, product_id="p.x", topics=["t"]) + assert ("iceberg.catalog.type" in kc) != ("iceberg.catalog.catalog-impl" in kc) + if "iceberg.catalog.type" in kc: + assert kc["iceberg.catalog.type"] in _CATALOG_UTIL_TYPES + + dbz = emit_debezium_iceberg_sink_config(resolved) + assert ("type" in dbz) != ("catalog-impl" in dbz) + if "type" in dbz: + assert dbz["type"] in _CATALOG_UTIL_TYPES + + +# --------------------------------------------------------------------------- +# Cross-emitter agreement: every emitter answers from the kind's row +# --------------------------------------------------------------------------- + +#: Every kind, plus absent and an unknown value. +MATRIX = ALL_KINDS + [None, "gravitino"] + + +def _row(kind: Optional[str], platform: str): + effective = canonical_catalog_kind(kind) or default_catalog_kind({"platform": platform}) + return catalog_kind_info(effective) + + +def _contract(binding: Dict[str, Any]) -> Dict[str, Any]: + return { + "fluidVersion": "0.7.6", + "kind": "DataProduct", + "id": "gold.orders", + "name": "orders", + "metadata": {"layer": "Gold"}, + "exposes": [ + { + "exposeId": "orders", + "kind": "table", + "binding": binding, + "contract": {"schema": [{"name": "id", "type": "string"}]}, + } + ], + } + + +@pytest.mark.parametrize("kind", MATRIX) +def test_dbt_snowflake_catalogs_yml_follows_the_row(kind): + from fluid_build.engines.dbt.catalogs_yml import generate_catalogs_yml + + binding = _binding( + kind, + platform="snowflake", + uri="http://c/catalog", + warehouse="s3://lake/p/", + iam_role_arn="arn:aws:iam::123456789012:role/sf", + ) + build = {"engine": "dbt", "execution": {"runtime": {"platform": "snowflake"}}} + content = generate_catalogs_yml(_contract(binding), build) + expected = _row(kind, "snowflake").snowflake_catalog_type + if expected is None: + # Snowflake has no catalog integration for it: the expose is skipped + # (and fluid validate says why), never written as Snowflake-managed. + assert content is None + return + assert content is not None + integration = yaml.safe_load(content)["catalogs"][0]["write_integrations"][0] + assert integration["catalog_type"] == expected + + +@pytest.mark.parametrize("kind", MATRIX) +def test_snowflake_iac_creates_a_volume_only_for_snowflake_managed(kind): + from fluid_build.iac import get_iac_plugin + + binding = _binding( + kind, + platform="snowflake", + warehouse="s3://lake/p/", + iam_role_arn="arn:aws:iam::123456789012:role/sf", + schema="PUBLIC", + ) + res = get_iac_plugin("snowflake").emit(_contract(binding)) + has_volume = bool(res.get("snowflake_external_volume")) + assert has_volume == (_row(kind, "snowflake").snowflake_catalog_type == "built_in") + + +@pytest.mark.parametrize("kind", MATRIX) +def test_aws_iac_creates_a_glue_table_only_for_glue(kind): + from fluid_build.iac.providers.aws import AwsIacPlugin + + binding = _binding(kind, platform="aws", bucket="lake", region="us-east-1") + if _row(kind, "aws").family == FAMILY_UNKNOWN: + # Whether a Glue table exists hangs on the answer, and apply does not + # run fluid validate: an unknown value is refused, not guessed. + from fluid_build.iac.base import UnsupportedBindingError + + with pytest.raises(UnsupportedBindingError, match="unknown-iceberg-catalog|does not know"): + AwsIacPlugin().emit(_contract(binding)) + return + res = AwsIacPlugin().emit(_contract(binding)) + has_glue_table = bool(res.get("aws_glue_catalog_table")) + assert has_glue_table == (_row(kind, "aws").family == FAMILY_GLUE) + # The bucket is the catalog's storage either way. + assert res.get("aws_s3_bucket") + + +@pytest.mark.parametrize("kind", MATRIX) +def test_sink_validator_demands_exactly_what_the_row_requires(kind): + from fluid_build.build_runners.kafka_connect.iceberg_sink_validation import ( + validate_iceberg_sink, + ) + + contract = _contract(_binding(kind, platform="gcp")) + contract["builds"] = [ + { + "id": "ingest", + "engine": "kafka-connect", + "properties": {"sink": {"format": "iceberg"}}, + } + ] + errors, _ = validate_iceberg_sink(contract) + row = _row(kind, "gcp") + if row.family == FAMILY_UNKNOWN: + assert any("gravitino" in e for e in errors) + return + for key in ("uri", "warehouse"): + demanded = any(f"binding.location.{key}" in e for e in errors) + assert demanded == (key in row.sink_requires), (key, errors) diff --git a/tests/test_iceberg_catalog_kind_schema.py b/tests/test_iceberg_catalog_kind_schema.py new file mode 100644 index 00000000..02b06675 --- /dev/null +++ b/tests/test_iceberg_catalog_kind_schema.py @@ -0,0 +1,137 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Schema surface for the Iceberg catalog kinds. + +``location.catalog`` is a free string the kind table in +``providers/_iceberg_catalog.py`` classifies, and ``sink.catalog`` is an enum. +These tests pin the schema's enum and prose to the table rather than to a +hand-kept list, so a kind the table does not know cannot pass the schema. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +import fluid_build +from fluid_build.providers._iceberg_catalog import ( + FAMILY_UNKNOWN, + canonical_catalog_kind, + catalog_kind_info, + known_catalog_kinds, +) +from fluid_build.schema_manager import FluidSchemaManager + +pytestmark = [pytest.mark.unit] + +_SCHEMA_DIR = Path(fluid_build.__file__).parent / "schemas" + + +def _schema(version: str) -> dict: + return json.loads((_SCHEMA_DIR / f"fluid-schema-{version}.json").read_text(encoding="utf-8")) + + +def _contract(version: str, *, sink_catalog=None, location=None) -> dict: + sink = {"format": "iceberg"} + if sink_catalog is not None: + sink["catalog"] = sink_catalog + return { + "fluidVersion": version, + "kind": "DataProduct", + "id": "compat.ops.iceberg_catalog_kind", + "name": "Iceberg Catalog Kind", + "description": "Schema coverage for the Iceberg catalog kinds.", + "domain": "ops", + "metadata": { + "layer": "Bronze", + "owner": {"team": "platform-ops", "email": "ops@example.com"}, + }, + "builds": [ + { + "id": "ingest_events", + "pattern": "acquisition", + "engine": "kafka-connect", + "properties": { + "source": { + "kind": "postgres", + "mode": "incremental_append", + "streams": ["public.events"], + }, + "sink": sink, + "kafka-connect": {"iceberg_sink_enabled": True}, + }, + } + ], + "exposes": [ + { + "exposeId": "events", + "kind": "table", + "version": "1.0.0", + "binding": { + "platform": "aws", + "format": "iceberg", + "location": {"database": "bronze", "table": "events", **(location or {})}, + }, + "contract": {"schema": [{"name": "id", "type": "integer", "required": True}]}, + } + ], + } + + +def _validate(contract: dict, version: str): + return FluidSchemaManager().validate_contract(contract, version) + + +_LAKEKEEPER_LOCATION = { + "catalog": "lakekeeper", + "uri": "http://lakekeeper:8181/catalog", + "warehouse": "demo", +} + + +@pytest.mark.parametrize("version", ["0.7.5", "0.7.6"]) +def test_base_contract_is_valid(version: str) -> None: + """Guard: a failure below is the catalog value's, not the fixture's.""" + result = _validate(_contract(version), version) + assert result.is_valid, result.errors + + +@pytest.mark.parametrize("version", ["0.7.5", "0.7.6"]) +def test_location_catalog_lakekeeper_is_a_free_string(version: str) -> None: + contract = _contract(version, location=dict(_LAKEKEEPER_LOCATION)) + result = _validate(contract, version) + assert result.is_valid, f"{version} refused location.catalog lakekeeper: {result.errors}" + + +@pytest.mark.parametrize("version", ["0.7.5", "0.7.6"]) +def test_sink_catalog_enum_names_only_kinds_the_table_knows(version: str) -> None: + """An enum member the kind table does not know would pass the schema and + then be refused by ``fluid validate`` (or, worse, hit an emitter fallback).""" + enum = _schema(version)["$defs"]["acquisitionSink"]["properties"]["catalog"]["enum"] + unknown = [v for v in enum if catalog_kind_info(v).family == FAMILY_UNKNOWN] + assert unknown == [] + + +def test_location_catalog_description_lists_every_kind() -> None: + """The prose is the only place a contract author learns the kinds, so it + must name every canonical kind in the table.""" + description = _schema("0.7.6")["$defs"]["bindingLocation"]["properties"]["catalog"][ + "description" + ] + canonical = [k for k in known_catalog_kinds() if canonical_catalog_kind(k) == k] + missing = [k for k in canonical if k not in description] + assert missing == [], f"location.catalog description omits {missing}" diff --git a/tests/test_policy_compiler.py b/tests/test_policy_compiler.py index fd00c2f2..63296e0c 100644 --- a/tests/test_policy_compiler.py +++ b/tests/test_policy_compiler.py @@ -14,6 +14,8 @@ """Tests for fluid_build/policy/compiler.py — access-policy → IAM bindings.""" +import pytest + from fluid_build.policy.compiler import ( SAFE_BQ_PERMS, SAFE_S3_PERMS, @@ -178,6 +180,93 @@ def test_no_bindings_warning(self): assert any("No IAM bindings" in w for w in warnings) +class TestIcebergCatalogRouting: + """An Iceberg expose is granted where its catalog lives. + + Every ``iceberg`` expose used to route to the AWS compiler on any + platform, so a Lakekeeper table, a Snowflake-managed table and a + ``platform: local`` table each got a ``glue.table`` grant on a Glue table + that does not exist (and ``policy-apply`` took ``provider: aws`` from it). + """ + + _LOC = {"bucket": "bkt", "database": "mydb", "table": "tbl", "region": "eu-west-1"} + + @staticmethod + def _types(bindings): + return sorted(b["resource_type"] for b in bindings) + + @staticmethod + def _catalog_warnings(warnings): + return [w for w in warnings if "cataloged in" in w] + + @pytest.mark.parametrize("catalog", [None, "glue", "GLUE"]) + def test_glue_catalog_keeps_s3_and_glue_grants(self, catalog): + loc = dict(self._LOC, **({"catalog": catalog} if catalog else {})) + bindings, warnings = compile_policy(_contract("aws", "iceberg", loc)) + assert self._types(bindings) == ["glue.table", "s3.bucket"] + assert self._catalog_warnings(warnings) == [] + + @pytest.mark.parametrize("catalog", ["lakekeeper", "Lakekeeper", "iceberg_rest", "polaris"]) + def test_rest_catalog_on_aws_gets_no_glue_grant(self, catalog): + loc = dict(self._LOC, catalog=catalog) + bindings, warnings = compile_policy(_contract("aws", "iceberg", loc)) + # The bucket is still the table's storage; the Glue table does not exist. + assert self._types(bindings) == ["s3.bucket"] + (warning,) = self._catalog_warnings(warnings) + assert "user@example.com" in warning and "['read']" in warning + assert "not AWS Glue" in warning + + def test_rest_catalog_warning_names_the_canonical_kind(self): + loc = dict(self._LOC, catalog="LakeKeeper") + _, warnings = compile_policy(_contract("aws", "iceberg", loc)) + assert "'lakekeeper'" in self._catalog_warnings(warnings)[0] + + def test_rest_catalog_without_bucket_compiles_nothing_and_says_so(self): + loc = {"database": "mydb", "table": "tbl", "catalog": "lakekeeper"} + bindings, warnings = compile_policy(_contract("aws", "iceberg", loc)) + assert bindings == [] + assert len(self._catalog_warnings(warnings)) == 1 + assert any("No IAM bindings" in w for w in warnings) + + def test_snowflake_managed_iceberg_gets_snowflake_rbac(self): + loc = {"database": "DB", "schema": "SCH", "table": "T"} + bindings, warnings = compile_policy(_contract("snowflake", "iceberg", loc)) + assert self._types(bindings) == ["snowflake.table"] + assert bindings[0]["grants"] == SAFE_SNOWFLAKE_PERMS["readData"] + # Snowflake IS the catalog: its grant is the whole enforcement. + assert self._catalog_warnings(warnings) == [] + + def test_external_catalog_on_snowflake_warns_grant_covers_snowflake_only(self): + loc = {"database": "DB", "schema": "SCH", "table": "T", "catalog": "lakekeeper"} + bindings, warnings = compile_policy(_contract("snowflake", "iceberg", loc)) + assert self._types(bindings) == ["snowflake.table"] + (warning,) = self._catalog_warnings(warnings) + assert "Snowflake readers only" in warning + + @pytest.mark.parametrize("platform", ["local", "azure"]) + def test_non_aws_platform_gets_no_aws_grant(self, platform): + bindings, warnings = compile_policy(_contract(platform, "iceberg", dict(self._LOC))) + assert [b for b in bindings if b["provider"] == "aws"] == [] + assert len(self._catalog_warnings(warnings)) == 1 + + def test_one_warning_per_grant_and_expose(self): + grants = [ + {"principal": "a@b.com", "permissions": ["read"]}, + {"principal": "c@d.com", "permissions": ["write"]}, + ] + loc = dict(self._LOC, catalog="lakekeeper") + _, warnings = compile_policy(_contract("aws", "iceberg", loc, grants=grants)) + catalog_warnings = self._catalog_warnings(warnings) + assert len(catalog_warnings) == 2 + assert "a@b.com" in catalog_warnings[0] and "c@d.com" in catalog_warnings[1] + + def test_non_iceberg_format_ignores_the_catalog_key(self): + loc = dict(self._LOC, catalog="lakekeeper") + bindings, warnings = compile_policy(_contract("aws", "parquet", loc)) + assert self._types(bindings) == ["glue.table", "s3.bucket"] + assert self._catalog_warnings(warnings) == [] + + class TestGcpBindingsInternal: def test_bigquery_no_dataset(self): bindings = [] @@ -204,6 +293,12 @@ def test_database_alias(self): assert len(glue) == 1 assert glue[0]["database"] == "myds" + def test_glue_false_emits_only_the_s3_statement(self): + bindings = [] + loc = {"bucket": "bkt", "database": "db", "table": "t"} + _compile_aws_bindings(bindings, "iceberg", loc, "user@a.com", ["read"], glue=False) + assert [b["resource_type"] for b in bindings] == ["s3.bucket"] + class TestSnowflakeBindingsInternal: def test_no_database(self):