diff --git a/.github/workflows/iac-tests.yml b/.github/workflows/iac-tests.yml index 7bbfce20..77940923 100644 --- a/.github/workflows/iac-tests.yml +++ b/.github/workflows/iac-tests.yml @@ -137,6 +137,31 @@ jobs: -q -p no:randomly --junitxml=state-drift.xml .venv/bin/python scripts/ci/assert_lane_coverage.py state-drift.xml \ --require "state drift=tests/iac/test_iac_state_drift_moto.py" + # The default state key names the provider, and the first apply after the + # upgrade moves the old key's state with `tofu init -migrate-state`. Only a + # real apply, move and plan prove the plan after it changes nothing. + - name: Stage 1 — per-provider state key and its migration (tofu vs moto, creds-free) + run: | + .venv/bin/python -m pytest tests/iac/test_iac_state_key_migration_moto.py \ + -q -p no:randomly --junitxml=state-key-migration.xml + .venv/bin/python scripts/ci/assert_lane_coverage.py state-key-migration.xml \ + --require "state key migration=tests/iac/test_iac_state_key_migration_moto.py" + # Governance parity (retention, CMEK, column restrictions, mapped + # principals): tofu validate of each governed GCP shape; a real plan and + # apply with terraform-provider-google against an in-process stand-in for + # the BigQuery REST API (the goccy emulator crashes the provider on + # apply), which is where the data-loss gate decides; and the Lake + # Formation excluded columns planned against moto. Fails if a file only + # skipped. + - name: Stage 1 — governance parity (tofu vs BigQuery stand-in and moto, creds-free) + run: | + .venv/bin/python -m pytest tests/iac/test_iac_gcp_governance_plan.py \ + tests/iac/test_iac_aws_column_restrictions.py tests/iac/test_iac_gcp_governance.py \ + -q -p no:randomly --junitxml=governance-parity.xml + .venv/bin/python scripts/ci/assert_lane_coverage.py governance-parity.xml \ + --require "GCP governance plan=tests/iac/test_iac_gcp_governance_plan.py" \ + --require "AWS column restrictions=tests/iac/test_iac_aws_column_restrictions.py" \ + --require "GCP governance validate=tests/iac/test_iac_gcp_governance.py" # ------------------------------------------------------------------- # Stage 2 — docker emulators. PR + push. diff --git a/.github/workflows/integration-emulated-heavy.yml b/.github/workflows/integration-emulated-heavy.yml index 4e48221d..66beee42 100644 --- a/.github/workflows/integration-emulated-heavy.yml +++ b/.github/workflows/integration-emulated-heavy.yml @@ -96,10 +96,13 @@ jobs: with: tofu_version: 1.12.0 + # ``local`` brings DuckDB: the embedded-SQL chain on the bigquery-emulator + # runs its SQL there. Without it that test skips, and the lane-coverage + # assert below fails the job. - name: Install with heavy-emulator extras run: | python -m pip install --upgrade pip - python -m pip install -e ".[dev,test-emulators-heavy]" + python -m pip install -e ".[dev,test-emulators-heavy,local]" # Skip-clean: a secret cannot gate a job-level `if:`. With no token # the LocalStack steps skip and the GCP half still runs — EXCEPT on @@ -198,6 +201,7 @@ jobs: --require "GCP emulator e2e=tests/iac/test_iac_gcp_emulator_e2e.py" --require "GCP cross-project=tests/iac/test_iac_cross_project_gcp_emulator.py" --require "BigQuery happy path=tests/providers/test_bigquery_emulated_happy_path.py" + --require "BigQuery embedded-SQL chain=tests/providers/test_bigquery_emulated_embedded_sql_chain.py" ) if [ "${{ steps.gate.outputs.localstack }}" = "true" ]; then requirements+=(--require "AWS LocalStack=tests/providers/test_aws_localstack_happy_path.py") diff --git a/.secrets.baseline b/.secrets.baseline index 71c9450f..13d9a9a0 100644 --- a/.secrets.baseline +++ b/.secrets.baseline @@ -139,7 +139,7 @@ "filename": ".github/workflows/iac-tests.yml", "hashed_secret": "a94a8fe5ccb19ba61c4c0873d391e987982fbbd3", "is_verified": false, - "line_number": 222 + "line_number": 247 } ], ".github/workflows/integration.yml": [ @@ -2154,5 +2154,5 @@ } ] }, - "generated_at": "2026-09-28T09:27:07Z" + "generated_at": "2026-09-28T10:03:17Z" } diff --git a/CHANGELOG.md b/CHANGELOG.md index fd764914..15ba18dc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Upgrade notes + +- **The first `fluid schedule-sync` after upgrading must retire the old DAG + of each env.** An env's Airflow DAGs now live in `__/` with + dag id `____`; 0.16.7 and earlier wrote them to + `/` as `__`. Left in place, the old DAG runs + beside the new one, and both apply the same product against the same state. + With `--delete-scope product` (the default) to a local path or a `git+ssh` + repository, the sync retires them itself: it deletes from `/` + only the DAGs rendered for the same product and env under the old id, and + keeps any other file. For every other transport it prints the step to take: + delete those DAG files at the destination once. `--delete-scope destination` + removes the old directory with the rest of what the sync does not ship. + The report's `superseded_scopes` records which case applied. + ## [0.16.7] — 2026-09-28 A Lake Formation grant that hides columns now applies on AWS, not only plans. diff --git a/HONESTLY_TESTED.md b/HONESTLY_TESTED.md index 0e34a95e..6101fd5d 100644 --- a/HONESTLY_TESTED.md +++ b/HONESTLY_TESTED.md @@ -46,6 +46,7 @@ Markers declared in `pyproject.toml`. | Stage 1 LF bucket-policy modes | `tests/iac/test_iac_lakeformation_bucket_policy.py`, `tests/iac/test_iac_lakeformation_bucket_policy_plan.py`, `tests/iac/test_iac_tofu_validate.py::test_lakeformation_bucket_policy_modes_pass_tofu_validate` | ✅ — the three `bucketPolicy` modes rendered (all-grantees byte-identical to the previous emit), `tofu validate` of each, and a real `tofu plan` against moto's STS proving the default keeps a cross-account grantee's statements and plans no bucket policy for a same-account one. | | Stage 1 LF column grants | `tests/iac/test_iac_lakeformation_column_grants.py`, `tests/iac/test_iac_lakeformation_column_permissions.py`, `tests/iac/test_iac_lakeformation_column_grants_plan.py` | ✅ — the three grant shapes rendered (no columns: `table`; `columns`: `table_with_columns.column_names`, no wildcard; `excludedColumns`: `wildcard = true` + `excluded_column_names`; refused at emit: both together, a column name the expose's schema does not declare, an exclusion of every column, and a column limit on a binding with no `location.table`), and a real `tofu plan` against moto of all three. Plan, not validate: the excluded shape without the wildcard passes `tofu validate` and fails `tofu plan` with "Missing required argument", and the file pins that too. A column-limited grant is emitted and planned with `SELECT` only: Lake Formation takes nothing else on `table_with_columns` ("Permissions modification is invalid" on the demo lab's apply, 28 Sept 2026), a `DESCRIBE` beside it is dropped because the column-limited `SELECT` already lets the principal see the table and Lake Formation refuses `DESCRIBE` to a principal holding a partial `SELECT`, and `ALTER`, `DROP`, `DELETE`, `INSERT` or `ALL` beside a column limit is refused at emit, as is a second grant for the same principal on the same table (Lake Formation refuses the table-level permissions to it, a table-level `SELECT` would read the withheld columns, and the provider reads the pair back as one). `DESCRIBE`'s grant option beside a column limit is refused (no resource can carry it). `SELECT`'s grant option is kept beside `columns` and refused beside `excludedColumns`: the guide's pages disagree (the permissions reference: "you can't include the grant option if column filtering is applied"; the console page offers it for simple column-based access), and *Data filtering limitations* is the specific rule followed: "To grant SELECT with the grant option and column filtering, you must use an include list, not an exclude list." **Not proven here**: moto stores any permission on any resource and `tofu plan` checks none, so the permission rules rest on the AWS LF developer guide and that failed apply; the fixed grant has **not yet been applied live**, and `SELECT` with the grant option on an include list has **not been tried live** (a live `GrantPermissions` of it settles the conflict). | | Stage 1 retention + encryption at rest | `tests/iac/test_iac_aws_retention_encryption.py`, `tests/iac/test_iac_aws_retention_encryption_moto.py`, `tests/iac/test_iac_tofu_validate.py::test_retention_and_kms_pass_tofu_validate`, `tests/cli/test_verify_athena.py` (storage section) | ✅ — `exposes[].lifecycle {retention, expire: true}` and `binding.encryption.kms` (fluid-schema 0.7.6) rendered (prefix-scoped rules, never a whole-bucket rule; key, alias, default SSE-KMS with a bucket key; key policy), `tofu validate` of each shape, and a real `tofu plan` / `apply` / `destroy` against moto: the key policy evaluated against the applying account, an object written with no encryption header landing SSE-KMS under the product key, `fluid verify`'s storage checks passing and then failing on a changed rule and an SSE-S3 object, and dropping the fields deleting the lifecycle configuration and scheduling the key's deletion 7 days out. An existing key is planned against moto too: named by alias, S3 is handed its key ARN; a key pending deletion and an RSA key fail the plan on the SSE configuration's preconditions; the AWS managed key named by its key ARN (which passes the name check) fails it with registerLocation (moto records every key as customer managed, so the test sets its record of alias/aws/s3 to `AWS`). A brownfield bucket carrying an operator's lifecycle rule and SSE-S3 is adopted through `fluid apply`'s own `_adopt_existing`, and the plan shows both configurations as in-place updates. `fluid verify` is covered with botocore stubs, including a key that is not `Enabled` and lifecycle rules filtered by tag, size or a narrower prefix that expire objects sooner. **Not proven here**: moto enforces neither key policies nor Lake Formation, so Athena reading an SSE-KMS location through Lake Formation-vended credentials (the service-linked role's key-policy statement) rests on the AWS documentation and has **not been run live**. | +| Stage 1 GCP governance parity (retention, CMEK, column restrictions, mapped principals) | `tests/iac/test_iac_gcp_governance.py`, `tests/iac/test_iac_gcp_governance_plan.py`, `tests/iac/test_iac_aws_column_restrictions.py`, `tests/cli/test_verify_bigquery_governance.py` | ✅ — rendered and `tofu validate`d: partition expiration (never a table TTL), a Cloud KMS key ring, key, service-agent grant and the dataset and table keys, a Data Catalog taxonomy, policy tags and fine-grained readers, and non-authoritative dataset members with the contract's logical principals mapped by `binding.principals` (placeholders and unmapped principals refused). A real `tofu plan` / `apply` with terraform-provider-google against an in-process stand-in for the BigQuery REST API (`tests/iac/_fake_bigquery.py`; the goccy emulator crashes the provider on apply): adding retention or a key to a live table plans its replacement, which the data-loss gate refuses without `--allow-data-loss`; a new retention is in place; the governed table plans clean; the dataset keeps its own access entries. The same stand-in shows a revoked reader is a member destroy the data-loss gate lets through, and that the first apply after the authoritative access list of 0.16.6 and earlier revokes a grant removed in the same change and then plans clean. AWS: the restriction becomes each Lake Formation grant's excluded columns, a real `tofu plan` against moto accepts them, and `fluid verify`'s Lake Formation check runs against moto's stored grants (and against a listed `TableWildcard` grant). `fluid verify` on BigQuery is covered with real `google.cloud.bigquery` Table/Dataset objects and a fake Data Catalog session. **Not proven here**: nothing ran against real BigQuery, Cloud KMS, Data Catalog or Lake Formation; no emulator enforces policy tags, keys or IAM, so a denied principal's refusal, the service agent's use of the key and a load into a tagged column rest on the providers' documentation. | | Stage 2 cross-account (two-account LocalStack) | `tests/integration/test_iac_cross_account_localstack.py` | ✅ 10 tests, gated on `FLUID_IAC_LIVE_XACCT=1` + a second LocalStack Pro carrying `lakeformation`. Producer stack applies in account A (`000000000000`); every grant names a role in account B (`222222222222`) — LocalStack derives the account from the access-key id. Verified live: LF permission + LF prefix-registration + `aws_s3_bucket_policy` all land and read back; the shared-**pool** bucket is referenced via `data.aws_s3_bucket`, its `GetObject` grant is scoped to `location.path` and its `ListBucket` carries the `s3:prefix` condition. Under `ENFORCE_IAM=1` the emitted bucket policy is proven to be the **deciding** access control: with a deliberately broad identity policy on the account-B role, the granted prefix reads and the sibling tenant's prefix is denied — and widening only the bucket policy flips that denial (causal control). **Still NOT proven here**: LF cross-account *authorization* (LocalStack rejects `GrantPermissions` under IAM enforcement even for a registered `DataLakeAdmin` — `docs/upstream-issues/localstack-lakeformation-grant-auth.md`), and cross-account Glue catalog sharing (no AWS RAM in LocalStack; account B gets `EntityNotFoundException`, which is state isolation, not a denial). | | Stage 1 Glue catalog enrichment | `tests/iac/test_iac_aws.py::TestAwsGlueCatalogEnrichment` (6) + `test_iac_tofu_validate.py[aws]` | ✅ — `aws_glue_catalog_table.description` + per-column comments + fluid_layer/fluid_product_type/fluid_domain/fluid_version/fluid_contract/forge.pii. parameters, absorbed from the retired `GlueCatalogRegistrar`. **Zero new schema fields** — reads existing `description`/`metadata.description`/`metadata.layer`/`metadata.productType`/`domain`/`fluidVersion`/`column.tags[]`/`column.description`. | | Stage 3 Glue catalog enrichment | `tests/iac/test_iac_aws_real_e2e.py::test_real_iceberg_on_glue_round_trip` | ✅ live — Description + Parameters + per-column Comments + `forge.pii.amount` verified via boto3 GetTable | @@ -245,7 +246,7 @@ fields: | Capability | Reads from | |---|---| | Cross-account S3 access | `binding.governance.lakeFormation.grants[].principal` (LF block); a bucket-policy statement is paired with each grantee in another account (`bucketPolicy: cross-account`, the default; `none` / `all-grantees` in fluid-schema 0.7.6) | -| Cross-project BQ access | `metadata.policies` (existing) → dataset `access[]` via `_bq_access_entries` — ⚠️ **but see the note below: `metadata.policies` does not pass `fluid validate`** | +| Cross-project BQ access | `metadata.policies` (existing) → `google_bigquery_dataset_iam_member` (was the dataset's authoritative `access[]`) — ⚠️ **but see the note below: `metadata.policies` does not pass `fluid validate`** | | Glue catalog enrichment | `description` / `metadata.{description,layer,productType}` / `domain` / `fluidVersion` / `column.{description,tags}` | | Snowflake catalog enrichment | same set as Glue | diff --git a/docs/apply.md b/docs/apply.md index 093c03b4..5c4aeba9 100644 --- a/docs/apply.md +++ b/docs/apply.md @@ -192,7 +192,15 @@ builds: non-parquet format is CSV, which is what that provider writes). An AWS binding naming a `location.bucket` and `location.path` reads `s3:////*.`, the prefix the duckdb acquisition runner - writes into and the Glue table `fluid apply` declares for it. A warehouse table, a stream, or a GCS/Azure prefix is an + writes into and the Glue table `fluid apply` declares for it. A GCP + `bigquery_table` binding reads that table (`..`, the + one `fluid apply` created, whatever `gs://` path the binding also carries): + when the build runs, the table is read through the BigQuery API + (`tabledata.list` pages, streamed) into a Parquet file under the build's + `.fluid/staging//inputs/`, the view reads that file, and the file is + removed after the build. A BigQuery `TIMESTAMP` reads as a DuckDB `TIMESTAMP` + holding the UTC wall clock, which is what the same SQL reads on the local and + aws targets. Another warehouse's table, a stream, or a GCS/Azure prefix is an `UnreadableBindingError` naming the platform. A `{{ env.X }}` in the upstream binding with `X` unset is an error, not an empty string. * **Explicit inputs win.** A `properties.parameters.inputs` entry whose `name` @@ -222,6 +230,49 @@ unchanged. An expose declaring `policy.privacy.masking` is refused (`MaskingNotAppliedError`): this path does not apply masking, and cleartext must not land silently. +When the first expose's binding is a GCP `bigquery_table`, the result is +written as Parquet under `.fluid/staging//` and one load job moves it +into that table (`WRITE_TRUNCATE`, `CREATE_NEVER`, the table's own schema), +the load the duckdb acquisition runner performs; a `gs://` `location.path` on +the binding is never written to. A failed or short load fails the build. Any +other landing this path cannot write, a `gs://` or other non-S3 URI, a GCS +bucket, an Azure, Snowflake or Databricks binding, is refused +(`EmbeddedSqlLandingError`) before the SQL runs, instead of being written to a +local file of that name. The load is recorded as a run under +`.fluid/runs///runs/`, the way the acquisition load is, so +`fluid verify` holds the table's count to the rows it landed. + +Also refused before anything is read (`EmbeddedSqlLandingError`): + +* a landing that resolves to one of the build's own inputs: a BigQuery table + the build reads (names compared case-insensitively; a project left to the + client matches any), or an S3 object inside a prefix it reads. The load + replaces the table, so it would overwrite another product's rows, and no + `--allow-data-loss` is ever asked for a data write; +* a further expose named in the build's `outputs` and bound to a cloud store + or a warehouse (an aws, gcp, azure, snowflake or databricks binding, or any + remote URI): this path lands only the first expose. A further local expose + or output port is not written either, and the build prints a warning. + +When the contract declares `sovereignty` and the build reads or loads a +BigQuery table, the locations those reads and the load actually use are held +to it by `fluid validate`'s rules (`EmbeddedSqlSovereigntyError`): every +BigQuery binding must name its region (without one it is `US`, the IaC's +default), the landing must be outside `deniedRegions`, inside +`allowedRegions` and in the declared `jurisdiction`, and, with `dataResidency` +and no `crossBorderTransfer` (the schema's defaults), every input must be in +the landing's jurisdiction. BigQuery's `EU` and `US` multi-regions count as +EU and US. `enforcementMode: strict` (the default) refuses, `advisory` warns, +`audit` logs. + +BigQuery reads and loads authenticate with Application Default Credentials +(gcloud ADC, an attached service account, or a Workload Identity Federation +`external_account` file in `GOOGLE_APPLICATION_CREDENTIALS`). With +`BIGQUERY_EMULATOR_HOST` set they go to that emulator with anonymous +credentials, so no real token is sent to it. A load job that reports no row +count (the goccy emulator's never do) is checked by counting the table after +the load, never assumed. + ### Skipped builds are not success If every build in a build-augmented mode was skipped — a missing dbt diff --git a/docs/governance-parity.md b/docs/governance-parity.md new file mode 100644 index 00000000..0e37d64c --- /dev/null +++ b/docs/governance-parity.md @@ -0,0 +1,240 @@ +# Governance parity: one contract, AWS and GCP + +A contract declares its governance once, in fields that name no cloud. Each +environment's overlay patches only `exposes[].binding`: the platform, the location, +and which real identities the contract's logical principals are on that cloud. +`fluid apply` then emits that cloud's own OpenTofu resources, and `fluid verify` +checks the live platform against the same derivation apply emitted from. + +The fields are in fluid-schema **0.7.6** (the preview). 0.7.5 GA is unchanged, so a +contract needs `fluidVersion: "0.7.6"` to validate with them. + +## The parity table + +| Policy (contract field) | AWS: what `fluid apply` emits | GCP: what `fluid apply` emits | What `fluid verify` checks | +|---|---|---|---| +| **Retention**: `exposes[].lifecycle {retention, expire: true}` | One S3 lifecycle rule per expose, filtered to the binding's prefix, expiring objects `retention` after they are written (`aws_s3_bucket_lifecycle_configuration`). | Daily partitions that expire `retention` after their day ends: `google_bigquery_table.time_partitioning {type: DAY, expiration_ms}`, by ingestion time, or on `binding.location.partitionBy` when it names one date or timestamp column (see "Retention counts from the partition's date" below). Never a table expiration, which would delete the product. | AWS: the enabled prefix rule's `Expiration.Days`, and no rule that expires sooner. GCP: the table's partition type, field and `expirationMs`, and that the table itself has no `expirationTime`. | +| **Encryption at rest**: `binding.encryption.kms` | `product`: a KMS key per bucket (`aws_kms_key`, rotation on, alias `alias/fluid//`) as the bucket's default SSE-KMS. `alias/...` or an ARN: that key. `none`: SSE-S3. | `product`: a key ring and key per dataset (`google_kms_key_ring` `fluid--`, `google_kms_crypto_key` `bigquery`, 90-day rotation) in the dataset's location, `roles/cloudkms.cryptoKeyEncrypterDecrypter` for the BigQuery service agent (`google_kms_crypto_key_iam_member`), used as the dataset's `default_encryption_configuration` and the table's `encryption_configuration`. `projects/.../cryptoKeys/...`: that key. `none`: Google-managed keys. Every table of one dataset must declare the same key (see "One dataset, one key"). | AWS: the key is `Enabled`, and the objects under the prefix are SSE-KMS with it. GCP: `kmsKeyName` of the table, and of the dataset's default when the product owns the dataset. | +| **Column restrictions**: `exposes[].policy.authz.columnRestrictions` | Each Lake Formation grant with `SELECT` excludes the restricted columns its principal may not read (`aws_lakeformation_permissions.table_with_columns.excluded_column_names`, with `wildcard`). | A Data Catalog taxonomy per product and dataset with fine-grained access control (`google_data_catalog_taxonomy`), a policy tag per set of restricted columns that share their readers (`google_data_catalog_policy_tag`), attached through the table schema's `policyTags`, and `roles/datacatalog.categoryFineGrainedReader` for exactly the allowed readers (`google_data_catalog_policy_tag_iam_member`). The restrictions' `tags` and `labels` go into the policy tag's description. | AWS: `lakeformation:ListPermissions` on tables; no `SELECT` of a denied principal, of a read grantee, or of `IAM_ALLOWED_PRINCIPALS` reaches a column it may not read, including grants made outside the contract. GCP: every restricted column carries a tag, the tag's fine-grained readers are exactly the derived set (a denied principal, or one granted outside the contract, fails), and the taxonomy enforces fine-grained access control. | +| **Access grants**: `accessPolicy.grants[]` | Not emitted: on AWS, access is the binding's `governance.lakeFormation.grants`. `fluid validate` warns only for an aws binding with no Lake Formation grants, where the contract's access intent is unenforced. | One non-authoritative `google_bigquery_dataset_iam_member` per role and member (and `google_storage_bucket_iam_member` for GCS), with the logical principal mapped to its GCP identity. | Not checked yet on either cloud. | + +A policy that a binding cannot apply is refused at `fluid validate`, `fluid plan` and +`fluid apply`, never dropped: retention, a key or a column restriction on a GCP +binding that is not a BigQuery table (GCS, Pub/Sub, Iceberg storage), a column +restriction on an AWS binding with no Lake Formation grants or on a non-Glue format, +an AWS key reference on GCP and a Cloud KMS key name on AWS. + +## Logical principals and `binding.principals` + +The base contract names principals as the business knows them: + +```yaml +accessPolicy: + grants: + - principal: group:data-platform@northwind.example + permissions: [read, select, query] +exposes: + - exposeId: candidates + policy: + authz: + columnRestrictions: + - principal: group:analysts@northwind.example + columns: [customer_id, msisdn] + access: deny +``` + +Each environment's overlay maps them, in the binding, to the identities they are on +that cloud: + +```yaml +# overlays/gcp.yaml +exposes: + - binding: + platform: gcp + principals: + group:data-platform@northwind.example: group:data-platform@northwind.com + group:analysts@northwind.example: group:analysts@northwind.com + serviceAccount:fluid-pipeline@northwind.example: >- + serviceAccount:fluid-pipeline@northwind-demo.iam.gserviceaccount.com + +# overlays/aws.yaml +exposes: + - binding: + platform: aws + principals: + group:analysts@northwind.example: arn:aws:iam::123456789012:role/fluid-demo-lab-analyst +``` + +* A value is one identity, a list of them, or `[]` for "no identity on this cloud" + (nothing is granted to it there). +* With `binding.principals` present, every principal the contract names for the + expose must be mapped; an unmapped one is refused (`principal-unmapped`). +* Without it, principals are used as written, as before, except that on GCP a + placeholder is refused (`principal-placeholder`): a principal in a reserved + top-level domain (`.example`, `.test`, `.invalid`, `.localhost`), and one that is + not an IAM member at all (`group:data-platform` with no domain, a bare `analysts`, + an unknown prefix such as `role:analyst`, or an unfilled `<>`). + BigQuery and Cloud Storage refuse either at apply, so none was ever a working grant. +* On AWS an unmapped restriction principal must already be an IAM ARN. + +The pattern follows ODCS v3, which declares `roles[]` once and binds them per server +(`servers[].roles`), and dbt `grants`, which resolve the grantee per target. + +## Column restriction semantics + +* A column named in any restriction is restricted. +* `deny`: the principal may not read the columns. +* `allow`: the columns are readable only by the principals an `allow` names. +* A deny beats an allow. A restriction never grants access: the readers are the + expose's readers. On GCP they are the `accessPolicy` read grantees and the + expose's own `policy.authz.readers` (whose table access is managed elsewhere, so + only the fine-grained reader role on the tag is granted to them); on AWS, the Lake + Formation `SELECT` grantees. An allowed principal that is not a reader is logged, + not added. +* A restriction on an expose with no reader is refused on both clouds: on AWS when + the binding has no Lake Formation grant (`column-restriction-unenforceable`), on + GCP when there is no read grant and no `policy.authz.readers` + (`column-restriction-no-readers`). The policy tag would otherwise lock the columns + for everyone, not only for the principals the restrictions name. +* A restriction's `tags` and `labels` are descriptive. On GCP they are written into + the policy tag's description; Lake Formation permissions have no field for them. +* On AWS a grant's hand-written `excludedColumns` keeps working. With a restriction + on the same expose the two must agree, or the emit is refused + (`column-restriction-conflict`). + +On GCP a denied principal gets an access error on the restricted columns; +`SELECT * EXCEPT (customer_id, msisdn)` still works for it. BigQuery dynamic data +masking (`google_bigquery_datapolicy_data_policy`, SHA-256 or nullify, for principals +holding `roles/bigquerydatapolicy.maskedReader`) is not emitted yet. It needs a third +principal set (who sees masked values rather than an error) that the contract has no +field for, and a column masked at landing (`policy.privacy.masking`) would be hashed +twice. + +## Dataset grants are no longer authoritative + +The grants were an authoritative `access` list on the dataset: it replaced every +entry the dataset had, including the ones BigQuery gives a new dataset (the project's +owners, writers and readers, and its creator), and anything granted elsewhere. They +are `google_bigquery_dataset_iam_member` resources now, which add their own binding +and leave the rest. The consequences: + +* A principal holding a basic role on the project (Viewer, Editor, Owner) keeps the + dataset access BigQuery's default entries give it. Restricted columns stay + protected by their policy tags whatever the dataset grants; keep basic roles off + projects that hold restricted data. +* A grant made outside the contract is no longer removed by the next apply. + `fluid verify` reports a fine-grained reader granted outside the contract; dataset + grants are not verified yet. +* The provider rewrites the dataset's access list without authorized-view entries + when it adds a member; forge-cli emits none. +* Each member resource is named from its role and member plus a hash of both, so + principals that differ only in `.`, `-` or `_` keep one grant each. + +### The first apply after upgrading revokes what the old list held + +Once the module stops setting `access`, the provider keeps it as Computed: an entry +the old authoritative list held and no member resource covers (a principal removed +in the same change, or a logical principal `binding.principals` now maps to another +identity) would stay on the dataset, unmanaged, and no plan would show it +(terraform-provider-google issue 8165). So the first `fluid apply` on a dataset whose +state holds an access list and no member resource sets `access`, for that one apply, +to the list less those entries. The provider revokes them and the member resources +are created after it; the apply prints what it revoked. For that one apply the list +is authoritative, as it was on every apply before: it is the list state recorded at +the last apply, so an entry added to the dataset by hand since then is removed, as +the old module removed it. The next apply finds the +member resources in state and leaves `access` unset. Only entries of the old +emitter's shape (a role and one user, group or domain) are revoked; special groups, +views and routines are kept. `fluid diff` makes the same change, so its plan shows +the revocation. A dataset whose every entry would be revoked cannot be narrowed that +way (an empty `access` plans nothing), and the apply is refused with the entries to +revoke by hand. + +### Revoking a grant is not data loss + +The data-loss gate refuses a plan that destroys a resource unless `--allow-data-loss` +is set. It counts only resources that hold data or policy: removing a +`google_bigquery_dataset_iam_member`, `google_bigquery_table_iam_member`, +`google_storage_bucket_iam_member`, `google_data_catalog_policy_tag_iam_member`, +`google_data_catalog_policy_tag`, `google_data_catalog_taxonomy` or an +`aws_lakeformation_permissions` revokes access and deletes nothing, so a revoked +reader or a lifted restriction applies without the flag. The apply lists them. A +key's IAM grant stays gated: without it BigQuery cannot decrypt the table. When the +plan's per-resource events do not account for every removal, every removal counts. + +## One dataset, one key + +BigQuery gives a table created without a key the dataset's default key. A table +declared unkeyed in a dataset another expose keys therefore gets the key anyway, the +provider plans removing it, and because `encryption_configuration` is ForceNew every +later plan replaces the table (terraform-provider-google issue 26193). Tables of one +dataset that declare different encryption are refused +(`encryption-kms-mixed-dataset`). A view stores no rows and carries no key; a keyed +view sets the dataset's default and must agree too. The dataset's default key no +longer depends on which expose comes first. + +## Retention counts from the partition's date + +BigQuery deletes a partition `retention` after the partition's own date, not after +its rows were written. Partitioned by ingestion time (no `partitionBy`, the default), +the date is the day the rows landed, so no row is deleted sooner than `retention` +after it was written, as with the S3 rule. With `binding.location.partitionBy`, the +date is the column's value: retention is the age of the event, and a backfill of rows +whose date is already older than `retention` lands in expired partitions and is +deleted at once. Naming the column is the opt-in to that; `fluid apply` logs it +(`bigquery_retention_event_time`). + +## Changes a live table cannot take in place + +BigQuery cannot partition an existing table, and the provider replaces a table whose +key changes (`encryption_configuration` is ForceNew). So the first apply that adds +`expire: true`, or a key, to a table that exists plans the table's **replacement**, +and `fluid apply` refuses it without `--allow-data-loss`. The next build lands the +data again. + +For partitioning, the replacement is made explicit: the provider plans adding +ingestion-time partitioning as an in-place update, which BigQuery then refuses, so +the table's `lifecycle.replace_triggered_by` names a `terraform_data` holding the +partitioning's shape. Changing `retention` later is an in-place update of +`expiration_ms`. Removing `expire` removes that trigger, which the data-loss gate +also refuses without `--allow-data-loss`; BigQuery cannot un-partition a table, so to +keep data longer, set a longer `retention` instead. + +## Prerequisites + +* The Cloud KMS API (`cloudkms.googleapis.com`) and the Data Catalog API + (`datacatalog.googleapis.com`) enabled on the project, for keys and policy tags. +* The identity running `fluid apply` needs, beyond BigQuery: `roles/cloudkms.admin` + (create key rings and keys, set their IAM), `roles/datacatalog.categoryAdmin` + (taxonomies, tags and their IAM), and `bigquery.datasets.update` on the datasets + (dataset IAM members). +* A key ring and a crypto key cannot be deleted on GCP. `tofu destroy` removes them + from state and schedules the key's versions for destruction (30 days by default); + the next apply adopts the same names (`discover_imports`). +* `fluid verify`'s column check calls the Data Catalog API with Application Default + Credentials, and needs `datacatalog.taxonomies.get` and + `datacatalog.taxonomies.getIamPolicy`. +* On AWS, `fluid verify`'s Lake Formation check lists the table permissions the + caller can see, so it must run as a Lake Formation administrator. When the + contract's own grants are not in the listing, it reports an error rather than a + pass. + +## What is proven, and what is not + +* `tofu validate` accepts every governed module shape (`tests/iac/test_iac_gcp_governance.py`). +* A real `tofu plan` and `apply` with terraform-provider-google, against an + in-process stand-in for the BigQuery REST API, shows adding retention or a key to a + live table plans its replacement (the data-loss gate refuses it), a new retention is + in place, the governed table then plans clean, and the dataset's own access entries + survive the grants (`tests/iac/test_iac_gcp_governance_plan.py`). +* A real `tofu plan` against moto accepts the Lake Formation grants with excluded + columns, and `fluid verify`'s Lake Formation check runs against moto's stored grants + (`tests/iac/test_iac_aws_column_restrictions.py`). +* The same stand-in shows a revoked reader plans one member destroy that the gate + lets through, and that moving from the authoritative access list of 0.16.6 and earlier to member + resources revokes a grant removed in the same change and then plans clean. +* **Not proven**: anything against real BigQuery, Cloud KMS, Data Catalog or Lake + Formation. No emulator enforces IAM, policy tags or keys: that a denied principal's + query is refused, that the BigQuery service agent can use the key, and that a load + into a policy-tagged column succeeds for the pipeline's identity rest on the + providers' documentation. diff --git a/docs/verify-gcp-bigquery.md b/docs/verify-gcp-bigquery.md new file mode 100644 index 00000000..493446c1 --- /dev/null +++ b/docs/verify-gcp-bigquery.md @@ -0,0 +1,65 @@ +# `fluid verify` on a GCP BigQuery binding + +An expose bound like this is provisioned by `fluid apply` as a BigQuery dataset +and table, and the build loads its rows into the table (an acquisition build +through the duckdb runner, an embedded-SQL build on DuckDB through the same +load): + +```yaml +binding: + platform: gcp + format: bigquery_table + location: + project: northwind-demo + dataset: demo_bronze + table: customer_subscriptions + region: europe-west1 +``` + +The table verify reads is named the way the load names it +(`build_runners/_bigquery_load.py::bigquery_load_target`): `{{ env.X }}` in the +binding is resolved as `fluid apply` resolves it, and a binding with no +`project` uses `GOOGLE_PROJECT` / `GOOGLE_CLOUD_PROJECT` / `GCLOUD_PROJECT` / +`CLOUDSDK_CORE_PROJECT`, then the client's own (Application Default +Credentials'), exactly as the load does. With none of them, verify is an error +naming the table, not a lookup of `..
`. + +| Check | How | Result when it fails | +|---|---|---| +| The table exists | `tables.get` | error (always exit 1); in a reference-only contract, INFO | +| Columns match the contract | the table's schema against `contract.schema`, types folded through the type map `fluid apply` declares them with | missing column / changed type: CRITICAL; extra column: INFO | +| Required / nullable | the table's field modes | WARNING | +| Location | the dataset's location against the binding's region | CRITICAL | +| It serves what the build loaded | `SELECT COUNT(*)` of the table, against the run records of the build that writes the expose, acquisition or embedded SQL (below) | CRITICAL | +| It is not empty | a count of 0 | CRITICAL; in a reference-only contract with no run of its own, INFO | +| Masked columns landed treated | for an expose with `policy.privacy.masking`, the same query counts each masked column's non-null values that lack their strategy's shape: `COUNTIF(col IS NOT NULL AND NOT REGEXP_CONTAINS(CAST(col AS STRING), @shape))`, the shape anchored `\A(?:...)\z` and bound as a query parameter | one such value: CRITICAL. Only counts leave the query, never a value | + +The count query needs `bigquery.jobs.create` on the project +(`roles/bigquery.jobUser`) as well as read access to the table; a query that +cannot run is an error, not a pass. `metadata.num_rows` in the report is the +counted rows (the table's own `numRows`, which leaves out the streaming buffer +and which the goccy emulator leaves unset, is kept as `table_num_rows`). + +### Which run the table is held to + +The rules of the Athena verifier (`docs/verify-aws-athena.md`), with one +difference: a run is this table's when its record's `facets.bigquery_load` +names the table, which the duckdb runner writes after the load, with the rows +the load job (or, on an emulator, the count after it) says arrived. A run that +succeeded with no BigQuery load landed somewhere else (a local or aws run from +the same contract directory) and is passed over. The newest run that may be +this table's is compared: equal for `full_refresh`, at least the sum since the +last full load for `incremental_append`. A failed run, or one whose record +does not parse, is not a count to hold the table to, and the count is then +reported without a gate. An embedded-SQL build on DuckDB that loads a BigQuery +table records its run the same way (`facets.bigquery_load`, `full_refresh`, +the rows the load landed), so a silver or gold table is held to its last load +exactly as bronze is. A failed embedded-SQL run is recorded without +`bigquery_load`, and the count is reported without a gate until the next run +succeeds. + +### Emulators + +With `BIGQUERY_EMULATOR_HOST` set, the client goes to that host with anonymous +credentials. The goccy bigquery-emulator runs the count query, the query +parameters and `REGEXP_CONTAINS`; it does not report `numRows`. diff --git a/examples/aws-glue-data-lake/README.md b/examples/aws-glue-data-lake/README.md index a7e306e8..39962738 100644 --- a/examples/aws-glue-data-lake/README.md +++ b/examples/aws-glue-data-lake/README.md @@ -16,6 +16,12 @@ End-to-end data lake on AWS using Glue for cataloging, ETL, and Iceberg tables. - AWS account with Glue, S3, and IAM permissions - `fluid` CLI installed (`pip install data-product-forge`) - AWS credentials configured (`aws configure` or env vars) +- `DATA_LAKE_BUCKET`, `AWS_REGION` and `AWS_ACCOUNT_ID` set: the contracts + read them as `{{ env.* }}` +- For `contract-database` and `contract-iceberg`, a Lake Formation + administrator to apply them: their `accessPolicy` is enforced on AWS by + the Lake Formation grants in the binding (the AWS emitter does not write + `accessPolicy` itself), on a location registered with Lake Formation ## Quick Start diff --git a/examples/aws-glue-data-lake/contract-database.fluid.yaml b/examples/aws-glue-data-lake/contract-database.fluid.yaml index c07f6711..38fdd45e 100644 --- a/examples/aws-glue-data-lake/contract-database.fluid.yaml +++ b/examples/aws-glue-data-lake/contract-database.fluid.yaml @@ -1,4 +1,4 @@ -fluidVersion: "0.7.1" +fluidVersion: "0.7.5" kind: DataProduct id: analytics.data_lake.sales_db name: Sales Data Lake (Glue Database + Table) @@ -47,6 +47,21 @@ exposes: path: curated/sales/transactions/ region: "{{ env.AWS_REGION }}" + # accessPolicy is the cloud-neutral intent, and the AWS emitter does + # not write it: on AWS these Lake Formation grants enforce it, one per + # accessPolicy grant (read/select -> SELECT + DESCRIBE, write/create -> + # INSERT, delete -> DELETE), on a location registered with Lake Formation. + # Set AWS_ACCOUNT_ID to the account that owns the roles before + # generating or applying the module. + governance: + lakeFormation: + registerLocation: true + grants: + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/data-analyst" + permissions: [SELECT, DESCRIBE] + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/data-engineer" + permissions: [SELECT, INSERT, DELETE, DESCRIBE] + contract: schema: - name: order_id diff --git a/examples/aws-glue-data-lake/contract-iceberg.fluid.yaml b/examples/aws-glue-data-lake/contract-iceberg.fluid.yaml index 9da86208..fe02bda6 100644 --- a/examples/aws-glue-data-lake/contract-iceberg.fluid.yaml +++ b/examples/aws-glue-data-lake/contract-iceberg.fluid.yaml @@ -1,4 +1,4 @@ -fluidVersion: "0.7.1" +fluidVersion: "0.7.5" kind: DataProduct id: analytics.data_lake.sales_iceberg name: Sales Iceberg Table @@ -49,6 +49,23 @@ exposes: path: iceberg/sales/curated/ region: "{{ env.AWS_REGION }}" + # accessPolicy is the cloud-neutral intent, and the AWS emitter does + # not write it: on AWS these Lake Formation grants enforce it, one per + # accessPolicy grant (read/select/query -> SELECT + DESCRIBE, write/create + # -> INSERT, update/delete -> DELETE: an Iceberg UPDATE or MERGE rewrites + # data files, so it needs INSERT and DELETE), on a location registered + # with Lake Formation. + # Set AWS_ACCOUNT_ID to the account that owns the roles before + # generating or applying the module. + governance: + lakeFormation: + registerLocation: true + grants: + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/data-analyst" + permissions: [SELECT, DESCRIBE] + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/data-engineer" + permissions: [SELECT, INSERT, DELETE, DESCRIBE] + iceberg: writeFormat: parquet snapshotRetention: diff --git a/examples/aws-iceberg-lakehouse/README.md b/examples/aws-iceberg-lakehouse/README.md index 86c8cf68..5f6a128f 100644 --- a/examples/aws-iceberg-lakehouse/README.md +++ b/examples/aws-iceberg-lakehouse/README.md @@ -15,6 +15,9 @@ Part of the **[FLUID examples](../README.md)**. · AWS provider · offline-revie - An `iceberg:` management block — snapshot retention (bounded time-travel history) and compaction (keeps the small-files problem in check). Values follow community norms: 5-day retention, 256 MB target file size. +- Lake Formation grants in the binding that enforce the `accessPolicy` on + AWS: SELECT for the analysts, and SELECT, INSERT and DELETE for the ETL + role, which an Iceberg `UPDATE`, `DELETE` or `MERGE` needs. ## Prerequisites @@ -40,6 +43,15 @@ cat /tmp/aws-iceberg/main.tf.json | `aws_glue_catalog_database.…sales` | Glue Data Catalog | The `sales` database | | `aws_glue_catalog_table.…orders` | Glue Data Catalog | Iceberg `orders` table (`table_type = ICEBERG`) | | `aws_s3_bucket.…acme_sales_lakehouse` | Amazon S3 | Backing object store | +| `aws_lakeformation_resource.…` | Lake Formation | Registers the table's S3 prefix | +| `aws_lakeformation_permissions.…` | Lake Formation | Two grants: the `accessPolicy` reader and writer, as IAM roles | +| `aws_s3_bucket_policy.…` | Amazon S3 | Direct-read statements for grantees in another account only; with the example's same-account roles it plans `count = 0` | + +The grant ARNs read `{{ env.AWS_ACCOUNT_ID }}`. Set it to the account that +owns the roles before `fluid generate iac` if you will run `tofu validate` +or `plan` on the module, and before `fluid apply`: left unset, the ARN stays +unresolved and the AWS provider refuses it. Applying the Lake Formation +resources needs a Lake Formation administrator. Confirm the Iceberg wiring in the emitted module: diff --git a/examples/aws-iceberg-lakehouse/contract.fluid.yaml b/examples/aws-iceberg-lakehouse/contract.fluid.yaml index 5fdfd0c6..70d8b53d 100644 --- a/examples/aws-iceberg-lakehouse/contract.fluid.yaml +++ b/examples/aws-iceberg-lakehouse/contract.fluid.yaml @@ -54,6 +54,23 @@ exposes: path: iceberg/sales/orders/ region: us-east-1 + # accessPolicy is the cloud-neutral intent, and the AWS emitter does + # not write it: on AWS these Lake Formation grants enforce it, one per + # accessPolicy grant (read/select -> SELECT + DESCRIBE, write/create -> + # INSERT, update/delete -> DELETE: an Iceberg UPDATE, DELETE or MERGE + # rewrites data files, so it needs INSERT and DELETE), on a location + # registered with Lake Formation. + # Set AWS_ACCOUNT_ID to the account that owns the roles before + # generating or applying the module. + governance: + lakeFormation: + registerLocation: true + grants: + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/analysts" + permissions: [SELECT, DESCRIBE] + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/orders-etl" + permissions: [SELECT, INSERT, DELETE, DESCRIBE] + # AWS Glue Iceberg table management: snapshot retention keeps # time-travel history bounded, compaction keeps the small-files # problem in check. Values follow community norms (5-day retention, diff --git a/examples/aws-medallion-lake/README.md b/examples/aws-medallion-lake/README.md index 777dc616..a03a7541 100644 --- a/examples/aws-medallion-lake/README.md +++ b/examples/aws-medallion-lake/README.md @@ -16,6 +16,9 @@ Part of the **[FLUID examples](../README.md)**. · AWS provider · offline-revie **Parquet** in the `curated/` prefix, with inline quality assertions. - Both zones are cataloged in the Glue Data Catalog (separate `iot_bronze` / `iot_silver` databases) so Athena can query either. +- One `accessPolicy` for the product, enforced on AWS by Lake Formation + grants in each zone's binding: the analysts read both zones, the ingest + role writes them. The raw shape is illustrated by [`sample_raw_readings.csv`](sample_raw_readings.csv) (the Bronze objects on S3 look like this — FLUID does not load it; it is here to @@ -47,6 +50,15 @@ cat /tmp/aws-medallion/main.tf.json | `aws_glue_catalog_database.…iot_silver` | Glue Data Catalog | Silver database | | `aws_glue_catalog_table.…sensor_readings` | Glue Data Catalog | Silver curated table (Parquet) | | `aws_s3_bucket.…acme_iot_lake` | Amazon S3 | Shared object store (raw/ + curated/ prefixes) | +| `aws_lakeformation_resource.…` | Lake Formation | Registers each zone's S3 prefix (raw/ and curated/) | +| `aws_lakeformation_permissions.…` | Lake Formation | Four grants, two per zone: the `accessPolicy` reader and writer, as IAM roles | +| `aws_s3_bucket_policy.…` | Amazon S3 | Direct-read statements for grantees in another account only; with the example's same-account roles it plans `count = 0` | + +The grant ARNs read `{{ env.AWS_ACCOUNT_ID }}`. Set it to the account that +owns the roles before `fluid generate iac` if you will run `tofu validate` +or `plan` on the module, and before `fluid apply`: left unset, the ARN stays +unresolved and the AWS provider refuses it. Applying the Lake Formation +resources needs a Lake Formation administrator. ## Architecture diff --git a/examples/aws-medallion-lake/contract.fluid.yaml b/examples/aws-medallion-lake/contract.fluid.yaml index 6943066f..8d2b6ebd 100644 --- a/examples/aws-medallion-lake/contract.fluid.yaml +++ b/examples/aws-medallion-lake/contract.fluid.yaml @@ -53,6 +53,21 @@ exposes: path: raw/iot/sensor_readings/ region: us-east-1 + # accessPolicy is the cloud-neutral intent, and the AWS emitter does + # not write it: on AWS each zone's Lake Formation grants enforce it, + # one per accessPolicy grant (read/select -> SELECT + DESCRIBE, + # write/create -> INSERT), on the zone's prefix registered with Lake + # Formation. Set AWS_ACCOUNT_ID to the account that owns the roles + # before generating or applying the module. + governance: + lakeFormation: + registerLocation: true + grants: + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/analysts" + permissions: [SELECT, DESCRIBE] + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/iot-ingest" + permissions: [SELECT, INSERT, DESCRIBE] + contract: schema: - name: device_id @@ -92,6 +107,16 @@ exposes: path: curated/iot/sensor_readings/ region: us-east-1 + # The same accessPolicy, enforced on the Silver prefix. + governance: + lakeFormation: + registerLocation: true + grants: + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/analysts" + permissions: [SELECT, DESCRIBE] + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/iot-ingest" + permissions: [SELECT, INSERT, DESCRIBE] + contract: schema: - name: device_id diff --git a/examples/aws-s3-glue-athena/README.md b/examples/aws-s3-glue-athena/README.md index 86433edd..09d587f0 100644 --- a/examples/aws-s3-glue-athena/README.md +++ b/examples/aws-s3-glue-athena/README.md @@ -12,8 +12,10 @@ Part of the **[FLUID examples](../README.md)**. · AWS provider · offline-revie catalog natively — an **Athena-queryable** table with no extra resource. - **Parquet** as the on-disk format, the single biggest lever on Athena scan cost (columnar + compressed ≈ an order of magnitude cheaper than raw text). -- An `accessPolicy` block that downstream compiles into IAM / Lake Formation - grants, and a `contract.quality[]` block of inline SQL assertions. +- An `accessPolicy` block, the cloud-neutral statement of who reads and who + writes, enforced on AWS by the binding's `governance.lakeFormation` grants + (the AWS emitter does not write `accessPolicy` itself), and a + `contract.quality[]` block of inline SQL assertions. ## Prerequisites @@ -37,16 +39,23 @@ cat /tmp/aws-lake/main.tf.json ## What gets generated -`fluid generate iac` emits a credential-free `main.tf.json` with three resources: +`fluid generate iac` emits a credential-free `main.tf.json` with these resources: | Resource | AWS service | Purpose | |----------|-------------|---------| | `aws_glue_catalog_database.…web_analytics` | Glue Data Catalog | The `web_analytics` database | | `aws_glue_catalog_table.…pageviews` | Glue Data Catalog | The `pageviews` table + column schema | | `aws_s3_bucket.…acme_web_analytics_lake` | Amazon S3 | Backing object store (curated zone) | +| `aws_lakeformation_resource.…` | Lake Formation | Registers the table's S3 prefix | +| `aws_lakeformation_permissions.…` | Lake Formation | Two grants: the `accessPolicy` reader and writer, as IAM roles | +| `aws_s3_bucket_policy.…` | Amazon S3 | Direct-read statements for grantees in another account only; with the example's same-account roles it plans `count = 0` | Apply it with OpenTofu (`tofu init && tofu apply`) once you have AWS -credentials — or apply through FLUID with `fluid apply`. +credentials — or apply through FLUID with `fluid apply`. Set `AWS_ACCOUNT_ID` +to the account that owns the roles before generating a module you will +validate, plan or apply: the grant ARNs read it (`{{ env.AWS_ACCOUNT_ID }}`), +and left unset the ARN stays unresolved and the AWS provider refuses it. The +Lake Formation resources need a Lake Formation administrator to apply them. ## Architecture diff --git a/examples/aws-s3-glue-athena/contract.fluid.yaml b/examples/aws-s3-glue-athena/contract.fluid.yaml index e85d78ce..26f24204 100644 --- a/examples/aws-s3-glue-athena/contract.fluid.yaml +++ b/examples/aws-s3-glue-athena/contract.fluid.yaml @@ -22,7 +22,7 @@ metadata: team: web-analytics email: web-analytics@example.com -# ─── Access Policy (compiled into IAM / Lake Formation grants) ───── +# ─── Access Policy (enforced on AWS by the binding's Lake Formation grants) ─ accessPolicy: grants: @@ -56,6 +56,21 @@ exposes: path: curated/web_analytics/pageviews/ region: us-east-1 + # accessPolicy is the cloud-neutral intent, and the AWS emitter does + # not write it: on AWS these Lake Formation grants enforce it, one per + # accessPolicy grant (read/select -> SELECT + DESCRIBE, write/create -> + # INSERT), on a location registered with Lake Formation. + # Set AWS_ACCOUNT_ID to the account that owns the roles before + # generating or applying the module. + governance: + lakeFormation: + registerLocation: true + grants: + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/analysts" + permissions: [SELECT, DESCRIBE] + - principal: "arn:aws:iam::{{ env.AWS_ACCOUNT_ID }}:role/etl" + permissions: [SELECT, INSERT, DESCRIBE] + contract: schema: - name: event_id diff --git a/examples/bitcoin-price-api-declarative-part-b/README.md b/examples/bitcoin-price-api-declarative-part-b/README.md index 1412b1c7..ba9bf76b 100644 --- a/examples/bitcoin-price-api-declarative-part-b/README.md +++ b/examples/bitcoin-price-api-declarative-part-b/README.md @@ -251,7 +251,6 @@ bq show --format=prettyjson <>:crypto_data.bitcoin_prices | j **What does NOT get created automatically (requires additional configuration)**: - IAM policies (requires `policy-compile` + manual Terraform apply) -- Column-level access control (requires BigQuery Policy Tags - paid tier) - Row-level security (requires BigQuery authorized views - paid tier) - Data masking (requires BigQuery DLP - paid tier) @@ -450,8 +449,29 @@ policy: - Supports compliance requirements (e.g., GDPR Article 25 - data minimization) **Implementation:** -- Free tier: Document only (not enforced) -- Paid tier: Use BigQuery Policy Tags + Data Catalog +- `fluid apply` writes a Data Catalog taxonomy and a policy tag on the two + columns, and grants the tag's fine-grained reader role to the expose's + readers, never to the interns (needs the Data Catalog API; see + [governance parity](../../docs/governance-parity.md)). + +**Mapping the principals:** the readers and the interns group are logical +names. The binding maps each to the identity it is on GCP, and once +`binding.principals` is present every principal the expose names must be +mapped (fluid-schema 0.7.6): + +```yaml +binding: + platform: gcp + principals: + group:data-analytics@company.com: group:data-analytics@example.com + "serviceAccount:looker@<>.iam.gserviceaccount.com": serviceAccount:looker@example.com + group:interns@company.com: group:interns@example.com + # ...one entry per reader +``` + +The `example.com` identities are placeholders to replace with your own. An +unmapped `<>` principal is refused at `fluid validate`, +since BigQuery would refuse it at apply. ### 5. Privacy Controls (Row-Level Security) diff --git a/examples/bitcoin-price-api-declarative-part-b/contract.fluid.yaml b/examples/bitcoin-price-api-declarative-part-b/contract.fluid.yaml index 31fec25a..cd33a367 100644 --- a/examples/bitcoin-price-api-declarative-part-b/contract.fluid.yaml +++ b/examples/bitcoin-price-api-declarative-part-b/contract.fluid.yaml @@ -1,4 +1,4 @@ -fluidVersion: "0.7.5" +fluidVersion: "0.7.6" kind: DataProduct id: crypto.bitcoin_prices_gcp_governed name: Bitcoin Price Index (FLUID Declarative + Governance) @@ -87,6 +87,19 @@ exposes: dataset: crypto_data table: bitcoin_prices region: europe-west3 # GDPR-compliant EU region + + # The readers and the restricted principal under policy.authz are + # logical names. binding.principals (fluid-schema 0.7.6) maps each to + # the identity it is on this GCP binding; once the block is present, + # fluid refuses any of them it does not map. The example.com + # identities are documentation placeholders: replace them with your + # own groups and service account, as you replace <>. + principals: + group:data-analytics@company.com: group:data-analytics@example.com + group:finance-team@company.com: group:finance-team@example.com + group:trading-desk@company.com: group:trading-desk@example.com + "serviceAccount:looker@<>.iam.gserviceaccount.com": serviceAccount:looker@example.com + group:interns@company.com: group:interns@example.com # GOVERNANCE POLICIES (schema-compliant) policy: @@ -108,7 +121,9 @@ exposes: - serviceAccount:data-pipeline@<>.iam.gserviceaccount.com - group:data-engineering@company.com - # Column-level access control (example - not enforced in free tier) + # Column-level access control: fluid apply enforces it with a Data + # Catalog policy tag the readers above may read and interns may not + # (needs the Data Catalog API; see docs/governance-parity.md) columnRestrictions: - principal: "group:interns@company.com" columns: diff --git a/fluid_build/build_runners/_bigquery_load.py b/fluid_build/build_runners/_bigquery_load.py index 4359bdb7..9b4f7752 100644 --- a/fluid_build/build_runners/_bigquery_load.py +++ b/fluid_build/build_runners/_bigquery_load.py @@ -29,6 +29,25 @@ ``google.cloud.bigquery`` is imported only when a load runs, so installs without the ``gcp`` extra are unaffected until they bind a BigQuery table. + +**Timestamps.** DuckDB writes a ``TIMESTAMP`` column as a Parquet timestamp +with ``isAdjustedToUTC=false``, which BigQuery reads as ``DATETIME``, not +``TIMESTAMP``. Before the load, every column the table declares ``TIMESTAMP`` +that the file holds as a naive timestamp is rewritten as ``TIMESTAMPTZ`` under +``TimeZone='UTC'``, so the Parquet column is UTC-adjusted and loads as the +table's ``TIMESTAMP``: the mapping python-bigquery itself uses +(``_pyarrow_helpers``: ``TIMESTAMP`` is ``pyarrow.timestamp("us", tz="UTC")``, +``DATETIME`` the naive one). The wall-clock value is read as UTC, which is how +BigQuery reads a zone-less timestamp literal too. + +**Emulators.** With ``BIGQUERY_EMULATOR_HOST`` set, python-bigquery already +sends every request to that host, but it still resolves Application Default +Credentials first: with none it raises, and with some it sends a real OAuth +token to a local emulator. :func:`bigquery_client` passes +``AnonymousCredentials`` instead, the way goccy/bigquery-emulator documents +its clients. The goccy emulator also reports no ``outputRows`` for a load +job, so a load whose job reports no count is checked by counting the table +afterwards; it is never taken as zero rows loaded, nor as success. """ from __future__ import annotations @@ -50,10 +69,47 @@ _MISSING_EXTRA = "loading into BigQuery needs the gcp extra: pip install 'data-product-forge[gcp]'" +#: The variable python-bigquery reads for an emulator endpoint +#: (``google.cloud.bigquery._helpers.BIGQUERY_EMULATOR_HOST``). +EMULATOR_HOST_ENV = "BIGQUERY_EMULATOR_HOST" + +#: DuckDB's names for a timestamp without a time zone, at every precision. +_NAIVE_TIMESTAMPS = frozenset({"TIMESTAMP", "TIMESTAMP_S", "TIMESTAMP_MS", "TIMESTAMP_NS"}) + + class BigQueryLoadError(RuntimeError): """The landed file could not be loaded, or loaded the wrong number of rows.""" +def emulator_host() -> Optional[str]: + """The BigQuery emulator endpoint this process is pointed at, if any.""" + host = os.environ.get(EMULATOR_HOST_ENV, "").strip() + return host or None + + +def bigquery_client(bigquery: Any, project: Optional[str]) -> Any: + """A ``bigquery.Client`` for ``project``: ADC, or anonymous on an emulator. + + Without an emulator this is exactly ``bigquery.Client(project=project)``, + so gcloud ADC, a VM's service account and Workload Identity Federation + (an ``external_account`` credential file) resolve as they always did. With + ``BIGQUERY_EMULATOR_HOST`` set, the client already talks to that host + (python-bigquery reads the variable itself) and gets no credentials to + send there: goccy/bigquery-emulator's documented client setup. + """ + if emulator_host() is None: + return bigquery.Client(project=project) + return bigquery.Client(project=project, credentials=_anonymous_credentials()) + + +def _anonymous_credentials() -> Any: + """``google.auth``'s ``AnonymousCredentials``, imported on first use. Tests replace this, + as they replace :func:`_bigquery_module`, so the unit lanes need no Google library.""" + from google.auth.credentials import AnonymousCredentials + + return AnonymousCredentials() + + def _bigquery_module() -> Any: """``google.cloud.bigquery``, imported on first use. Tests replace this.""" try: @@ -69,7 +125,11 @@ def bigquery_load_target( """The table a binding loads into, or None when it is not a BigQuery table. Resolved with the IaC's own helpers, so the load names the dataset, table - and location ``_emit_bigquery`` created. + and location ``_emit_bigquery`` created. A binding that names no region + gives ``location: None``, never a guessed ``US``: :func:`load_file` then + runs the job where the table itself is (its ``location``), which is the + only place a load job can run. The guessed default could send a job for + an EU table to the US multi-region. """ from ..iac.providers.gcp import BIGQUERY_TABLE, _bq_table_name, resolve_gcp_target @@ -83,7 +143,7 @@ def bigquery_load_target( "project": project, "dataset": loc.get("dataset") or "default", "table": _bq_table_name(expose, loc), - "location": loc.get("region") or loc.get("location") or "US", + "location": loc.get("region") or loc.get("location") or None, } @@ -108,6 +168,92 @@ def unsupported_reason(mode: str, sink_format: str, stream_count: int) -> Option return None +def _timestamp_columns(schema: Any) -> list: + """The names of the top-level columns the table declares ``TIMESTAMP``.""" + names = [] + for field in schema or []: + field_type = str(getattr(field, "field_type", "") or "").upper() + name = getattr(field, "name", None) + if name and field_type == "TIMESTAMP": + names.append(str(name)) + return names + + +def _quote_ident(name: str) -> str: + return '"' + name.replace('"', '""') + '"' + + +def utc_adjusted_copy(path: str, schema: Any) -> Optional[str]: + """A copy of the Parquet file at ``path`` whose ``TIMESTAMP`` columns are UTC-adjusted. + + ``schema`` is the destination table's. ``None`` when no column needs it: + the table declares no ``TIMESTAMP`` column, or the file already holds each + one as ``TIMESTAMPTZ``. The copy sits beside ``path`` and the caller + removes it after the load; ``path`` itself (the build's landed file, which + the run record names) is left as it is. + """ + wanted = {name.lower(): name for name in _timestamp_columns(schema)} + if not wanted: + return None + import duckdb + + con = duckdb.connect(":memory:") + try: + # Read as UTC: the cast below turns the naive wall clock into an instant. + con.execute("SET TimeZone = 'UTC'") + described = con.execute("DESCRIBE SELECT * FROM read_parquet(?)", [str(path)]).fetchall() + naive = [ + str(row[0]) + for row in described + if str(row[0]).lower() in wanted and str(row[1]).upper() in _NAIVE_TIMESTAMPS + ] + if not naive: + return None + replaced = ", ".join( + f"CAST({_quote_ident(c)} AS TIMESTAMPTZ) AS {_quote_ident(c)}" for c in naive + ) + base, _ext = os.path.splitext(str(path)) + out = f"{base}.bq-load.parquet" + # A path is not a bindable parameter in COPY ... TO; quote it as a literal. + target = "'" + out.replace("'", "''") + "'" + con.execute( + f"COPY (SELECT * REPLACE ({replaced}) FROM read_parquet(?)) " + f"TO {target} (FORMAT parquet)", + [str(path)], + ) + return out + finally: + con.close() + + +def quote_bq_ident(name: str) -> str: + """``name`` as a GoogleSQL quoted identifier. + + Backtick-quoted identifiers take the string-literal escapes (GoogleSQL + lexical structure, "Quoted identifiers"), so a backslash and a backtick + are escaped and nothing in ``name`` can close the quote. + """ + return "`" + str(name).replace("\\", "\\\\").replace("`", "\\`") + "`" + + +def quote_table_id(table_id: str) -> str: + """``project.dataset.table`` as three quoted GoogleSQL identifiers.""" + parts = str(table_id).rsplit(".", 2) + if len(parts) != 3 or not all(parts): + raise BigQueryLoadError(f"{table_id!r} is not a project.dataset.table id") + return ".".join(quote_bq_ident(part) for part in parts) + + +def _count_table(client: Any, table_id: str, location: Optional[str]) -> int: + """``COUNT(*)`` of ``table_id``, read back through a query.""" + sql = f"SELECT COUNT(*) FROM {quote_table_id(table_id)}" + rows = list(client.query(sql, location=location).result()) + try: + return int(list(rows[0].values())[0]) + except (IndexError, TypeError, ValueError) as exc: + raise BigQueryLoadError(f"counting {table_id} returned no readable count") from exc + + def load_file( path: str, target: Mapping[str, Any], @@ -122,32 +268,75 @@ def load_file( The table's own schema is passed to the job. Without it, a Parquet file whose columns are all optional cannot load into REQUIRED columns (googleapis/python-bigquery#2373), and ``WRITE_TRUNCATE`` would replace - the declared schema with the file's. + the declared schema with the file's. A ``TIMESTAMP`` column the file holds + as a naive timestamp is loaded from a UTC-adjusted copy + (:func:`utc_adjusted_copy`). + + The count is the job's ``outputRows``. A job that reports none (the goccy + emulator never does) is checked by counting the table: for + ``WRITE_TRUNCATE`` the table must then hold exactly the file's rows, and + for ``WRITE_APPEND`` it must have grown by them, which needs the count + before the load; that is taken on an emulator only, and a real job with no + count and no earlier count is refused rather than assumed. """ reason = unsupported_reason(mode, sink_format, 1) if reason: raise BigQueryLoadError(reason) bigquery = _bigquery_module() - client = bigquery.Client(project=target["project"]) + client = bigquery_client(bigquery, target["project"]) table_id = f"{client.project}.{target['dataset']}.{target['table']}" # Never create: the table is the IaC's, and a load that made its own would # carry the file's schema, not the contract's. table = client.get_table(table_id) + # The job runs where the table is: the binding's region when it names + # one, otherwise the table's own location, read above, never a default. + location = target.get("location") or getattr(table, "location", None) + appending = _WRITE_DISPOSITION[mode] == "WRITE_APPEND" + before: Optional[int] = None + if appending and emulator_host() is not None: + before = _count_table(client, table_id, location) job_config = bigquery.LoadJobConfig( source_format=_SOURCE_FORMAT[sink_format], write_disposition=_WRITE_DISPOSITION[mode], create_disposition="CREATE_NEVER", schema=table.schema, ) - with open(path, "rb") as fh: - job = client.load_table_from_file( - fh, table_id, job_config=job_config, location=target["location"] + upload = utc_adjusted_copy(path, table.schema) if sink_format == "parquet" else None + try: + with open(upload or path, "rb") as fh: + job = client.load_table_from_file( + fh, table_id, job_config=job_config, location=location + ) + job.result() + finally: + if upload is not None: + try: + os.remove(upload) + except OSError as exc: # the load's outcome stands; the copy is reported + logger.warning("bigquery.load copy_not_removed path=%s error=%s", upload, exc) + rows_from = "output_rows" + if job.output_rows is not None: + loaded = int(job.output_rows) + elif not appending: + loaded = _count_table(client, table_id, location) + rows_from = "count_after_load" + elif before is not None: + loaded = _count_table(client, table_id, location) - before + rows_from = "count_after_load" + else: + raise BigQueryLoadError( + f"the load job into {table_id} reported no output row count, and without the " + "table's count before an append the rows it added cannot be checked" ) - job.result() - loaded = int(job.output_rows or 0) if loaded != expected_rows: raise BigQueryLoadError( f"loaded {loaded} rows into {table_id}, but the landed file holds {expected_rows}" ) - logger.info("bigquery.load table=%s rows=%d job=%s", table_id, loaded, job.job_id) - return {"table": table_id, "rows": loaded, "job_id": job.job_id} + logger.info( + "bigquery.load table=%s rows=%d rows_from=%s job=%s", + table_id, + loaded, + rows_from, + job.job_id, + ) + return {"table": table_id, "rows": loaded, "rows_from": rows_from, "job_id": job.job_id} diff --git a/fluid_build/build_runners/_bigquery_read.py b/fluid_build/build_runners/_bigquery_read.py new file mode 100644 index 00000000..7a289593 --- /dev/null +++ b/fluid_build/build_runners/_bigquery_read.py @@ -0,0 +1,174 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Read a BigQuery table into a local Parquet file an embedded-SQL build's DuckDB reads. + +An embedded-SQL build on DuckDB whose ``consumes[]`` entry resolves to a +BigQuery table (the upstream's gcp ``bigquery_table`` binding) cannot read it +the way it reads an S3 prefix: DuckDB has no BigQuery reader of its own, and +``gs://`` is not where the table lives. So the table is read through the +BigQuery API, page by page, into a Parquet file under the build's +``.fluid/staging``, and the build's view over it is the ordinary local-file +view every other input gets. The SQL does not change. + +Borrowed, not built: + +* The read is python-bigquery's own: ``Client.list_rows(table)`` and + ``RowIterator.to_arrow_iterable()``, one Arrow record batch per page, written + with ``pyarrow.parquet.ParquetWriter`` as they arrive, so the whole table is + never held in memory. This is the pattern Google documents for BigQuery to + DuckDB (``to_arrow`` into a DuckDB relation; the MotherDuck and + ``davidgasquez.com/duckdb-bq-storage`` write-ups use it), streamed. +* Types follow the DuckDB BigQuery extension's documented read mapping + (hafenkran/duckdb-bigquery, ``docs/concepts/data-types.md``): a BigQuery + ``TIMESTAMP`` reads as a DuckDB ``TIMESTAMP`` holding the UTC wall clock, + not a ``TIMESTAMPTZ`` whose rendering would follow the runner's time zone. + That is also what the same SQL reads on the local and aws targets, where + the upstream lands naive timestamps, so one query means one thing on every + target. + +Why not the extension itself (``ATTACH 'project=...' AS bq (TYPE bigquery)``): +it is a native binary fetched from DuckDB's community repository at run +time (``INSTALL bigquery FROM community``), outside the pip install and its +lockfile, built for one DuckDB release line at a time while forge-cli accepts +``duckdb>=1.0``; its endpoint overrides are per-function parameters rather +than ``ATTACH`` options, so the attached catalog cannot be pointed at an +emulator in CI; and it would authenticate through google-cloud-cpp while the +load that follows (``_bigquery_load``) authenticates through google-auth, two +credential chains to configure and debug for one build. Reading with the +client the load already uses keeps one chain: gcloud ADC, a VM service account +and Workload Identity Federation (an ``external_account`` file named by +``GOOGLE_APPLICATION_CREDENTIALS``) all work with no key. + +The read uses the REST ``tabledata.list`` pages. The BigQuery Storage Read API +is faster for large tables, but it needs ``bigquery.readsessions.create`` and +the ``google-cloud-bigquery-storage`` package, neither of which the ``gcp`` +extra or the documented pipeline role grants; it is a follow-up, not a silent +fallback. +""" + +from __future__ import annotations + +import logging +import os +from pathlib import Path +from typing import Any, Dict, Optional + +from . import _bigquery_load +from ._bigquery_load import BigQueryLoadError, bigquery_client + +_MISSING_PYARROW = ( + "reading a BigQuery table into DuckDB needs pyarrow: pip install 'data-product-forge[gcp]'" +) + + +class BigQueryReadError(RuntimeError): + """An upstream BigQuery table could not be read into the build.""" + + +def missing_dependency() -> Optional[str]: + """Why a BigQuery read cannot run here, or ``None`` when it can.""" + try: + # Through the module, so a test that replaces ``_bigquery_module`` there + # replaces it for the read as well as the load. + _bigquery_load._bigquery_module() + except BigQueryLoadError as exc: + return str(exc) + try: + import pyarrow # noqa: F401 + import pyarrow.parquet # noqa: F401 + except ImportError: + return _MISSING_PYARROW + return None + + +def _naive_schema(schema: Any) -> Any: + """``schema`` with each zoned timestamp as a naive one of the same unit. + + Arrow stores a timestamp as an offset from the epoch in UTC whatever its + ``tz``, so a cast that drops the zone keeps the value, which then reads as + the UTC wall clock. + """ + import pyarrow as pa + + return pa.schema( + [ + ( + f.with_type(pa.timestamp(f.type.unit)) + if pa.types.is_timestamp(f.type) and f.type.tz is not None + else f + ) + for f in schema + ], + metadata=schema.metadata, + ) + + +def stage_table(table_id: str, dest: Path, *, logger: logging.Logger) -> Dict[str, Any]: + """Write every row of BigQuery ``table_id`` to the Parquet file ``dest``. + + ``table_id`` is ``project.dataset.table``; its project may be empty, and + the client's own (``GOOGLE_PROJECT`` and friends, else ADC's) is used. + Returns ``{"table", "rows", "path"}``, the table as the client resolved + it. A table that cannot be read (missing, forbidden) is a + :class:`BigQueryReadError` naming it; an empty table stages an empty file + with the table's columns, which the SQL then reads as zero rows. + """ + problem = missing_dependency() + if problem: + raise BigQueryReadError(problem) + import pyarrow as pa + import pyarrow.parquet as pq + + bigquery = _bigquery_load._bigquery_module() + project, dataset, table = table_id.rsplit(".", 2) + client = bigquery_client(bigquery, project or None) + resolved = f"{client.project}.{dataset}.{table}" + try: + bq_table = client.get_table(resolved) + except Exception as exc: # noqa: BLE001 - NotFound, Forbidden: all name the table + raise BigQueryReadError( + f"could not read BigQuery table {resolved}: {type(exc).__name__}: {exc}" + ) from exc + dest.parent.mkdir(parents=True, exist_ok=True) + partial = dest.with_name(dest.name + ".partial") + rows = 0 + writer = None + try: + for batch in client.list_rows(bq_table).to_arrow_iterable(): + schema = _naive_schema(batch.schema) + if writer is None: + writer = pq.ParquetWriter(str(partial), schema) + # Table.cast, not RecordBatch.cast (pyarrow 16+): the pin is pyarrow>=14. + writer.write_table(pa.Table.from_batches([batch]).cast(schema)) + rows += batch.num_rows + if writer is None: + # No page at all: an empty table. Its columns still come from BigQuery. + empty = client.list_rows(bq_table, max_results=0).to_arrow() + schema = _naive_schema(empty.schema) + writer = pq.ParquetWriter(str(partial), schema) + writer.write_table(empty.cast(schema)) + writer.close() + writer = None + os.replace(partial, dest) + finally: + if writer is not None: + writer.close() + if partial.exists(): + partial.unlink() + logger.info("bigquery.read table=%s rows=%d staged=%s", resolved, rows, dest) + return {"table": resolved, "rows": rows, "path": str(dest)} + + +__all__ = ["BigQueryReadError", "missing_dependency", "stage_table"] diff --git a/fluid_build/build_runners/_embedded_sql_io.py b/fluid_build/build_runners/_embedded_sql_io.py index d7d20372..e5ce4fcb 100644 --- a/fluid_build/build_runners/_embedded_sql_io.py +++ b/fluid_build/build_runners/_embedded_sql_io.py @@ -43,9 +43,13 @@ through the duckdb acquisition runner's own key rule (``_object_store_uri``), read as the glob of that prefix's files of the binding format, which is also what the Glue table ``fluid apply`` declares - for the binding serves. Anything this engine cannot read (a warehouse - table, a stream, a GCS or Azure prefix) is an :class:`UnreadableBindingError` - naming the platform. + for the binding serves; a GCP ``bigquery_table`` binding as that table, + named by the IaC's own rule (``_bigquery_load.bigquery_load_target``, the + table ``fluid apply`` created and the upstream's build loads), read + through the BigQuery API into a staged Parquet file when the build runs + (``_bigquery_read``). Anything else this engine cannot read (another + warehouse's table, a stream, a GCS or Azure prefix) is an + :class:`UnreadableBindingError` naming the platform. Each entry the SQL reads becomes one DuckDB view named by its ``exposeId``, a quoted identifier that must pass ``validate_ident``. What the SQL reads is @@ -65,10 +69,16 @@ **Lands.** When the first expose's binding is an AWS object-store binding, the result is written to exactly the object the acquisition runner writes for that binding (``_object_store_uri`` + ``_file_within_prefix``), inside the prefix -the Glue table points at, so ``fluid verify --env aws`` counts it. A local -binding is unchanged. An expose declaring ``policy.privacy.masking`` is -refused: this path does not apply masking, and cleartext must not land -silently. +the Glue table points at, so ``fluid verify --env aws`` counts it. When it is +a GCP ``bigquery_table`` binding, the result is staged as Parquet under the +build's ``.fluid/staging`` and one load job moves it into that table +(``_bigquery_load.load_file``, the acquisition runner's own load), whatever +``location.path`` the binding also carries: a ``gs://`` path there is not a +file this path writes. A local binding is unchanged. Any other landing (a +``gs://`` or other non-S3 URI, a GCS bucket, another warehouse) is refused +with :class:`EmbeddedSqlLandingError` rather than written to a local file of +that name. An expose declaring ``policy.privacy.masking`` is refused: this +path does not apply masking, and cleartext must not land silently. Borrowed, not built: @@ -156,6 +166,13 @@ class MaskingNotAppliedError(EmbeddedSqlLandingError): code: str = "MaskingNotAppliedError" +@dataclass +class EmbeddedSqlSovereigntyError(EmbeddedSqlLandingError): + """A BigQuery read or landing the contract's ``sovereignty`` block does not allow.""" + + code: str = "EmbeddedSqlSovereigntyError" + + _EXPLICIT_INPUT_FIX = ( "Or bind it by hand: a builds[].properties.parameters.inputs entry named " "'{name}' (with the path to read) wins over the consumes entry and is used as is." @@ -180,6 +197,15 @@ class ResolvedInput: platform: str contract_path: Path region: Optional[str] = None + #: A BigQuery upstream: its ``project.dataset.table`` (the project empty + #: when the client's own is used), read into ``read_path`` before the SQL. + table: Optional[str] = None + #: The local Parquet file a BigQuery upstream was staged into. + read_path: Optional[str] = None + #: Whether the upstream's binding names its region. A BigQuery table whose + #: binding names none is in the IaC's default location (``US``), which a + #: ``sovereignty`` block must not be satisfied by (:func:`refuse_sovereignty_breach`). + region_declared: bool = True @property def view(self) -> str: @@ -188,20 +214,28 @@ def view(self) -> str: def as_input_spec(self) -> Dict[str, Any]: """The local provider's input spec (``_register_inputs``) for this entry.""" + if self.table is not None and self.read_path is None: + raise RuntimeError( + f"{_entry(self.product_id, self.expose_id)}: BigQuery table {self.table} " + "was not staged before the SQL ran" + ) spec: Dict[str, Any] = { "table": self.view, - "path": self.uri, - "format": self.format, + "path": self.read_path or self.uri, + "format": "parquet" if self.read_path else self.format, "quoted": True, "productId": self.product_id, "exposeId": self.expose_id, } - if self.region: + if self.region and self.table is None: spec["region"] = self.region return spec def record(self) -> Dict[str, str]: - return {"productId": self.product_id, "exposeId": self.expose_id, "uri": self.uri} + rec = {"productId": self.product_id, "exposeId": self.expose_id, "uri": self.uri} + if self.read_path: + rec["staged"] = self.read_path + return rec @dataclass(frozen=True) @@ -239,6 +273,46 @@ def as_output_spec(self) -> Dict[str, Any]: return spec +@dataclass(frozen=True) +class BigQueryLanding: + """The BigQuery table an embedded-SQL build loads its result into. + + ``project`` is ``None`` when neither the binding nor the environment names + one, and the client's own is used, as the acquisition runner's load does. + ``staged`` is the local Parquet file the SQL writes and the load reads, + set when the build runs. + """ + + project: Optional[str] + dataset: str + table: str + location: str + staged: Optional[str] = None + #: Whether the binding names the region; ``location`` is otherwise the + #: IaC's default (``US``), as :attr:`ResolvedInput.region_declared`. + region_declared: bool = True + #: The expose the result lands in, for the run record. + expose_id: str = "result" + + @property + def table_id(self) -> str: + return f"{self.project or ''}.{self.dataset}.{self.table}" + + def load_target(self) -> Dict[str, Any]: + """The target ``_bigquery_load.load_file`` takes. + + An undeclared region passes no location, so the job runs where the + table itself is (``load_file`` reads the table's own), never in the + IaC's default guessed on the table's behalf. + """ + return { + "project": self.project, + "dataset": self.dataset, + "table": self.table, + "location": self.location if self.region_declared else None, + } + + @dataclass class EmbeddedSqlIO: """Everything :func:`plan_embedded_sql_io` decided, before any SQL runs.""" @@ -248,6 +322,14 @@ class EmbeddedSqlIO: lineage_only: List[LineageOnlyInput] = field(default_factory=list) landing: Optional[Landing] = None workspace_root: Optional[Path] = None + bigquery_landing: Optional[BigQueryLanding] = None + #: What the plan let through but the operator must hear about: a further + #: output this path does not write, an advisory sovereignty finding. + warnings: List[str] = field(default_factory=list) + + @property + def bigquery_inputs(self) -> List[ResolvedInput]: + return [r for r in self.inputs if r.table is not None] # ── Helpers ───────────────────────────────────────────────────────────── @@ -472,6 +554,21 @@ def _unreadable(where: str, platform: str, fmt: str, detail: str) -> UnreadableB ) +def _require_bigquery_reader(where: str) -> None: + """Refuse a BigQuery upstream before any SQL runs when it cannot be read here.""" + from ._bigquery_read import missing_dependency + + problem = missing_dependency() + if problem: + raise UnreadableBindingError( + what=f"{where}: the upstream is a BigQuery table, and this install cannot read one", + why=problem, + fix="Install the gcp extra where this build runs: pip install 'data-product-forge[gcp]'", + doc=doc_url(), + extras={"platform": "gcp", "format": "bigquery_table"}, + ) + + def _read_uri(path: str, platform: str, fmt: str, where: str) -> Tuple[str, str, str]: """A binding whose ``location.path`` is already a URI: only ``s3://`` reads.""" from .duckdb.runner import _FILE_FORMAT_EXT @@ -557,6 +654,56 @@ def _read_s3_prefix( return f"{prefix}*.{_FILE_FORMAT_EXT[fmt]}", fmt, "aws" +#: The dataset location ``iac/providers/gcp.py::_emit_bigquery`` gives a +#: binding that names no region. +_IAC_DEFAULT_BIGQUERY_LOCATION = "US" + +#: The ``binding.location`` keys a BigQuery table is named by. +_BIGQUERY_LOCATION_KEYS = ("project", "dataset", "table", "view", "region", "location") + + +def _bigquery_target(expose: Mapping[str, Any], where: str) -> Optional[Dict[str, Any]]: + """The BigQuery table ``expose`` is bound to, or ``None`` when it is not one. + + Named by the IaC's own rule (``_bigquery_load.bigquery_load_target``, which + reads ``iac/providers/gcp.py``), so this is the table ``fluid apply`` + created and the acquisition runner loads. ``{{ env.* }}`` in the naming + keys is resolved first, refusing an unset or credential-shaped variable + (:func:`_resolve_env`), and the project falls back to the environment + (``GOOGLE_PROJECT`` and friends) as the load's does. + """ + from ._bigquery_load import bigquery_load_target + + raw_binding = expose.get("binding") + binding: Mapping[str, Any] = raw_binding if isinstance(raw_binding, Mapping) else {} + raw_loc = binding.get("location") + loc: Mapping[str, Any] = raw_loc if isinstance(raw_loc, Mapping) else {} + if bigquery_load_target(binding, expose) is None: + return None + resolved_loc: Dict[str, Any] = dict(loc) + for key in _BIGQUERY_LOCATION_KEYS: + if loc.get(key): + resolved_loc[key] = _resolve_env( + loc.get(key), field_name=f"location.{key}", where=where + ) + target = bigquery_load_target({**dict(binding), "location": resolved_loc}, expose) + if target is not None: + declared = bool(resolved_loc.get("region") or resolved_loc.get("location")) + # ``bigquery_load_target`` names no location for a binding that + # declares none, so the load job runs where the table is + # (``BigQueryLanding.load_target``). The sovereignty checks reason + # about the dataset ``_emit_bigquery`` creates for such a binding, + # which is the IaC's own default, ``US``; they must also know the + # location was not declared. + target["location"] = target.get("location") or _IAC_DEFAULT_BIGQUERY_LOCATION + target["location_declared"] = declared + return target + + +def _bigquery_table_id(target: Mapping[str, Any]) -> str: + return f"{target.get('project') or ''}.{target['dataset']}.{target['table']}" + + def _read_location( expose: Mapping[str, Any], upstream_path: Path, where: str, *, local_provider: bool = False ) -> Tuple[str, str, str, Optional[str]]: @@ -812,6 +959,27 @@ def _bind_consumes( upstream, expose = _upstream_expose( upstream_path, product_id, expose_id, env, logger or LOG ) + where = _entry(product_id, expose_id) + bigquery = _bigquery_target(expose, where) + if bigquery is not None: + # Before the location.path check: a bigquery_table binding may also + # carry a gs:// staging path, which is not where the table's rows are. + _require_bigquery_reader(where) + table_id = _bigquery_table_id(bigquery) + bound.resolved.append( + ResolvedInput( + product_id=product_id, + expose_id=expose_id, + uri=f"bigquery://{table_id.lstrip('.')}", + format="bigquery_table", + platform="gcp", + contract_path=upstream_path, + region=str(bigquery["location"]), + table=table_id, + region_declared=bool(bigquery.get("location_declared")), + ) + ) + continue uri, fmt, platform, region = _read_location( expose, upstream_path, @@ -901,6 +1069,331 @@ def refuse_unapplied_masking(contract: Mapping[str, Any], build: Mapping[str, An ) +#: ``binding.platform`` values whose storage this path cannot write, and whose +#: expose the local provider would otherwise write as a local file: another +#: cloud's object store or a warehouse. A gcp ``bigquery_table`` is checked +#: before this, and lands through a load job; a GCS bucket or any other gcp +#: resource does not. Platforms another stage delivers from the landed file +#: (an output port such as pgvector) keep the local write. +_UNLANDABLE_PLATFORMS = frozenset({"gcp", "azure", "snowflake", "databricks"}) + + +def _refuse_unlandable(binding: Mapping[str, Any], path: Any, platform: str, where: str) -> None: + """Refuse a landing this path would otherwise write to a local file. + + The local provider writes whatever ``location.path`` it is given to the + local filesystem, so a ``gs://`` path became a file named ``gs:/...`` and a + GCS or other-cloud binding a local file, and the build reported success + with nothing where the contract said it would be. + """ + fmt = _normalize_format(binding.get("format")) or "unset" + text = str(path or "") + scheme = text.split("://", 1)[0].lower() if is_remote_uri(text) else "" + if scheme and scheme != "s3": + raise EmbeddedSqlLandingError( + what=f"{where}: the embedded-SQL path cannot land in a {scheme}:// location", + why=( + f"The binding (platform {platform or 'unset'!r}, format {fmt!r}) names {text}. " + "This path writes a local file, an S3 object, or loads a BigQuery table, and " + "the local writer would have created a file named after that URI instead." + ), + fix=( + "Bind the expose as a gcp bigquery_table (it is loaded through a load job), " + "an aws bucket prefix, or a local path, or land it with another engine." + ), + doc=doc_url(), + extras={"platform": platform or "unset", "format": fmt, "scheme": scheme}, + ) + if platform in _UNLANDABLE_PLATFORMS: + raise EmbeddedSqlLandingError( + what=f"{where}: the embedded-SQL path cannot land a {platform} binding ({fmt})", + why=( + "This path writes a local file, an S3 object for an aws binding naming a " + "bucket, or loads a gcp bigquery_table; a " + f"{platform} binding of format {fmt!r} is none of them, and writing it as a " + "local file would report success with nothing where the contract says." + ), + fix=("Bind the expose as one of those, or land it with an engine for that platform."), + doc=doc_url(), + extras={"platform": platform, "format": fmt}, + ) + + +def _require_bigquery_writer(where: str) -> None: + from ._bigquery_read import missing_dependency + + problem = missing_dependency() + if problem: + raise EmbeddedSqlLandingError( + what=f"{where}: the expose is a BigQuery table, and this install cannot load one", + why=problem, + fix="Install the gcp extra where this build runs: pip install 'data-product-forge[gcp]'", + doc=doc_url(), + ) + + +def bigquery_landing( + contract: Mapping[str, Any], build: Mapping[str, Any] +) -> Optional[BigQueryLanding]: + """The BigQuery table the result is loaded into, or ``None`` when it lands elsewhere. + + The first expose, as for every landing here (the local provider writes + exposes[0]). A further expose the build names in ``outputs`` is not + landed by this path at all (:func:`further_outputs`). + """ + exposes = _landed_exposes(contract, build) + if not exposes: + return None + expose = exposes[0] + expose_id = str(expose.get("exposeId") or expose.get("id") or "result") + where = f"expose {expose_id}" + target = _bigquery_target(expose, where) + if target is None: + return None + _require_bigquery_writer(where) + return BigQueryLanding( + project=target.get("project") or None, + dataset=str(target["dataset"]), + table=str(target["table"]), + location=str(target["location"]), + region_declared=bool(target.get("location_declared")), + expose_id=expose_id, + ) + + +#: ``binding.platform`` values whose expose a further output would have to be +#: written to off this machine: an object store or a warehouse this path +#: writes only for the first expose, if at all. +_REMOTE_OUTPUT_PLATFORMS = frozenset({"aws"}) | _UNLANDABLE_PLATFORMS + + +def further_outputs(contract: Mapping[str, Any], build: Mapping[str, Any]) -> List[str]: + """Refuse a further output this path would not land remotely; warn about a local one. + + The local provider writes one result, the first expose's + (``LocalProvider._derive_actions_from_contract``), and this path lands + that one only: in S3, in a BigQuery table, or as a local file. Every + other expose the build names in ``outputs`` is written nowhere. One bound + to a cloud store or a warehouse (a BigQuery table, a ``gs://`` or S3 + prefix, another platform's binding) is refused: the build would report + success with nothing where that contract says. A local one, or an output + port another stage delivers (pgvector, kafka, ...), is returned as a + warning to print, which is what the build did before, said out loud. + """ + exposes = _landed_exposes(contract, build) + first_id = str(exposes[0].get("exposeId") or exposes[0].get("id") or "?") if exposes else "?" + warnings: List[str] = [] + for extra in exposes[1:]: + extra_id = str(extra.get("exposeId") or extra.get("id") or "?") + raw_binding = extra.get("binding") + binding: Mapping[str, Any] = raw_binding if isinstance(raw_binding, Mapping) else {} + raw_loc = binding.get("location") + loc: Mapping[str, Any] = raw_loc if isinstance(raw_loc, Mapping) else {} + platform = str(binding.get("platform") or "").strip().lower() + path = str(loc.get("path") or "") + if platform in _REMOTE_OUTPUT_PLATFORMS or is_remote_uri(path): + raise EmbeddedSqlLandingError( + what=( + f"expose {extra_id}: the embedded-SQL path lands one result, the first " + f"expose's ({first_id})" + ), + why=( + f"The build names {extra_id!r} in outputs and binds it to " + f"{platform or 'a remote location'} ({path or 'no path'}), which nothing " + "would write: the build would report success without it." + ), + fix="Land one expose per embedded-SQL build, or give this one its own build.", + doc=doc_url(), + extras={"exposeId": extra_id, "platform": platform or "unset"}, + ) + warnings.append( + f"expose {extra_id} is named in the build's outputs, but this path writes only " + f"the first expose ({first_id}); {extra_id} is not written by this build" + ) + return warnings + + +def bigquery_staging_dir(contract_dir: Path, build_id: Any) -> Path: + """``.fluid/staging/`` beside the contract, where the acquisition + runner stages a BigQuery-bound file too (``_bigquery_staging_path``).""" + from .duckdb.runner import _path_part + + return Path(contract_dir) / ".fluid" / "staging" / _path_part(build_id or "build") + + +def stage_bigquery_io( + io: EmbeddedSqlIO, contract_dir: Path, build_id: Any, *, logger: logging.Logger +) -> Tuple[EmbeddedSqlIO, List[Path], List[Dict[str, Any]]]: + """Read each BigQuery upstream into a local Parquet file; name the result's staged file. + + Returns ``(io with every BigQuery input staged and the landing's staged + path set, the staged input files, one fact per table read)``. The caller + removes the staged inputs after the build: they are copies of another + product's table. + """ + import dataclasses + + from ._bigquery_read import stage_table + from .duckdb.runner import _path_part + + root = bigquery_staging_dir(contract_dir, build_id) + staged: List[Path] = [] + reads: List[Dict[str, Any]] = [] + inputs: List[ResolvedInput] = [] + try: + for r in io.inputs: + if r.table is None: + inputs.append(r) + continue + dest = root / "inputs" / f"{_path_part(r.expose_id)}.parquet" + staged.append(dest) + reads.append({**stage_table(r.table, dest, logger=logger), "view": r.view}) + inputs.append(dataclasses.replace(r, read_path=str(dest))) + except Exception: + remove_staged(staged) + raise + landing = io.bigquery_landing + if landing is not None: + landing = dataclasses.replace( + landing, staged=str(root / f"{_path_part(landing.table)}.parquet") + ) + return dataclasses.replace(io, inputs=inputs, bigquery_landing=landing), staged, reads + + +def remove_staged(paths: Sequence[Path]) -> None: + """Delete staged upstream copies; a file already gone is not an error.""" + for path in paths: + try: + Path(path).unlink() + except FileNotFoundError: + pass + except OSError as exc: # pragma: no cover - reported, never raised + LOG.warning("embedded_sql_staged_input_not_removed path=%s error=%s", path, exc) + + +def load_bigquery_landing(landing: BigQueryLanding, *, logger: logging.Logger) -> Dict[str, Any]: + """Load the staged result into its table: the acquisition runner's own load. + + ``WRITE_TRUNCATE``: the embedded-SQL result replaces the table, as it + replaces the local file or the S3 object on the other targets. The count + the load is held to is read back from the staged file. + """ + from ._bigquery_load import load_file + from .duckdb.runner import _count_file_rows + + if landing.staged is None: + raise EmbeddedSqlLandingError( + what=f"BigQuery table {landing.table_id}: the result was not staged", + why="The load reads the staged Parquet file the SQL writes, and none was named.", + fix="Report this: the build must stage the result before it loads it.", + doc=doc_url(), + ) + return load_file( + landing.staged, + landing.load_target(), + mode="full_refresh", + sink_format="parquet", + expected_rows=_count_file_rows(landing.staged, "parquet"), + logger=logger, + ) + + +def write_bigquery_run_record( + contract: Mapping[str, Any], + build: Mapping[str, Any], + contract_dir: Path, + landing: BigQueryLanding, + *, + started_at: str, + facts: Optional[Mapping[str, Any]] = None, + error: Optional[str] = None, + logger: logging.Logger = LOG, +) -> Optional[str]: + """Record a BigQuery-landing run where ``fluid verify`` reads the acquisition runs. + + The acquisition runner records each run under + ``/.fluid/runs///runs/`` (``FileStateStore``), + with ``facets.bigquery_load`` naming the table and the rows the load + landed, and the BigQuery verifier holds the table's ``COUNT(*)`` to the + newest such run (``_verify_bigquery``). This writes the same record for an + embedded-SQL build, so silver and gold are held to the rows their load + landed, as bronze is: ``records_total`` is the load's count, which + :func:`load_bigquery_landing` has already held to the staged file's rows + (``rows_from: write``), and the mode is ``full_refresh`` (the load is + ``WRITE_TRUNCATE``). It is dbt's pattern too: ``run_results.json`` keeps + each node's ``relation_name`` and ``adapter_response.rows_affected``. + + A failed run is recorded without ``bigquery_load``: whether it changed + the table is unknown, so verify reports the count without a comparison + until the next run succeeds. Returns the run id, or ``None`` when ids the + state store refuses (``validate_identifier``) leave nothing to record. + """ + from ._acquisition_common import generate_run_id, utc_now_iso + from ._ids import IdentifierViolation, validate_identifier + from ._state import FileStateStore + + try: + product_id = validate_identifier(str(contract.get("id") or ""), kind="contract.id") + build_id = validate_identifier(str(build.get("id") or ""), kind="build.id") + except IdentifierViolation as exc: + logger.warning("embedded_sql_run_record_skipped reason=%s", type(exc).__name__) + return None + run_id = generate_run_id() + succeeded = facts is not None and error is None + facets: Dict[str, Any] = {"engine": "duckdb", "pattern": "embedded-logic"} + if succeeded and facts is not None: + facets["bigquery_load"] = dict(facts) + facets["landed"] = { + "mode": "full_refresh", + "rows_from": "write", + "destinations": {landing.expose_id: f"bigquery://{facts.get('table')}"}, + } + record: Dict[str, Any] = { + "run_id": run_id, + "state": "succeeded" if succeeded else "failed", + "started_at": started_at, + "finished_at": utc_now_iso(), + "records_total": int(facts["rows"]) if succeeded and facts is not None else 0, + "streams": [], + "facets": facets, + } + if error is not None: + record["error"] = error + FileStateStore(Path(contract_dir) / ".fluid").write_run_record(product_id, build_id, record) + return run_id + + +def refuse_unlandable_first_expose(contract: Mapping[str, Any]) -> None: + """For a DuckDB SQL build with no inline SQL: refuse what it would land locally. + + Such a build keeps the provider's old handling (no consumes resolution and + no remote landing), so a BigQuery table, a GCS path or another platform's + binding would be written as a local file of that name. + """ + exposes = [e for e in contract.get("exposes") or [] if isinstance(e, Mapping)] + if not exposes: + return + expose = exposes[0] + expose_id = str(expose.get("exposeId") or expose.get("id") or "result") + where = f"expose {expose_id}" + binding = expose.get("binding") if isinstance(expose.get("binding"), Mapping) else {} + if _bigquery_target(expose, where) is not None: + raise EmbeddedSqlLandingError( + what=f"{where}: only a build with inline properties.sql loads a BigQuery table", + why=( + "This build has no inline SQL, so it runs the provider's multi-stage path, " + "which writes local files and would land nothing in the table." + ), + fix="Give the build its SQL in properties.sql.", + doc=doc_url(), + ) + raw_loc = binding.get("location") + loc: Mapping[str, Any] = raw_loc if isinstance(raw_loc, Mapping) else {} + _refuse_unlandable( + binding, loc.get("path"), str(binding.get("platform") or "").strip().lower(), where + ) + + def object_store_landing( contract: Mapping[str, Any], build: Mapping[str, Any] ) -> Optional[Landing]: @@ -917,6 +1410,11 @@ def object_store_landing( the stream (``_file_within_prefix``). A ``path`` that is already an ``s3://`` URI is used as the runner uses it. No bucket, or no path, keeps the local write, as it does in the runner. + + ``None`` for a BigQuery table too, which :func:`bigquery_landing` lands. + A landing this path cannot perform (another URI scheme, a platform that is + neither local, aws nor a BigQuery table) is refused + (:func:`_refuse_unlandable`) instead of becoming a local file. """ from .duckdb.runner import _file_within_prefix, _object_store_uri @@ -927,16 +1425,19 @@ def object_store_landing( binding = expose.get("binding") if isinstance(expose.get("binding"), Mapping) else {} raw_loc = binding.get("location") loc: Mapping[str, Any] = raw_loc if isinstance(raw_loc, Mapping) else {} + expose_id = str(expose.get("exposeId") or expose.get("id") or "result") + where = f"expose {expose_id}" + if _bigquery_target(expose, where) is not None: + return None # a BigQuery table: :func:`bigquery_landing` loads it raw_path = loc.get("path") + platform = str(binding.get("platform") or "").strip().lower() + _refuse_unlandable(binding, raw_path, platform, where) if not raw_path: return None - platform = str(binding.get("platform") or "").strip().lower() names_bucket = platform == "aws" and bool(loc.get("bucket")) if not names_bucket and not is_remote_uri(str(raw_path)): return None # a local binding: unchanged - expose_id = str(expose.get("exposeId") or expose.get("id") or "result") - where = f"expose {expose_id}" path = str(_resolve_env(raw_path, field_name="location.path", where=where)) resolved_loc: Dict[str, Any] = { **dict(loc), @@ -946,7 +1447,10 @@ def object_store_landing( region = _resolve_env(loc.get("region"), field_name="location.region", where=where) uri: Optional[str] if is_remote_uri(path): - uri = path if path.lower().startswith("s3://") else None + # A resolved path that is still a non-S3 URI is refused like one written + # out: the local provider would write a file named "gs:/..." instead. + _refuse_unlandable(binding, path, platform, where) + uri = path else: resolved_loc["bucket"] = _resolve_env( loc.get("bucket"), field_name="location.bucket", where=where @@ -971,6 +1475,286 @@ def object_store_landing( ) +def _same_bigquery_table(input_table: str, landing: BigQueryLanding) -> bool: + """Whether a BigQuery input and the landing may be one table. + + Names compared case-insensitively. A project left to the client (empty on + either side) matches any project: both resolve to the client's own at run + time, so they may well be the same, and the doubt is refused, not assumed. + """ + project, dataset, table = input_table.rsplit(".", 2) + if (dataset.lower(), table.lower()) != (landing.dataset.lower(), landing.table.lower()): + return False + return not project or not landing.project or project.lower() == landing.project.lower() + + +def _s3_read_prefix(uri: str) -> str: + """The prefix an S3 input reads: the glob's directory, or the one object itself.""" + last = uri.rsplit("/", 1)[-1] + if any(ch in last for ch in "*?["): + return uri[: len(uri) - len(last)] + return uri + + +def refuse_landing_into_input(io: EmbeddedSqlIO) -> None: + """Refuse a landing that would write into one of the build's own inputs. + + The same rule Dagster applies to an asset whose dependency is its own key + (``_validate_self_deps``, "Asset ... depends on itself"), on what the two + resolve to rather than on product ids: :func:`_refuse_self_consume` already + refuses a contract consuming its own id, but an overlay naming an + upstream's table (a typo, or two products left to the ``default`` + dataset with the same expose id) passed it, and the build's + ``WRITE_TRUNCATE`` replaced another product's rows with its query result. + That is a data write, not a planned delete, so ``--allow-data-loss`` is + never asked. An S3 landing inside the prefix an input reads is refused for + the same reason: the object would become rows of that product's table. + """ + if io.bigquery_landing is not None: + _refuse_loading_an_input(io.bigquery_landing, io.inputs) + if io.landing is not None: + _refuse_landing_in_an_input_prefix(io.landing, io.inputs) + + +def _refuse_loading_an_input(bq: BigQueryLanding, inputs: Sequence[ResolvedInput]) -> None: + for r in inputs: + if r.table is None or not _same_bigquery_table(r.table, bq): + continue + raise EmbeddedSqlLandingError( + what=( + f"expose {bq.expose_id}: the result would be loaded into BigQuery table " + f"{bq.table_id}, which {_entry(r.product_id, r.expose_id)} reads" + ), + why=( + f"The load replaces the table (WRITE_TRUNCATE), so {r.product_id}'s rows " + "would be overwritten with this build's query result." + ), + fix=( + "Bind this expose to its own dataset and table in the overlay, and name " + "the project on both bindings so they cannot resolve to the same table." + ), + doc=doc_url(), + extras={"table": bq.table_id, "productId": r.product_id}, + ) + + +def _refuse_landing_in_an_input_prefix(landing: Landing, inputs: Sequence[ResolvedInput]) -> None: + for r in inputs: + if r.table is not None or not r.uri.lower().startswith("s3://"): + continue + prefix = _s3_read_prefix(r.uri) + inside = prefix.endswith("/") and landing.uri.startswith(prefix) + if landing.uri != prefix and not inside: + continue + raise EmbeddedSqlLandingError( + what=( + f"the result would land at {landing.uri}, inside {prefix}, which " + f"{_entry(r.product_id, r.expose_id)} reads" + ), + why=( + f"Every object under that prefix is a row source of {r.product_id}'s " + "table, so this build would add its result to another product's data." + ), + fix="Bind this expose to a prefix of its own in the overlay.", + doc=doc_url(), + extras={"uri": landing.uri, "productId": r.product_id}, + ) + + +#: BigQuery's two multi-regions, by the jurisdiction Google's own location +#: value groups put them in: ``in:eu-locations`` lists ``EU`` and +#: ``in:us-locations`` lists ``US`` (Resource Manager, "Restricting resource +#: locations"). The region table the sovereignty validator reads +#: (``policy.sovereignty.region_jurisdiction_map``) carries single regions only. +_BIGQUERY_MULTI_REGIONS = {"eu": "EU", "us": "US"} + + +def _jurisdiction(location: str) -> str: + from fluid_build.policy.sovereignty import region_jurisdiction_map + + table = region_jurisdiction_map() + text = str(location or "").strip() + return ( + table.get(text) + or table.get(text.lower()) + or _BIGQUERY_MULTI_REGIONS.get(text.lower()) + or "Unknown" + ) + + +#: A sovereignty finding: ``(severity, message)``, the severity ``severity_for``'s. +_Finding = Tuple[str, str] + + +def refuse_sovereignty_breach(contract: Mapping[str, Any], io: EmbeddedSqlIO) -> List[str]: + """Hold the BigQuery tables this build reads and loads to the contract's ``sovereignty``. + + ``fluid validate`` checks a binding's declared ``region`` and skips one + that declares none, and a BigQuery binding with no region is created, + loaded and read in ``US`` (the IaC's default). Reading an EU upstream + through this path and loading the result is a copy the build itself + makes, so it is checked here, before anything is read, by the validator's + own rules (``policy.sovereignty.SovereigntyValidator``) on the locations + the reads and the load actually use: + + * a BigQuery input or landing whose binding names no region; + * the landing's location in ``deniedRegions`` (an error in every mode), + outside ``allowedRegions``, or outside ``jurisdiction``; + * with ``dataResidency`` and no ``crossBorderTransfer`` (the schema's + defaults), an input in another jurisdiction than the landing. + + A finding blocks at the severity ``enforcementMode`` gives it + (``severity_for``: strict refuses, advisory warns, audit logs). Returns + the warnings to print; an unknown jurisdiction is a warning, never an + agreement. Nothing is checked for a build with no BigQuery read or load. + """ + raw = contract.get("sovereignty") + sovereignty: Mapping[str, Any] = raw if isinstance(raw, Mapping) else {} + if not sovereignty or (io.bigquery_landing is None and not io.bigquery_inputs): + return [] + from fluid_build.policy.sovereignty import ( + DEFAULT_ENFORCEMENT_MODE, + EnforcementMode, + severity_for, + ) + + try: + mode = EnforcementMode(sovereignty.get("enforcementMode", DEFAULT_ENFORCEMENT_MODE)) + except ValueError: + mode = EnforcementMode.STRICT # a mode the schema does not know enforces, not less + blocking = severity_for(mode) + findings = [ + *_undeclared_region_findings(io, blocking), + *_landing_region_findings(io.bigquery_landing, sovereignty, blocking), + *_cross_border_findings(io, sovereignty, blocking), + ] + errors = [m for sev, m in findings if sev == "error"] + if errors: + more = f" (and {len(errors) - 1} more below)" if len(errors) > 1 else "" + raise EmbeddedSqlSovereigntyError( + what=f"the contract's sovereignty block refuses this build: {errors[0]}{more}", + why="; ".join(errors), + fix=( + "Name a region on every BigQuery binding (the overlay's location.region), " + "inside sovereignty.allowedRegions and the declared jurisdiction." + ), + doc=doc_url(), + extras={"enforcementMode": mode.value, "findings": errors}, + ) + for message in (m for sev, m in findings if sev == "info"): + LOG.info("embedded_sql_sovereignty_audit %s", message) + return [f"sovereignty: {m}" for sev, m in findings if sev == "warning"] + + +def _undeclared_region_findings(io: EmbeddedSqlIO, blocking: str) -> List[_Finding]: + """A BigQuery input or landing whose location is the IaC's default, not declared.""" + findings: List[_Finding] = [ + ( + blocking, + f"{_entry(r.product_id, r.expose_id)}: BigQuery table {r.table} names no region, " + f"so it is read from the default location {r.region}", + ) + for r in io.bigquery_inputs + if not r.region_declared + ] + bq = io.bigquery_landing + if bq is not None and not bq.region_declared: + findings.append( + ( + blocking, + f"expose {bq.expose_id}: BigQuery table {bq.table_id} names no region, so it " + f"would be loaded in {bq.location}", + ) + ) + return findings + + +def _landing_region_findings( + bq: Optional[BigQueryLanding], sovereignty: Mapping[str, Any], blocking: str +) -> List[_Finding]: + """The validator's checks 1 to 3 on the location the load goes to.""" + if bq is None: + return [] + from fluid_build.policy.sovereignty import UNCONSTRAINED_JURISDICTIONS + + where = f"expose {bq.expose_id}: BigQuery table {bq.table_id}" + allowed = [str(x) for x in sovereignty.get("allowedRegions") or []] + findings: List[_Finding] = [] + # Denied is an error in every mode, as the validator's check 1 is. + if bq.location in [str(x) for x in sovereignty.get("deniedRegions") or []]: + findings.append(("error", f"{where}: region {bq.location} is explicitly denied")) + if allowed and bq.location not in allowed: + findings.append( + ( + blocking, + f"{where}: region {bq.location} is not in allowedRegions ({', '.join(allowed)})", + ) + ) + jurisdiction = sovereignty.get("jurisdiction") + if not jurisdiction or jurisdiction in UNCONSTRAINED_JURISDICTIONS: + return findings + found = _jurisdiction(bq.location) + if found == "Unknown": + findings.append(("warning", f"{where}: region {bq.location} has no known jurisdiction")) + elif found not in (jurisdiction, "Global"): + findings.append( + ( + blocking, + f"{where}: region {bq.location} is in {found}, not the required " + f"jurisdiction {jurisdiction}", + ) + ) + return findings + + +def _cross_border_findings( + io: EmbeddedSqlIO, sovereignty: Mapping[str, Any], blocking: str +) -> List[_Finding]: + """The validator's check 4 across the inputs and the landing: one jurisdiction.""" + from fluid_build.policy.sovereignty import ( + DEFAULT_CROSS_BORDER_TRANSFER, + DEFAULT_DATA_RESIDENCY, + ) + + residency = sovereignty.get("dataResidency", DEFAULT_DATA_RESIDENCY) + if not residency or sovereignty.get("crossBorderTransfer", DEFAULT_CROSS_BORDER_TRANSFER): + return [] + bq = io.bigquery_landing + land_at = bq.location if bq is not None else (io.landing.region if io.landing else None) + if land_at is None: + if io.landing is None: + return [] # a local file: no cloud location to compare + return [ + ( + blocking, + f"the result lands at {io.landing.uri}, whose binding names no region, so a " + "BigQuery input's transfer to it cannot be checked", + ) + ] + land_j = _jurisdiction(land_at) + findings: List[_Finding] = [] + for r in (r for r in io.inputs if r.region): + read_j = _jurisdiction(str(r.region)) + entry = _entry(r.product_id, r.expose_id) + if "Unknown" in (land_j, read_j): + findings.append( + ( + "warning", + f"{entry} is read from {r.region} and the result lands in {land_at}; one " + "has no known jurisdiction, so the transfer cannot be verified", + ) + ) + elif read_j != land_j: + findings.append( + ( + blocking, + f"{entry} is read from {r.region} ({read_j}) and the result lands in " + f"{land_at} ({land_j}), and crossBorderTransfer is false", + ) + ) + return findings + + def plan_embedded_sql_io( contract: Mapping[str, Any], build: Mapping[str, Any], @@ -985,32 +1769,53 @@ def plan_embedded_sql_io( :func:`object_store_landing`); ``build`` may be either. """ refuse_unapplied_masking(contract, build) + warnings = further_outputs(contract, build) + bq_landing = bigquery_landing(contract, build) landing = object_store_landing(contract, build) bound = _bind_consumes(contract, build, contract_dir, env=env, logger=logger) - return EmbeddedSqlIO( + io = EmbeddedSqlIO( inputs=bound.resolved, covered=bound.covered, lineage_only=bound.lineage_only, landing=landing, workspace_root=bound.workspace_root, + bigquery_landing=bq_landing, + warnings=warnings, ) + # Before anything is read or staged: both are about what the build would + # write where, which is known from the plan alone. + refuse_landing_into_input(io) + io.warnings.extend(refuse_sovereignty_breach(contract, io)) + return io __all__ = [ + "BigQueryLanding", "ConsumesResolutionError", "CoveredInput", "EmbeddedSqlIO", "EmbeddedSqlLandingError", + "EmbeddedSqlSovereigntyError", "FILE_FORMATS", "Landing", "LineageOnlyInput", "MaskingNotAppliedError", "ResolvedInput", "UnreadableBindingError", + "bigquery_landing", + "bigquery_staging_dir", "explicit_input_names", + "further_outputs", + "load_bigquery_landing", "object_store_landing", "plan_embedded_sql_io", + "refuse_landing_into_input", + "refuse_sovereignty_breach", "refuse_unapplied_masking", + "refuse_unlandable_first_expose", "relations_read", + "remove_staged", "resolve_consumes", + "stage_bigquery_io", + "write_bigquery_run_record", ] diff --git a/fluid_build/build_runners/base.py b/fluid_build/build_runners/base.py index ad576219..65097ac5 100644 --- a/fluid_build/build_runners/base.py +++ b/fluid_build/build_runners/base.py @@ -352,7 +352,7 @@ def _print_embedded_sql_io(io: Any) -> None: f' ⬅ consumes {r.product_id}/{r.expose_id} as view "{r.view}": {r.uri}', markup=False, ) - if io.inputs: + if io.inputs and io.bigquery_landing is None: cprint( " (no run record or lineage event on this path: the resolved inputs are " "listed here and in runtime/out/local_apply_log.jsonl)", @@ -360,6 +360,16 @@ def _print_embedded_sql_io(io: Any) -> None: ) if io.landing is not None: cprint(f" ➡ lands {io.landing.uri}", markup=False) + if io.bigquery_landing is not None: + cprint( + f" ➡ lands BigQuery table {io.bigquery_landing.table_id} " + f"({io.bigquery_landing.location}): staged as Parquet, then one load job " + "(WRITE_TRUNCATE), recorded in the build's run record", + markup=False, + ) + for warning in getattr(io, "warnings", None) or []: + cprint(f" ⚠️ {warning}", markup=False) + LOG.warning("embedded_sql_plan_warning %s", warning) def _print_action_errors(results: List[Dict[str, Any]], io: Any) -> None: @@ -399,6 +409,12 @@ def _bind_embedded_sql_io(actions: List[Dict[str, Any]], io: Any) -> None: ] if io.landing is not None: action["outputs"] = [io.landing.as_output_spec()] + if io.bigquery_landing is not None: + # The staged file the load reads, never the binding's own + # location.path (a gs:// staging prefix is not a local file). + if io.bigquery_landing.staged is None: + raise RuntimeError("the BigQuery landing was not staged before the SQL ran") + action["outputs"] = [{"path": io.bigquery_landing.staged, "format": "parquet"}] def _local_sql_actions( @@ -552,8 +568,15 @@ def _execute_embedded_sql_build( runs, and so does one naming this contract's own id. The result lands in S3 when the first expose is an AWS object-store binding, at the object the duckdb acquisition runner would write for it; a local - binding is unchanged. An expose declaring ``policy.privacy.masking`` is - refused on this path, which does not apply it. + binding is unchanged. An entry resolving to a GCP ``bigquery_table`` is + read through the BigQuery API into a staged Parquet file first, and a + first expose bound to one is loaded into that table by one load job after + the SQL (``_embedded_sql_io.stage_bigquery_io`` / + ``load_bigquery_landing``); a failed or short load fails the build. Any + other landing this path cannot write (a ``gs://`` path, a GCS bucket, + another warehouse) is refused before the SQL runs. An expose declaring + ``policy.privacy.masking`` is refused on this path, which does not apply + it. Returns 0 on success, 1 on failure. """ @@ -577,12 +600,9 @@ def _execute_embedded_sql_build( cprint(f"🔷 Build '{build_id}' (embedded-SQL / {engine_label})") io = None - # Only a build carrying inline SQL reads the views: one without it (a - # multi-stage ``engine: sql`` build) keeps the provider's old handling. - has_inline_sql = bool(str((build.get("properties") or {}).get("sql") or "").strip()) - if platform in LOCAL_SQL_PLATFORMS and has_inline_sql: - io = _plan_duckdb_io(unresolved_contract, build, contract_dir, env=env) - if io is None: + if platform in LOCAL_SQL_PLATFORMS: + planned, io = _plan_local_sql(unresolved_contract, build, contract_dir, env=env) + if not planned: return 1 if dry_run: @@ -617,9 +637,65 @@ def _execute_embedded_sql_build( LOG.exception("embedded_sql_build_error build_id=%s", build_id) return 1 + return _run_local_sql(build, contract, contract_dir, io) + + +def _plan_local_sql( + unresolved_contract: Dict[str, Any], + build: Dict[str, Any], + contract_dir: Path, + *, + env: Optional[str], +) -> Tuple[bool, Any]: + """``(planned, io)`` for a build on the local DuckDB engine; ``planned`` False if refused. + + Only a build carrying inline SQL reads the views (``io``): one without it + (a multi-stage ``engine: sql`` build) keeps the provider's old handling, + except that a first expose it would write as a local file of the wrong + kind (a BigQuery table, a ``gs://`` path) is refused. + """ + has_inline_sql = bool(str((build.get("properties") or {}).get("sql") or "").strip()) + if has_inline_sql: + io = _plan_duckdb_io(unresolved_contract, build, contract_dir, env=env) + return io is not None, io + from fluid_build._errors import FluidUserError + + from ._embedded_sql_io import refuse_unlandable_first_expose + + try: + refuse_unlandable_first_expose(unresolved_contract) + except FluidUserError as exc: + _print_typed_error(exc) + LOG.error("embedded_sql_io_refused build_id=%s code=%s", build.get("id"), exc.code) + return False, None + return True, None + + +def _run_local_sql( + build: Dict[str, Any], contract: Dict[str, Any], contract_dir: Path, io: Any +) -> int: + """Run the build on the local provider's DuckDB; 0 on success, 1 on failure. + + A BigQuery upstream is staged first and its copy removed afterwards; a + BigQuery landing is loaded after the SQL, and a failed load fails the build. + """ + import time + + from ._acquisition_common import utc_now_iso + + build_id = build.get("id", "unknown") + staged_inputs: List[Path] = [] + # A BigQuery landing is recorded as a run, as the acquisition load is, so + # ``fluid verify`` holds the table to the rows it landed. + planned_landing = io.bigquery_landing if io is not None else None + outcome: Dict[str, Any] = {} + started_at = utc_now_iso() try: from fluid_build.providers.local.local import LocalProvider + if io is not None and (io.bigquery_inputs or io.bigquery_landing is not None): + io, staged_inputs = _stage_bigquery(io, contract_dir, build_id) + # ``anchor_dir``: a relative ``location.path`` lands under the # source contract's directory, the same place the acquisition # runners write and ``fluid verify`` reads (it used to land under @@ -640,15 +716,90 @@ def _execute_embedded_sql_build( written_files.extend(r.get("written", [])) for p in written_files: cprint(f" 📁 {p}") + if io is not None and io.bigquery_landing is not None: + return _load_bigquery_result(io.bigquery_landing, build_id, outcome) return 0 else: cprint(f" ❌ Failed: {failed} action(s) failed") _print_action_errors(result.get("results") or [], io) + outcome["error"] = f"{failed} action(s) failed" return 1 except Exception as exc: cprint(f" ❌ Embedded-SQL build '{build_id}' error: {_redacted(exc)}") LOG.exception("embedded_sql_build_error build_id=%s", build_id) + outcome["error"] = _redacted(exc) + return 1 + finally: + if staged_inputs: + from ._embedded_sql_io import remove_staged + + remove_staged(staged_inputs) + if planned_landing is not None: + _record_bigquery_run( + build, contract, contract_dir, planned_landing, started_at, outcome + ) + + +def _record_bigquery_run( + build: Dict[str, Any], + contract: Dict[str, Any], + contract_dir: Path, + landing: Any, + started_at: str, + outcome: Dict[str, Any], +) -> None: + """Write the run record of a BigQuery-landing build; never changes the build's result.""" + from ._embedded_sql_io import write_bigquery_run_record + + facts = outcome.get("facts") + error = outcome.get("error") or (None if facts else "the build did not reach the load") + try: + run_id = write_bigquery_run_record( + contract, build, contract_dir, landing, started_at=started_at, facts=facts, error=error + ) + except Exception as exc: # noqa: BLE001 - reported; the load's outcome stands + cprint(f" ⚠️ the run record could not be written: {_redacted(exc)}", markup=False) + LOG.warning("embedded_sql_run_record_failed error=%s", type(exc).__name__) + return + if run_id is not None: + LOG.info("embedded_sql_run_recorded build_id=%s run_id=%s", build.get("id"), run_id) + + +def _stage_bigquery(io: Any, contract_dir: Path, build_id: Any) -> Tuple[Any, List[Path]]: + """Read every BigQuery upstream into a staged Parquet file, printing each read.""" + from ._embedded_sql_io import stage_bigquery_io + + io, staged, reads = stage_bigquery_io(io, contract_dir, build_id, logger=LOG) + for read in reads: + cprint( + f' ⬇ read {int(read["rows"]):,} row(s) from BigQuery table {read["table"]} ' + f'for view "{read["view"]}"', + markup=False, + ) + return io, staged + + +def _load_bigquery_result(landing: Any, build_id: Any, outcome: Dict[str, Any]) -> int: + """Load the staged result into its BigQuery table; 0 only when the rows arrived. + + ``outcome`` gets the load's facts, or its error, for the run record. + """ + from ._embedded_sql_io import load_bigquery_landing + + try: + facts = load_bigquery_landing(landing, logger=LOG) + except Exception as exc: # noqa: BLE001 - a failed or short load fails the build + cprint(f" ❌ BigQuery load into {landing.table_id} failed: {_redacted(exc)}") + LOG.error("embedded_sql_bigquery_load_failed build_id=%s", build_id) + outcome["error"] = _redacted(exc) return 1 + outcome["facts"] = facts + cprint( + f" ⬆ loaded {int(facts['rows']):,} row(s) into BigQuery table {facts['table']} " + f"(job {facts.get('job_id')}, rows from {facts.get('rows_from')})", + markup=False, + ) + return 0 def _manifest_env(path: Any) -> Optional[str]: @@ -723,6 +874,54 @@ def _run_env(args: argparse.Namespace, plan_data: Optional[Dict[str, Any]] = Non return None +def _runs_dir(contract_dir: Path, product_id: str, build_id: str) -> Optional[Path]: + """Where a build's run records are (``FileStateStore``), for ids the store accepts.""" + from ._ids import IdentifierViolation, validate_identifier + + try: + validate_identifier(product_id, kind="contract.id") + validate_identifier(build_id, kind="build.id") + except IdentifierViolation: + return None + return contract_dir / ".fluid" / "runs" / product_id / build_id / "runs" + + +def _run_ids(contract_dir: Path, product_id: str, build_id: str) -> Set[str]: + """The run ids a build has recorded so far.""" + runs = _runs_dir(contract_dir, product_id, build_id) + try: + return {p.stem for p in runs.glob("*.json")} if runs and runs.is_dir() else set() + except OSError: + return set() + + +def _report_build( + report: Any, + contract_dir: Path, + product_id: str, + build_id: str, + result: int, + runs_before: Optional[Set[str]], +) -> None: + """Record one build on ``report``, with the newest run record it wrote, if any. + + Run ids sort by time (``generate_run_id``), so the newest id this build + added is its run. Nothing is read when no ``fluid apply`` report is open. + """ + if report is None: + return + run: Optional[Dict[str, Any]] = None + runs = _runs_dir(contract_dir, product_id, build_id) + added = sorted(_run_ids(contract_dir, product_id, build_id) - (runs_before or set())) + if runs is not None and added: + try: + loaded = json.loads((runs / f"{added[-1]}.json").read_text(encoding="utf-8")) + run = loaded if isinstance(loaded, dict) else None + except (OSError, ValueError): + run = None + report.record_build(build_id=build_id, status="succeeded" if result == 0 else "failed", run=run) + + def run_builds_from_args( args: argparse.Namespace, logger: logging.Logger, @@ -750,6 +949,10 @@ def run_builds_from_args( root): the contract itself, the contract a bundle's MANIFEST records, or the contract a plan records (through its bundle when it was planned from one). See :func:`fluid_build._contract_loader.source_contract_path`. + + Each build is recorded on the running ``fluid apply``'s Command Center + report, when there is one (``observability.apply_run``): its status and + the run record it wrote, and why the build phase failed. """ # Deferred imports to avoid circular import at module-load time: # base.py -> python.runner -> base.py (for _resolve_env_placeholders). @@ -909,11 +1112,19 @@ def run_builds_from_args( if _b.get("id"): validate_identifier(_b["id"], kind="build.id") + # The running ``fluid apply``'s run report, if any: each build is recorded on it. + from fluid_build.observability.apply_run import current_apply_run + + report = current_apply_run() + product_id = str(contract.get("id") or "") + # Filter builds if specific ID requested if args.build_id: builds = [b for b in builds if b.get("id") == args.build_id] if not builds: LOG.error(f"Build not found: {args.build_id}") + if report is not None: + report.build_failed(f"build_not_found:{args.build_id}") return 1 # The overlay env the contract above was loaded with, decided once and @@ -936,6 +1147,7 @@ def run_builds_from_args( for build in builds: build_id = build.get("id", "unknown") + runs_before = _run_ids(contract_path.parent, product_id, build_id) if report else None if is_acquisition_build(build): sample_rows = getattr(args, "sample_rows", None) @@ -946,6 +1158,7 @@ def run_builds_from_args( dry_run=args.dry_run, sample_rows=sample_rows, ) + _report_build(report, contract_path.parent, product_id, build_id, result, runs_before) if result == 0: total_executed += 1 else: @@ -987,6 +1200,8 @@ def run_builds_from_args( expected = (contract_path.parent / repository / "dbt_project.yml").resolve() cprint(f"\n⚠️ Build '{build_id}' - dbt project not found: {expected}") total_skipped += 1 + if report is not None: + report.record_build(build_id=build_id, status="skipped") continue result = execute_dbt_build( @@ -1019,6 +1234,8 @@ def run_builds_from_args( "For Python builds, create the script at the expected path above." ) total_skipped += 1 + if report is not None: + report.record_build(build_id=build_id, status="skipped") continue # Execute build @@ -1033,6 +1250,7 @@ def run_builds_from_args( force_run=force_run, ) + _report_build(report, contract_path.parent, product_id, build_id, result, runs_before) if result == 0: total_executed += 1 else: @@ -1067,6 +1285,8 @@ def run_builds_from_args( total_skipped, ) return 0 + if report is not None: + report.build_failed("builds_all_skipped") console_error( f"Every build was skipped ({total_skipped}/{len(builds)}) — nothing " "was transformed and no rows were produced. Fix the missing build " diff --git a/fluid_build/cli/_apply_cc_report.py b/fluid_build/cli/_apply_cc_report.py new file mode 100644 index 00000000..325889b6 --- /dev/null +++ b/fluid_build/cli/_apply_cc_report.py @@ -0,0 +1,537 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""``fluid apply`` reports each run to the Command Center, best effort. + +The Command Center records CLI runs at ``POST /api/v1/executions`` (a run +starts, ``status: running``) and ``PATCH /api/v1/executions/{id}`` (it ends: +``success`` / ``failed``, the Command Center sets ``completed_at`` and +``duration_seconds``). forge-cli has shipped a client for exactly that API +since the observability module landed (``observability/reporter.py``, +``CommandCenterReporter``: async queue, circuit breaker, SSRF host gate), but +nothing called it (``cli/bootstrap.py::get_reporter`` has no callers), so a +Jenkins run left no trace. This wires that client into ``fluid apply`` +rather than adding another one. + +**Where it reports, and as whom.** Wherever ``fluid publish`` already does: +the ``fluid-command-center`` catalog configuration (``FLUID_CC_ENDPOINT``, +``FLUID_API_KEY`` or ``FLUID_BEARER_TOKEN``, and the organization from +``FLUID_CC_ORG_ID``, ``organization_id`` or the ``organization`` slug in the +product's ``fluid.config.yaml``, resolved by the publisher's own +``resolve_organization_id``), so a pipeline configured to publish reports its +applies with no new setting. The reporter's own ``FLUID_COMMAND_CENTER_URL`` +/ ``FLUID_COMMAND_CENTER_API_KEY`` are the fallback. ``FLUID_COMMAND_CENTER_ENABLED=false`` +turns it off. Without an organization the run is not sent: the Command +Center would store it untagged, where no read path shows it. + +**What it says.** The run's product id, contract version and ``fluidVersion``, +the contract hash the Command Center keys contract versions by, the +``--env`` it was applied for, the provider, the apply mode, the state +location, the planned and applied change counts and the address of every +resource the module declares, and the timings. A build-augmented apply +(``--mode amend-and-build`` / ``replace-and-build``) adds each build it ran: +its id, how it ended, and, from the run record the build wrote, the run id, +the table it loaded and the rows it landed (``facets.bigquery_load``, +``facets.landed``, ``records_total``). Its phase is ``build``, and a failed +build names itself in the error event (``build_failed:``). **Never a secret**: no header, +no environment value, no tofu output (its text can carry attribute values); +a failure is reported by its typed event name and exit code only. + +**Best effort.** A Command Center that is down, slow, refusing or +misconfigured costs the apply a warning line and at most the reporter's +timeout, never its exit code (the OpenLineage client's posture: failures +are logged, not raised). Every step below swallows its own errors. +""" + +from __future__ import annotations + +import asyncio +import functools +import logging +import os +import platform +import time +import uuid +from datetime import datetime, timezone +from typing import Any, Callable, Dict, List, Mapping, Optional + +_LOG = logging.getLogger(__name__) + + +_OFF_VALUES = {"0", "false", "no", "off"} + +#: Why nothing is reported when nothing is configured: silent, by design. +_NOT_CONFIGURED = "no Command Center is configured" + + +def current_report() -> Optional["ApplyRunReport"]: + """The report the running ``fluid apply`` fills, or ``None``. + + Held in ``observability.apply_run``, where ``build_runners`` reads it too. + """ + from fluid_build.observability.apply_run import current_apply_run + + return current_apply_run() + + +def _utc_now() -> str: + return datetime.now(timezone.utc).isoformat() + + +def _runner() -> str: + """The CC's ``runner`` tag: ``jenkins`` under Jenkins (which sets + ``JENKINS_URL`` for every build step), else ``cli``.""" + return "jenkins" if os.environ.get("JENKINS_URL") else "cli" + + +class ApplyRunReport: + """One ``fluid apply`` run, as the Command Center's executions API takes it.""" + + def __init__(self, args: Any, logger: logging.Logger) -> None: + self.logger = logger + self.execution_id = str(uuid.uuid4()) + self.started_at = _utc_now() + self._t0 = time.monotonic() + self.contract_path = str(getattr(args, "contract", "") or "") or None + self.environment: Optional[str] = getattr(args, "env", None) or None + self.mode = str(getattr(args, "mode", "") or "") or None + self.provider: Optional[str] = None + self.metadata: Dict[str, Any] = {} + self.result: Dict[str, Any] = {} + self._reporter: Any = None + self._disabled_reason: Optional[str] = None + self._began = False + self._finished = False + #: Why the build phase failed (``build_failed:``), when it did. + self._build_event: Optional[str] = None + + # -- filled by the apply engine ------------------------------------ + + def identify( + self, + *, + contract: Mapping[str, Any], + provider: Optional[str] = None, + environment: Optional[str] = None, + ) -> None: + """What the run is for, as soon as the engine knows it; registers nothing. + + A run refused before :meth:`begin` (a sovereignty refusal in the + emitter, a provider that cannot be resolved) is then still reported + with its product, contract version and platform. + """ + try: + if environment: + self.environment = environment + self.metadata.update(_contract_facts(contract)) + if provider: + self.provider = provider + self.metadata["platform"] = provider + except Exception as exc: # noqa: BLE001 - reporting never fails an apply + _LOG.debug("command center report: identify failed: %s", type(exc).__name__) + + def begin( + self, + *, + contract: Mapping[str, Any], + provider: str, + environment: Optional[str] = None, + state: Optional[str] = None, + ) -> None: + """The engine knows the contract and the provider: register the run.""" + try: + self.provider = provider + if environment: + self.environment = environment + self.metadata.update(_contract_facts(contract)) + self.metadata["platform"] = provider + if state: + self.metadata["state"] = state + self._register() + except Exception as exc: # noqa: BLE001 - reporting never fails an apply + _LOG.debug("command center report: begin failed: %s", type(exc).__name__) + + def record_infra( + self, + *, + planned: Mapping[str, Any], + applied: Optional[Mapping[str, Any]], + resources: List[str], + dry_run: bool, + ) -> None: + """What the plan and the apply did, and which resources the module holds.""" + self.result.update( + { + "planned_changes": {k: int(planned.get(k, 0)) for k in ("add", "change", "remove")}, + "applied_changes": ( + None + if applied is None + else {k: int(applied.get(k, 0)) for k in ("add", "change", "remove")} + ), + "resources": list(resources), + "dry_run": bool(dry_run), + } + ) + + def record_build( + self, + *, + build_id: str, + status: str, + run: Optional[Mapping[str, Any]] = None, + ) -> None: + """One build of a build-augmented apply: ``succeeded``, ``failed`` or ``skipped``. + + ``run`` is the run record the build wrote, if it wrote one; only its + id, the table it loaded, where it landed and the rows are kept. + """ + try: + entry: Dict[str, Any] = {"build_id": str(build_id), "status": str(status)} + entry.update(_build_facts(run)) + self.result.setdefault("builds", []).append(entry) + if status == "failed": + self.build_failed(f"build_failed:{build_id}") + except Exception as exc: # noqa: BLE001 - reporting never fails an apply + _LOG.debug("command center report: record_build failed: %s", type(exc).__name__) + + def build_failed(self, event: str) -> None: + """The build phase failed for ``event``; the first reason is the one reported.""" + if not self._build_event: + self._build_event = str(event) + + # -- the run's end ------------------------------------------------- + + def finish(self, status: str, *, event: Optional[str] = None, exit_code: int = 0) -> None: + """Close the run: ``success`` or ``failed``. Safe to call once, never raises.""" + if self._finished: + return + self._finished = True + try: + if not self._began: + if not self.metadata.get("product_id"): + # Refused before the engine read the contract (plan + # binding, a declared env with no overlay): the base + # document still says which product this run was for. + self.metadata.update(_base_contract_facts(self.contract_path)) + self._register() + reporter = self._reporter + if reporter is None: + return + finished_at = _utc_now() + duration = round(time.monotonic() - self._t0, 3) + timings = { + "started_at": self.started_at, + "finished_at": finished_at, + "duration_seconds": duration, + } + result = dict(self.result, exit_code=int(exit_code), **timings) + if not event and status != "success": + # A build that failed returns an exit code, not an exception. + event = self._build_event + if event: + result["error_event"] = str(event) + if self.result.get("dry_run"): + phase = "plan" + elif self.result.get("builds") or self._build_event: + phase = "build" + else: + phase = "apply" + reporter.update_execution( + self.execution_id, + status=status, + progress=100.0, + current_phase=phase, + error_message=(f"fluid apply failed: {event}" if event else None), + result=result, + ) + reporter.stop(timeout=float(reporter.config.timeout) * 2 + 1) + self._say_outcome(reporter) + except Exception as exc: # noqa: BLE001 - reporting never fails an apply + _LOG.debug("command center report: finish failed: %s", type(exc).__name__) + + # -- internals ----------------------------------------------------- + + def _register(self) -> None: + self._began = True + reporter = self._start_reporter() + if reporter is None: + return + from fluid_build import __version__ + + metadata = dict(self.metadata) + metadata.setdefault("environment", self.environment) + metadata["mode"] = self.mode + reporter.register_execution( + execution_id=self.execution_id, + command="apply", + contract_path=self.contract_path, + provider=self.provider, + environment=self.environment, + runner=_runner(), + cli_version=str(__version__), + python_version=platform.python_version(), + metadata=metadata, + ) + + def _start_reporter(self) -> Any: + if self._reporter is not None: + return self._reporter + config, reason = _command_center_config(self.logger) + if config is None: + self._disabled_reason = reason + if reason and reason != _NOT_CONFIGURED: + # Configured but unusable: say so once. Unconfigured is silent. + self._say(f" command center: run not reported ({reason})") + return None + from fluid_build.observability.reporter import CommandCenterReporter + + reporter = CommandCenterReporter(config) + reporter.start() + if not reporter.running: + # Refused by the reporter itself: its SSRF host gate, or no + # ``requests``. It has logged which. + self._disabled_reason = "the Command Center reporter did not start" + self._say(f" command center: run not reported ({self._disabled_reason})") + return None + self._reporter = reporter + return reporter + + def _say_outcome(self, reporter: Any) -> None: + sent, failed = reporter.stats.get("sent", 0), reporter.stats.get("failed", 0) + if failed or sent < 2: + self._say( + f" command center: run {self.execution_id} not fully reported " + f"({sent} of 2 requests accepted); the apply's result is unaffected" + ) + else: + self._say(f" command center: run {self.execution_id} reported") + + @staticmethod + def _say(line: str) -> None: + try: + from fluid_build.cli.console import cprint + + cprint(line) + except Exception: # noqa: BLE001 + pass + + +def _build_facts(run: Optional[Mapping[str, Any]]) -> Dict[str, Any]: + """The run id, the loaded table, the landed destinations and rows of a run record. + + Nothing else from the record: its facets can carry engine output. + """ + if not isinstance(run, Mapping): + return {} + facts: Dict[str, Any] = {} + if run.get("run_id"): + facts["run_id"] = str(run["run_id"]) + facets = run.get("facets") if isinstance(run.get("facets"), Mapping) else {} + load = facets.get("bigquery_load") + if isinstance(load, Mapping): + if load.get("table"): + facts["table"] = str(load["table"]) + if isinstance(load.get("rows"), int): + facts["rows"] = int(load["rows"]) + landed = facets.get("landed") + destinations = landed.get("destinations") if isinstance(landed, Mapping) else None + if isinstance(destinations, Mapping) and destinations: + facts["destinations"] = {str(k): str(v) for k, v in destinations.items()} + succeeded = str(run.get("state") or "").lower() == "succeeded" + if "rows" not in facts and succeeded and isinstance(run.get("records_total"), int): + facts["rows"] = int(run["records_total"]) + return facts + + +def _contract_facts(contract: Mapping[str, Any]) -> Dict[str, Any]: + """Identity facts only: nothing from the contract's body beyond its id and versions.""" + facts: Dict[str, Any] = { + "product_id": contract.get("id"), + "product_name": contract.get("name"), + "contract_version": contract.get("version"), + "fluid_version": contract.get("fluidVersion"), + } + try: + import yaml + + from fluid_build.providers.catalogs.fluid_cc.provider import command_center_contract_hash + + facts["contract_hash"] = command_center_contract_hash( + yaml.safe_dump(dict(contract), sort_keys=False) + ) + except Exception: # noqa: BLE001 - a hash is a join key, not required + pass + return facts + + +def _base_contract_facts(contract_path: Optional[str]) -> Dict[str, Any]: + """Identity facts from the base contract (or a plan's embedded one), no overlay. + + For a run that failed before the engine loaded the contract, possibly + because its overlay could not be applied. The product id and versions + are the base's (an overlay rebinds, it does not rename). No contract + hash: the Command Center keys versions by the hash of the compiled + contract, which this run never produced. + """ + if not contract_path: + return {} + try: + if contract_path.endswith(".json"): + import json + + with open(contract_path, encoding="utf-8") as handle: + plan = json.load(handle) + contract = plan.get("contract") if isinstance(plan, dict) else None + else: + from fluid_build.loader import load_contract + + contract = load_contract(contract_path) + except Exception: # noqa: BLE001 - an unreadable contract is simply not described + return {} + if not isinstance(contract, Mapping): + return {} + facts = _contract_facts(contract) + facts.pop("contract_hash", None) + return {k: v for k, v in facts.items() if v is not None} + + +def _command_center_config(logger: logging.Logger): + """``(CommandCenterConfig, None)`` to report with, or ``(None, reason)``.""" + from fluid_build.observability.config import CommandCenterConfig + + enabled = os.environ.get("FLUID_COMMAND_CENTER_ENABLED") + if enabled is not None and enabled.strip().lower() in _OFF_VALUES: + return None, "FLUID_COMMAND_CENTER_ENABLED is off" + + published = _publish_path_config(logger) + if published is not None: + return published + env_config = CommandCenterConfig.from_environment() + if env_config.is_configured(): + org = (os.environ.get("FLUID_CC_ORG_ID") or "").strip() + if org and _header_safe(org): + env_config.headers = {**env_config.headers, "X-Organization-Id": org} + return env_config, None + return None, _NOT_CONFIGURED + + +def _publish_path_config(logger: logging.Logger): + """The ``fluid publish`` target, with its credential and its organization. + + ``None`` when the catalog configuration names no Command Center at all; + ``(None, reason)`` when it names one this run cannot report to. + """ + from fluid_build.config_manager import COMMAND_CENTER_CANONICAL_NAME, FluidConfig + from fluid_build.observability.config import CommandCenterConfig + + try: + catalog = FluidConfig().get_catalog_config(COMMAND_CENTER_CANONICAL_NAME) or {} + except Exception: # noqa: BLE001 - an unreadable config is "not configured" + return None + endpoint = str(catalog.get("endpoint") or "").strip() + if not endpoint or not catalog.get("enabled", True): + return None + # The built-in defaults name ``http://localhost:8000`` with no key, so an + # endpoint alone is not a configured Command Center: a credential is. + from fluid_build.providers.common import get_auth_headers + + probe = get_auth_headers(endpoint, catalog.get("auth")) + if not (probe.get("X-API-Key") or probe.get("Authorization")): + return None + # The reporter's SSRF gate, before the organization lookup below sends the + # credential anywhere: loopback or an allow-listed host, never a private + # or cloud-metadata address (observability/reporter.py). + from fluid_build.observability.reporter import _command_center_host_allowed + + if not _command_center_host_allowed(endpoint): + return None, ( + "its host resolves to a private or metadata address; allow it with " + "FLUID_COMMAND_CENTER_HOST_ALLOWLIST" + ) + try: + from fluid_build.providers.catalogs.fluid_cc import FluidCommandCenterProvider + + provider = FluidCommandCenterProvider(catalog) + org_id = _run_coroutine(provider.resolve_organization_id()) + headers = dict(provider._headers()) + except Exception as exc: # noqa: BLE001 - typed CC errors say what is missing + return None, f"the Command Center organization could not be settled ({type(exc).__name__})" + api_key = headers.pop("X-API-Key", None) + extra = {k: v for k, v in headers.items() if k in ("Authorization", "X-Organization-Id")} + if org_id and _header_safe(org_id): + extra["X-Organization-Id"] = org_id + timeout = _timeout(catalog.get("timeout")) + config = CommandCenterConfig( + url=provider.endpoint, api_key=api_key, timeout=timeout, headers=extra + ) + if not config.is_configured(): + return None, "the Command Center catalog configuration has no credential" + return config, None + + +def _timeout(value: Any) -> int: + try: + seconds = int(value) + except (TypeError, ValueError): + return 5 + return max(1, min(seconds, 10)) + + +def _header_safe(value: str) -> bool: + from fluid_build.providers.catalogs.fluid_cc.provider import _is_valid_org_id + + return bool(_is_valid_org_id(value)) + + +def _run_coroutine(coro: Any) -> Any: + """Run ``coro`` to completion from sync code, even under a running loop.""" + try: + asyncio.get_running_loop() + except RuntimeError: + return asyncio.run(coro) + import concurrent.futures + + with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool: + return pool.submit(asyncio.run, coro).result() + + +def reports_apply_run(fn: Callable[..., int]) -> Callable[..., int]: + """Decorate ``fluid apply``'s ``run``: one Command Center run per invocation.""" + + @functools.wraps(fn) + def wrapper(args: Any, logger: logging.Logger) -> int: + from fluid_build.observability.apply_run import ( + reset_current_apply_run, + set_current_apply_run, + ) + + from ._common import CLIError + + report = ApplyRunReport(args, logger) + token = set_current_apply_run(report) + try: + rc = fn(args, logger) + except CLIError as exc: + report.finish("failed", event=exc.event, exit_code=exc.exit_code) + raise + except BaseException as exc: + report.finish("failed", event=type(exc).__name__, exit_code=1) + raise + else: + report.finish("success" if rc == 0 else "failed", exit_code=rc) + return rc + finally: + reset_current_apply_run(token) + + return wrapper diff --git a/fluid_build/cli/_apply_opentofu_engine.py b/fluid_build/cli/_apply_opentofu_engine.py index 8c14a04b..a87c91be 100644 --- a/fluid_build/cli/_apply_opentofu_engine.py +++ b/fluid_build/cli/_apply_opentofu_engine.py @@ -29,7 +29,7 @@ import json import logging from contextlib import contextmanager -from dataclasses import dataclass +from dataclasses import dataclass, replace from pathlib import Path from typing import Any, Dict, List, Mapping, Optional, Tuple @@ -38,13 +38,24 @@ from fluid_build.iac.backend import ( STATE_BACKEND_ENV, backend_location, + legacy_default_backend, parse_backend, resolve_state_backend_spec, ) from fluid_build.iac.base import UnsupportedBindingError from fluid_build.iac.credentials import build_tofu_env, credential_report from fluid_build.iac.naming import safe_ident +from fluid_build.iac.state_migration import ( + PENDING, + StateMigrationError, + StateReconciliation, + other_clouds, + read_state, + records_backend, +) +from fluid_build.iac.state_migration import reconcile_state_key as _reconcile_state +from ._apply_cc_report import current_report from ._common import CLIError, load_contract_with_overlay, resolve_env_templates_in_contract from ._logging import info, warn from .generate_iac import _resolve_provider, native_actions @@ -63,6 +74,14 @@ def apply_via_opentofu(args, logger: logging.Logger) -> int: contract = _load_contract(args, logger) provider = _resolve_provider(contract, getattr(args, "provider", None) or "auto") + report = current_report() + if report is not None: + # Known from here on, so a refusal before the run is registered (a + # sovereignty refusal in the emitter, an init that fails) still + # reaches the Command Center with its product, version and platform. + # A run refused earlier is described from the base contract instead + # (``ApplyRunReport.finish``). + report.identify(contract=contract, provider=provider, environment=_applied_env(args)) plugin = get_iac_plugin(provider) if plugin is None: @@ -129,14 +148,64 @@ def apply_via_opentofu(args, logger: logging.Logger) -> int: cprint(f" state: {state_line}") cprint(f" credentials: {', '.join(present) if present else 'none detected in environment'}") - init = runner.tofu_init(str(workdir), backend=backend is not None, env=env) + # The Command Center hears about the run now that the product, the + # provider and the environment are known (best effort; see + # cli/_apply_cc_report.py). Addresses and counts only, never tofu output. + if report is not None: + report.begin( + contract=contract, + provider=provider, + environment=_applied_env(args), + state=target.location, + ) + + init = runner.tofu_init( + str(workdir), + backend=backend is not None, + env=env, + reconfigure=recorded_legacy_backend(target), + ) if not init.ok: raise CLIError(1, "opentofu_init_failed", {"error": _tail(init.stderr or init.stdout)}) + dry_run = bool(getattr(args, "dry_run", False)) + if dry_run: + # A dry-run is plan only and writes no state, so it never moves any + # either: while the move is pending it plans against the old key, the + # read-only path ``fluid diff`` takes (``read_target``). The copy is + # left to the first real apply, which a plan-only CI role with + # read-only state access never runs. + read = read_target(target, provider, env, logger) + if read is not target: + target = read + module, actions = emit_module(plugin, contract, target, logger) + module_path.write_text(module, encoding="utf-8") + init = runner.tofu_init(str(workdir), backend=True, env=env, reconfigure=True) + if not init.ok: + raise CLIError( + 1, "opentofu_init_failed", {"error": _tail(init.stderr or init.stdout)} + ) + else: + # State a previous release kept at the key without the provider moves + # to this provider's key (OpenTofu's own ``init -migrate-state``), so + # the first apply after the upgrade does not plan every resource as + # new (see ``iac.state_migration``). Before anything reads the state. + reconcile_state_key(target, provider, env, logger, migrate=True) + + # One state, two clouds: a key that does not name the provider (the + # shared ``fluid/terraform.tfstate`` a bucket-only --state-backend gives a + # contract without packaging, or an explicit key used for both) can hold + # the other cloud's resources, which this plan would destroy. + guard_state_shared_with_another_cloud(target, provider, env) + # Pre-plan region guard. A module now pins the region its bindings name, # and moving a contract's resources to it would not show as a destroy. _guard_region_move(plugin, contract, str(workdir), env) + # A dataset an older forge-cli applied with an authoritative access list has + # its stale grants revoked once, before its grants become member resources. + _reconcile_with_state(plugin, module_path, str(workdir), env, logger) + # Pre-plan ownership-transition guard (RFC-packaging-modes.md file 10). # Runs BEFORE _adopt_existing — brownfield adoption is precisely the # mechanism that would re-own a shared pool — and before `tofu plan`, @@ -153,6 +222,22 @@ def apply_via_opentofu(args, logger: logging.Logger) -> int: raise CLIError(1, "opentofu_plan_failed", {"error": _tail(plan.stderr or plan.stdout)}) changes = runner.change_summary(plan) cprint(f"\n tofu plan: +{changes['add']} ~{changes['change']} -{changes['remove']}") + if report is not None: + report.record_infra( + planned=changes, + applied=None, + resources=_module_addresses(module), + dry_run=bool(getattr(args, "dry_run", False)), + ) + + # Revoking a grant destroys a member resource and loses no data: only the + # removals of data-bearing resources reach the data-loss gate. + data_changes, revoked = _data_bearing_changes(changes, runner.planned_removals(plan)) + if revoked: + cprint( + f" {len(revoked)} access grant(s) or policy tag(s) removed (access revoked, no " + "data lost; not gated): " + ", ".join(revoked) + ) # Report what the plan's ``lifecycle.ignore_changes`` deliberately hides. # Without this, a contract whose column types no longer match the live @@ -164,22 +249,22 @@ def apply_via_opentofu(args, logger: logging.Logger) -> int: # Data-loss gate — `tofu` has no CTAS/CLONE data snapshot (see # AUTOGEN_SPIKE.md, risk R1), so a destructive plan fails closed. allow_data_loss = bool(getattr(args, "allow_data_loss", False)) - if _data_loss_blocked(changes, allow_data_loss): + if _data_loss_blocked(data_changes, allow_data_loss): raise CLIError( 1, "opentofu_data_loss_gate", { - "error": f"plan destroys {changes['remove']} resource(s); `tofu` does not " + "error": f"plan destroys {data_changes['remove']} resource(s); `tofu` does not " "snapshot data — re-run with --allow-data-loss to proceed" }, ) - if allow_data_loss and int(changes.get("remove", 0)) > 0: + if allow_data_loss and int(data_changes.get("remove", 0)) > 0: # Audit-trail: every destructive apply through the override is # logged at WARNING so CI log-scrapers + operators have a # paper-trail. Matches the same posture as the native engine's # _verify_plan_binding bypass warning. cprint( - f"\n ⚠️ --allow-data-loss: {changes['remove']} resource(s) will be " + f"\n ⚠️ --allow-data-loss: {data_changes['remove']} resource(s) will be " "DESTROYED and no pre-replace snapshot is taken — this engine has " "no CTAS/CLONE step, so `fluid rollback` will have no restore point." ) @@ -188,7 +273,7 @@ def apply_via_opentofu(args, logger: logging.Logger) -> int: "will be destroyed by `tofu apply` with NO pre-replace snapshot " "(`fluid rollback` has no restore point). Provider: %s. Plan changes: " "+%d ~%d -%d.", - int(changes.get("remove", 0)), + int(data_changes.get("remove", 0)), provider, changes["add"], changes["change"], @@ -198,11 +283,11 @@ def apply_via_opentofu(args, logger: logging.Logger) -> int: logger, "opentofu_destructive_gate_override", provider=provider, - resources_to_destroy=int(changes.get("remove", 0)), + resources_to_destroy=int(data_changes.get("remove", 0)), **changes, ) - if bool(getattr(args, "dry_run", False)): + if dry_run: cprint("\ndry-run: plan only — not applying.") info(logger, "opentofu_apply_dry_run", provider=provider, **changes) return 0 @@ -225,11 +310,45 @@ def apply_via_opentofu(args, logger: logging.Logger) -> int: record(applied) cprint(f"\n tofu apply complete: +{applied['add']} ~{applied['change']} -{applied['remove']}") + if report is not None: + report.record_infra( + planned=changes, applied=applied, resources=_module_addresses(module), dry_run=False + ) info(logger, "opentofu_apply_ok", provider=provider, **applied) return 0 +def _applied_env(args) -> Optional[str]: + """The overlay env this apply is for: ``--env``, else the bundle's own.""" + env = getattr(args, "env", None) + if env: + return str(env) + bundle = getattr(args, "bundle", None) + if not bundle: + return None + try: + from fluid_build.forge.core.bundle import read_bundle_source + + source = read_bundle_source(Path(bundle)) or {} + except Exception: # noqa: BLE001 - the env is a report field, not a gate + return None + return str(source["env"]) if source.get("env") else None + + +def _module_addresses(module_text: str) -> List[str]: + """``type.name`` of every resource the emitted module declares.""" + try: + doc = json.loads(module_text) + except ValueError: + return [] + out: List[str] = [] + for rtype, by_name in sorted((doc.get("resource") or {}).items()): + if isinstance(by_name, dict): + out.extend(f"{rtype}.{name}" for name in sorted(by_name)) + return out + + @dataclass(frozen=True) class StateTarget: """Where ``fluid apply`` keeps one contract's OpenTofu workdir and state.""" @@ -239,6 +358,9 @@ class StateTarget: backend: Optional[Dict[str, Any]] #: ``--state-backend``, ``FLUID_STATE_BACKEND`` or ``default``. origin: str + #: Where a release before the provider-keyed default kept this state, when + #: that differs from ``backend`` (``iac.backend.legacy_default_backend``). + legacy_backend: Optional[Dict[str, Any]] = None @property def location(self) -> str: @@ -269,12 +391,21 @@ def resolve_state_target(args, contract: Mapping[str, Any], provider: str) -> St # bucket instead of the workspace it wipes after every run. That job # applies every product with the one value, so a bucket-only value keys # state per contract for all of them, packaging block or not. + # + # The provider is part of every per-contract default key, so one contract + # applied to aws and to gcp (``--env`` overlays) keeps two states: with + # one, each cloud's plan read the other's resources as orphans to destroy. backend_spec, backend_origin = resolve_state_backend_spec(getattr(args, "state_backend", None)) + per_contract = backend_origin == STATE_BACKEND_ENV try: backend = parse_backend( backend_spec, contract, - per_contract_default=backend_origin == STATE_BACKEND_ENV, + per_contract_default=per_contract, + provider=provider, + ) + legacy = legacy_default_backend( + backend_spec, contract, per_contract_default=per_contract, provider=provider ) except ValueError as exc: raise CLIError( @@ -292,7 +423,139 @@ def resolve_state_target(args, contract: Mapping[str, Any], provider: str) -> St / provider / safe_ident(contract.get("id") or "contract") ) - return StateTarget(workdir=workdir, backend=backend, origin=backend_origin) + return StateTarget( + workdir=workdir, backend=backend, origin=backend_origin, legacy_backend=legacy + ) + + +def recorded_legacy_backend(target: StateTarget) -> bool: + """True when the workdir's ``.terraform/`` recorded the pre-provider-key backend. + + A plain ``tofu init`` stops there with "Backend configuration changed"; + the caller inits with ``-reconfigure`` instead, and :func:`reconcile_state_key` + moves the state, so nothing is left behind. + """ + return target.legacy_backend is not None and records_backend( + target.workdir, target.legacy_backend + ) + + +def reconcile_state_key( + target: StateTarget, + provider: str, + env: Mapping[str, str], + logger: logging.Logger, + *, + migrate: bool, +) -> Optional[StateReconciliation]: + """Bring state a previous release kept without the provider in its key along. + + Runs after the workdir's ``tofu init`` on ``target.backend``. + ``migrate=True`` (``fluid apply``) copies it with ``tofu init + -migrate-state``; ``migrate=False`` (the read-only drift pass) only + reports it, and :func:`read_target` then points the read at the old key. + ``None`` when there is no old key to look at. + """ + if target.backend is None or target.legacy_backend is None: + return None + try: + outcome = _reconcile_state( + workdir=target.workdir, + current=target.backend, + legacy=target.legacy_backend, + provider=provider, + env=env, + migrate=migrate, + logger=logger, + ) + except StateMigrationError as exc: + raise CLIError( + 1, + exc.code, + { + "error": str(exc), + "state": backend_location(target.backend), + "legacy_state": backend_location(target.legacy_backend), + }, + ) + line = outcome.summary() + if line: + cprint(f" state move: {line}") + info( + logger, + "opentofu_state_key_reconciled", + outcome=outcome.outcome, + state=backend_location(target.backend), + legacy_state=backend_location(target.legacy_backend), + resources=outcome.resources, + ) + return outcome + + +def _key_names_provider(backend: Mapping[str, Any], provider: str) -> bool: + """True when the backend's key (or GCS prefix) has ``provider`` as a path segment.""" + if "s3" in backend: + path = str((backend.get("s3") or {}).get("key") or "") + elif "gcs" in backend: + path = str((backend.get("gcs") or {}).get("prefix") or "") + else: + return False + return provider in path.split("/") + + +def guard_state_shared_with_another_cloud( + target: StateTarget, provider: str, env: Mapping[str, str] +) -> None: + """Refuse a remote state whose key names no provider and holds another cloud's resources. + + The per-provider default keys (``fluid///...``) cannot be + shared by two clouds, and local state lives in a per-provider workdir, so + only a remote key that does not name the provider is read (one ``tofu + state pull``). Whose resources they are is the migration's own rule + (``state_migration.other_clouds``). Without this, the gcp plan on a key + the aws apply wrote reads the aws resources as orphans, and + ``--allow-data-loss`` destroys them. + """ + if target.backend is None or _key_names_provider(target.backend, provider): + return + location = backend_location(target.backend) + try: + doc = read_state(target.workdir, env) + except StateMigrationError as exc: + raise CLIError(1, exc.code, {"error": str(exc), "state": location}) + others = sorted(other_clouds(doc.resources, provider)) + if not others: + return + raise CLIError( + 1, + "state_shared_with_another_provider", + { + "error": ( + f"{location} holds resources of the {', '.join(others)} provider, and this is " + f"the {provider} apply: its plan would read them as orphans to destroy. Give " + "each provider its own state, with a key that names the provider in " + "--state-backend (for example fluid///terraform.tfstate) or a " + f"bucket-only {STATE_BACKEND_ENV}, which keys state by contract and provider" + ), + "state": location, + "providers": others, + }, + ) + + +def read_target( + target: StateTarget, provider: str, env: Mapping[str, str], logger: logging.Logger +) -> StateTarget: + """The target a read-only caller reads: the old key while its move is pending. + + Call after the workdir's init on ``target.backend``. When this returns a + different target, the caller re-emits the module on its backend and + re-inits with ``-reconfigure``. + """ + outcome = reconcile_state_key(target, provider, env, logger, migrate=False) + if outcome is None or outcome.outcome != PENDING: + return target + return replace(target, backend=dict(outcome.legacy)) def emit_module( @@ -686,6 +949,101 @@ def _report_suppressed_drift( warn(logger, "opentofu_suppressed_drift_reported", tables=[r["table"] for r in drift]) +#: Resource types whose destroy revokes access and deletes no data: a member grant, +#: a Lake Formation permission, a policy tag or its taxonomy. Removing one is how a +#: contract revokes a reader or lifts a column restriction, so the data-loss gate +#: does not count it (it did, and a revocation needed --allow-data-loss, the flag +#: that also lets the same plan drop tables). The same split policy-as-code gates +#: over ``tofu show -json`` make by resource type (OPA's Terraform tutorial weighs +#: deletes per type). A key's IAM grant is not here: without it BigQuery cannot +#: decrypt the table. +ACCESS_ONLY_RESOURCE_TYPES = frozenset( + { + "google_bigquery_dataset_iam_member", + "google_bigquery_table_iam_member", + "google_storage_bucket_iam_member", + "google_data_catalog_policy_tag_iam_member", + "google_data_catalog_policy_tag", + "google_data_catalog_taxonomy", + "aws_lakeformation_permissions", + } +) + + +def _data_bearing_changes( + changes: Mapping[str, int], removals: List[Tuple[str, str]] +) -> Tuple[Dict[str, int], List[str]]: + """``(changes with only data-bearing removals, addresses of access-only removals)``. + + Fails closed: when the plan's per-resource events do not account for every + removal in its summary (an older ``tofu``, a truncated stream), every removal + counts, as before. + """ + counted = {key: int(changes.get(key, 0)) for key in ("add", "change", "remove")} + if len(removals) != counted["remove"]: + return counted, [] + revoked = [addr for addr, kind in removals if kind in ACCESS_ONLY_RESOURCE_TYPES] + counted["remove"] -= len(revoked) + return counted, revoked + + +def _reconcile_with_state( + plugin: Any, + module_path: Path, + workdir: str, + env: Mapping[str, str], + logger: logging.Logger, + *, + announce: bool = True, +) -> List[Dict[str, Any]]: + """Let the plugin patch the written module against the current state. + + Optional plugin capability ``reconcile_state(module, state_resources)``: the GCP + plugin uses it once per dataset an older forge-cli applied with an authoritative + access list, so a grant the contract no longer makes is revoked rather than left + unmanaged (``iac/providers/gcp.py::reconcile_legacy_dataset_access``). A no-op for + a fresh workdir and for plugins without it. ``fluid diff`` calls it too + (``announce=False``), so its plan shows the revocation the apply will make. + """ + reconcile = getattr(plugin, "reconcile_state", None) + if not callable(reconcile): + return [] + state = runner.tofu_state_resources(workdir, env=env) + if not state: + return [] + module = json.loads(module_path.read_text(encoding="utf-8")) + reports = reconcile(module, state) + if not reports: + return [] + blocked = [r for r in reports if r.get("blocked")] + if blocked: + raise CLIError( + 1, + "opentofu_dataset_access_unreconciled", + { + "error": "state holds dataset(s) an older forge-cli applied with an " + "authoritative access list, and every entry of it is a grant the contract " + "no longer makes, so it cannot be narrowed in place: " + + "; ".join(f"{r['dataset']}: {', '.join(r['revoked'])}" for r in blocked), + "remediation": [ + "Revoke those entries on the dataset (bq update or the console), or " + "keep one of them in accessPolicy for this apply, then re-run." + ], + }, + ) + module_path.write_text(json.dumps(module, indent=2), encoding="utf-8") + if not announce: + return reports + for report in reports: + cprint( + f"\n dataset {report['dataset']}: its access list was written by an older " + "forge-cli; this apply revokes the entries no grant of the contract covers: " + + ", ".join(report["revoked"]) + ) + warn(logger, "gcp_dataset_access_reconciled", datasets=reports) + return reports + + def _data_loss_blocked(changes: Mapping[str, int], allow_data_loss: bool) -> bool: """A plan that removes resources is blocked unless ``--allow-data-loss`` is set.""" return int(changes.get("remove", 0)) > 0 and not allow_data_loss diff --git a/fluid_build/cli/_diff_state.py b/fluid_build/cli/_diff_state.py index 2fdef8ca..cf0c78c3 100644 --- a/fluid_build/cli/_diff_state.py +++ b/fluid_build/cli/_diff_state.py @@ -184,7 +184,14 @@ def _plan_and_classify( from fluid_build.iac import runner from fluid_build.iac.credentials import build_tofu_env - from ._apply_opentofu_engine import _guard_region_move, _tail, emit_module + from ._apply_opentofu_engine import ( + _guard_region_move, + _reconcile_with_state, + _tail, + emit_module, + read_target, + recorded_legacy_backend, + ) workdir: Path = target.workdir location = target.location @@ -200,18 +207,45 @@ def _plan_and_classify( env = build_tofu_env() env.update(plugin.credential_env(env)) - init = runner.tofu_init(str(workdir), backend=target.backend is not None, env=env) + init = runner.tofu_init( + str(workdir), + backend=target.backend is not None, + env=env, + reconfigure=recorded_legacy_backend(target), + ) if not init.ok: return _error( "OpenTofu could not initialise the apply's workdir: " + _tail(init.stderr or init.stdout, _MAX_DETAIL_CHARS), location, ) + # State a previous release kept at the key without the provider is + # read where it is until ``fluid apply`` moves it: this pass writes + # no state, and a drift gate that runs before the first upgraded + # apply must still see the real one. + try: + read = read_target(target, plugin.name, env, logger) + except CLIError as exc: + return _error(_cli_error_text(exc), location) + if read is not target: + target, location = read, read.location + module, _actions = emit_module(plugin, contract, target, logger) + module_path.write_text(module, encoding="utf-8") + init = runner.tofu_init(str(workdir), backend=True, env=env, reconfigure=True) + if not init.ok: + return _error( + "OpenTofu could not initialise the apply's workdir on the old state key: " + + _tail(init.stderr or init.stdout, _MAX_DETAIL_CHARS), + location, + ) # The apply refuses to run when state holds the contract's resources in # another region; the refresh would find them gone and this pass would # call that "deleted outside the apply". try: _guard_region_move(plugin, contract, str(workdir), env) + # The apply's one-time revocation of an older access list's stale + # grants, so this plan shows what the apply will do. + _reconcile_with_state(plugin, module_path, str(workdir), env, logger, announce=False) except CLIError as exc: return _error(_cli_error_text(exc), location) diff --git a/fluid_build/cli/_logging.py b/fluid_build/cli/_logging.py index 0e91c45d..72eafe24 100644 --- a/fluid_build/cli/_logging.py +++ b/fluid_build/cli/_logging.py @@ -38,19 +38,36 @@ def setup_logging(level: str = "INFO", file: str | None = None) -> logging.Logge return logger +#: The envelope keys every event carries. A payload key of the same name is +#: kept as ``extra_`` instead of replacing the envelope's (the event +#: name is ``message``; a provider result with its own ``message`` would +#: otherwise have renamed the event). +_ENVELOPE_KEYS = frozenset({"time", "level", "name", "message"}) + + def _event(level: str, name: str, payload: Dict[str, Any]) -> str: + # Renamed, not dropped: the pattern of WebbPulse/webbpulse-python#129 + # (``extra_`` for an ``extra`` key that collides with a LogRecord + # attribute). structlog's hynek/structlog#842 drops such keys instead, + # which would lose the provider's own explanation here. + fields = {(f"extra_{k}" if k in _ENVELOPE_KEYS else k): v for k, v in payload.items()} return json.dumps( { "time": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), "level": level, "name": "fluid.cli", "message": name, - **payload, + **fields, } ) -def info(logger: logging.Logger, message: str, **payload: Any) -> None: +# ``logger`` and ``message`` are positional-only (PEP 570), so a payload may +# carry keys of those names: ``info(logger, "policy_apply_result", **res)`` +# raised ``TypeError: info() got multiple values for argument 'message'`` +# for every provider whose result has a ``message`` (GCP's policy applier +# does), and failed stage 8 of every generated gcp pipeline. +def info(logger: logging.Logger, message: str, /, **payload: Any) -> None: """Emit a structured INFO event to the log sink. Routed at DEBUG level for the human-facing console handler so the @@ -67,7 +84,7 @@ def info(logger: logging.Logger, message: str, **payload: Any) -> None: logger.debug(_event("INFO", message, payload)) -def warn(logger: logging.Logger, message: str, **payload: Any) -> None: +def warn(logger: logging.Logger, message: str, /, **payload: Any) -> None: """Emit a structured WARNING event — stays at WARNING level. Warnings are user-relevant ("we didn't break, but you should know @@ -76,5 +93,5 @@ def warn(logger: logging.Logger, message: str, **payload: Any) -> None: logger.warning(_event("WARNING", message, payload)) -def error(logger: logging.Logger, message: str, **payload: Any) -> None: +def error(logger: logging.Logger, message: str, /, **payload: Any) -> None: logger.error(_event("ERROR", message, payload)) diff --git a/fluid_build/cli/_verify_athena.py b/fluid_build/cli/_verify_athena.py index 4c54105a..89dbfc08 100644 --- a/fluid_build/cli/_verify_athena.py +++ b/fluid_build/cli/_verify_athena.py @@ -802,8 +802,15 @@ def _run_total(record: Mapping[str, Any]) -> Optional[int]: return total if total >= 0 else None +#: ``(record, table location) -> landed there: yes, no, or unknown``. +WroteInto = Callable[[Optional[Mapping[str, Any]], str], Optional[bool]] + + def _landed_since_full_load( - records: List[Optional[Dict[str, Any]]], table_location: str + records: List[Optional[Dict[str, Any]]], + table_location: str, + *, + wrote_into: Optional[WroteInto] = None, ) -> Tuple[int, List[str], bool]: """Rows the runs since the last full load put in the table, those runs, and whether the walk reached that full load. @@ -818,10 +825,11 @@ def _landed_since_full_load( floor, because an append removes no row; it is the whole floor only when the walk reached a full load. """ + wrote_into = wrote_into or _wrote_into total = 0 runs: List[str] = [] for record in records: - wrote = _wrote_into(record, table_location) + wrote = wrote_into(record, table_location) if wrote is False: continue if record is None or wrote is None: @@ -838,16 +846,30 @@ def _landed_since_full_load( def _landed_rows( - contract: Mapping[str, Any], expose_id: str, workdir: Path, table_location: str + contract: Mapping[str, Any], + expose_id: str, + workdir: Path, + table_location: str, + *, + wrote_into: Optional[WroteInto] = None, + embedded_sql_records: bool = False, ) -> Tuple[Optional[int], Dict[str, Any]]: """What the landing build's runs put in ``table_location``, when a record says. Compared only for an acquisition build, and only with a run that recorded landing its rows in this table and counting them at the write. Anything - else is reported with the reason and never gates. + else is reported with the reason and never gates. ``wrote_into`` decides + which runs landed in the table: by ``facets.landed.destinations`` here, + by ``facets.bigquery_load.table`` for a BigQuery table + (``_verify_bigquery``). With ``embedded_sql_records`` (the BigQuery + verifier), an embedded-SQL build's runs are read too: its BigQuery load + records what it landed as an acquisition run does + (``_embedded_sql_io.write_bigquery_run_record``). """ + wrote_into = wrote_into or _wrote_into from fluid_build.build_runners._acquisition_common import is_acquisition_build from fluid_build.build_runners._ids import IdentifierViolation, validate_identifier + from fluid_build.build_runners.base import is_embedded_sql_build builds = _landing_builds(contract, expose_id) if not builds: @@ -859,7 +881,8 @@ def _landed_rows( "note": f"{len(builds)} builds write this expose ({ids}); no single run to compare", } build = builds[0] - if not is_acquisition_build(dict(build)): + counted_embedded_sql = embedded_sql_records and is_embedded_sql_build(dict(build)) + if not is_acquisition_build(dict(build)) and not counted_embedded_sql: # A transformation's run record counts what its engine ran: the dbt # runner's records_total is the number of nodes in run_results.json. return None, { @@ -885,7 +908,7 @@ def _landed_rows( # Runs that landed in another target (a local run from the same directory) # never touched this table, so the newest run that may have is the one. skipped = 0 - while skipped < len(records) and _wrote_into(records[skipped], table_location) is False: + while skipped < len(records) and wrote_into(records[skipped], table_location) is False: skipped += 1 if skipped == len(records): return None, { @@ -912,7 +935,7 @@ def _landed_rows( } if skipped: info["other_target_runs_skipped"] = skipped - if _wrote_into(record, table_location) is None: + if wrote_into(record, table_location) is None: info["source"] = "none" info["note"] = ( f"run {record.get('run_id')} does not record where it landed its rows, so it " @@ -936,7 +959,7 @@ def _landed_rows( return None, info if info["rule"] == RULE_AT_LEAST_CUMULATIVE: total, info["runs"], info["reached_full_load"] = _landed_since_full_load( - records[skipped:], table_location + records[skipped:], table_location, wrote_into=wrote_into ) return total, info @@ -965,32 +988,37 @@ def _row_count_dimension( table_id: str, *, reference_only: bool = False, + engine: str = "Athena", ) -> Dict[str, Any]: - """``status`` is ``pass``, ``fail`` (CRITICAL) or ``info`` (reported, never gated).""" + """``status`` is ``pass``, ``fail`` (CRITICAL) or ``info`` (reported, never gated). + + ``engine`` names what counted the rows in the messages (``BigQuery`` for + ``_verify_bigquery``, which holds a table to its runs by the same rules). + """ run = _runs_phrase(info) rule = info.get("rule") if count == 0 and landed is None and reference_only: # Bug 6's case with the table present: apply creates the Glue table, # and the pipeline that owns the rows may not have run yet. message = ( - f"Athena counted 0 rows in {table_id}; not gated, because the contract is " + f"{engine} counted 0 rows in {table_id}; not gated, because the contract is " "reference-only and the pipeline that owns the table may not have written it yet" ) status = "info" elif count == 0: - message = f"Athena counted 0 rows in {table_id}" + message = f"{engine} counted 0 rows in {table_id}" if landed is not None: message += f"; {run} landed {landed:,}" status = "fail" elif landed is not None and rule == RULE_EQUAL and count != landed: message = ( - f"Athena counted {count:,} rows in {table_id}; {run} landed {landed:,} " + f"{engine} counted {count:,} rows in {table_id}; {run} landed {landed:,} " "and is a full refresh" ) status = "fail" elif landed is not None and rule == RULE_AT_LEAST_CUMULATIVE and count < landed: message = ( - f"Athena counted {count:,} rows in {table_id}; {run} landed {landed:,} " + f"{engine} counted {count:,} rows in {table_id}; {run} landed {landed:,} " f"{_since_phrase(info)}, and an append removes no row, so the table holds " "fewer rows than its runs landed" ) @@ -1267,6 +1295,11 @@ def _severity( "encryption:key-disabled": ( "Enable the key again (kms:EnableKey, e.g. aws kms enable-key --key-id )" ), + "columnRestrictions": ( + "Re-apply so each Lake Formation grant excludes the columns the contract restricts " + "from its principal, and revoke any SELECT granted outside the contract that reaches " + "them (including one to IAM_ALLOWED_PRINCIPALS on the table)" + ), "encryption:key-not-enabled": ( "Bring the key back to the Enabled state (see its KeyState in kms:DescribeKey and " "the KMS key states table); until then S3 can neither write nor read its objects" @@ -1277,7 +1310,7 @@ def _severity( def _storage_problems(dimensions: Mapping[str, Any]) -> List[Tuple[str, str]]: """``(message, action)`` for every failed storage dimension.""" problems: List[Tuple[str, str]] = [] - for name in ("retention", "encryption"): + for name in ("retention", "encryption", "columnRestrictions"): dimension = dimensions.get(name) or {} if dimension.get("status") != "fail": continue diff --git a/fluid_build/cli/_verify_bigquery.py b/fluid_build/cli/_verify_bigquery.py new file mode 100644 index 00000000..b341d124 --- /dev/null +++ b/fluid_build/cli/_verify_bigquery.py @@ -0,0 +1,213 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""``fluid verify``'s data dimensions for a BigQuery table: rows, and masking. + +The BigQuery verifier used to read only the table's metadata: its schema, and +the dataset's location. A table that held no rows, or that held a cleartext +``msisdn`` where the contract says it lands hashed, passed ``--strict``. These +are the two dimensions the Glue + Athena verifier already reports +(``_verify_athena``), by the same rules, from one GoogleSQL query: + +* **row_count.** ``SELECT COUNT(*)`` of the table, held to the run records of + the build that lands the expose (an acquisition build, or an embedded-SQL + build whose load records its run the same way), by the mode the run recorded + (equal for ``full_refresh``, at least the sum since the last full load for + ``incremental_append``, reported for anything else). A run counts as this + table's when its ``facets.bigquery_load`` names the table: the load the + duckdb runner performs records it, with the rows the load job (or, on an + emulator, the count after it) says arrived. A run that succeeded without a + BigQuery load landed somewhere else (a local or aws run from the same + contract directory) and is passed over; a failed one may have been this + table's, and ends the comparison. An empty table fails, except for a + reference-only contract with no run of its own. ``metadata.num_rows`` is + not the count: it leaves out the streaming buffer, and the goccy emulator + leaves it unset. +* **masking.** For each ``policy.privacy.masking`` rule, the non-null values + of the column that do not have the strategy's shape, counted in the same + query (``_verify_masking.bigquery_plan``); one is CRITICAL. Only counts come + back. + +A query that cannot run is an error, which ``fluid verify`` fails on: the +dimensions are unproven, not passed. The query needs ``bigquery.jobs.create`` +(``roles/bigquery.jobUser``) on the project as well as read access to the table. +""" + +from __future__ import annotations + +import logging +from pathlib import Path +from typing import Any, Dict, List, Mapping, Optional, Tuple + +LOG = logging.getLogger("fluid.cli.verify.bigquery") + +#: How long the count query may run before verify reports it as an error. +QUERY_TIMEOUT_SECONDS = 300.0 + + +def _loaded_into(record: Optional[Mapping[str, Any]], table_id: str) -> Optional[bool]: + """Whether a run loaded ``table_id``: yes, no, or unknown (``None``).""" + if record is None: + return None + facets = record.get("facets") + load = facets.get("bigquery_load") if isinstance(facets, Mapping) else None + if isinstance(load, Mapping) and load.get("table"): + return str(load["table"]).lower() == table_id.lower() + if str(record.get("state") or "").lower() == "succeeded": + # A BigQuery-bound run only succeeds after its load, which it records. + return False + return None + + +def _count_query( + client: Any, + bigquery: Any, + table_id: str, + selects: List[str], + params: List[Tuple[str, str]], + location: Optional[str], +) -> Tuple[int, List[int], Dict[str, Any]]: + """``(rows, the value of each extra select, query facts)``.""" + from fluid_build.build_runners._bigquery_load import quote_table_id + + selected = ", ".join(["COUNT(*)", *selects]) + sql = f"SELECT {selected} FROM {quote_table_id(table_id)}" + job_config = bigquery.QueryJobConfig( + query_parameters=[bigquery.ScalarQueryParameter(n, "STRING", v) for n, v in params] + ) + job = client.query(sql, job_config=job_config, location=location) + rows = list(job.result(timeout=QUERY_TIMEOUT_SECONDS)) + try: + cells = list(rows[0].values()) + # COUNT(*) always has a value; a count over no rows may come back NULL + # from an engine that sums (DuckDB's count_if does), which is zero. + values = [int(cells[0])] + [int(v or 0) for v in cells[1:]] + except (IndexError, TypeError, ValueError, AttributeError) as exc: + raise RuntimeError(f"the count query on {table_id} returned no readable count") from exc + if len(values) != 1 + len(selects): + raise RuntimeError(f"the count query on {table_id} returned {len(values)} value(s)") + return values[0], values[1:], {"sql": sql, "job_id": getattr(job, "job_id", None)} + + +def _escalate( + severity: Dict[str, Any], problems: List[Tuple[str, str]], row_count: Mapping[str, Any] +) -> Dict[str, Any]: + """The schema severity, raised to CRITICAL by a bad count or untreated values. + + As ``_verify_athena._severity``: an empty table of a reference-only + contract is INFO, reported and never gated. + """ + if not problems: + if row_count.get("status") == "info" and severity.get("level") == "SUCCESS": + return { + "level": "INFO", + "impact": "LOW", + "symbol": "🔵", + "remediation": "NONE", + "reason": row_count.get("message"), + "actions": ["Run the pipeline that owns the table, then verify again"], + } + return severity + already = severity.get("level") == "CRITICAL" + return { + "level": "CRITICAL", + "impact": "HIGH", + "symbol": "🔴", + "remediation": "MANUAL_INTERVENTION_REQUIRED", + "reason": "; ".join( + ([severity.get("reason", "")] if already else []) + [p[0] for p in problems] + ), + "actions": (list(severity.get("actions") or []) if already else []) + + [p[1] for p in problems], + } + + +def add_data_dimensions( + result: Dict[str, Any], + *, + client: Any, + bigquery: Any, + bq_table: Any, + expose: Mapping[str, Any], + contract: Mapping[str, Any], + workdir: Path, + reference_only: bool = False, + location: Optional[str] = None, +) -> Dict[str, Any]: + """``result`` (``verify_bigquery_table``'s) with ``row_count`` and ``masking`` added. + + Mutates and returns ``result``: its dimensions, severity, status and + ``metadata.num_rows`` (the counted rows; the table's own figure is kept as + ``metadata.table_num_rows``). A count that cannot be taken turns the result + into an error. + """ + from fluid_build.cli._verify_athena import _landed_rows, _row_count_dimension + from fluid_build.cli._verify_masking import bigquery_plan, severity_problem + + table_id = str(result["table_id"]) + expose_id = str(expose.get("exposeId") or expose.get("id") or "") + columns = [ + str(f.name) for f in getattr(bq_table, "schema", None) or [] if getattr(f, "name", None) + ] + selects, params, finish_masking = bigquery_plan(expose, columns) + try: + count, masking_values, query = _count_query( + client, bigquery, table_id, selects, params, location + ) + except Exception as exc: # noqa: BLE001 - every failure is reported, not raised + LOG.warning("verify_bigquery_count_failed table=%s error=%s", table_id, type(exc).__name__) + result["status"] = "error" + result["error"] = f"BigQuery could not count {table_id}: {exc}" + return result + try: + landed, info = _landed_rows( + contract, + expose_id, + workdir, + table_id, + wrote_into=_loaded_into, + embedded_sql_records=True, + ) + except Exception as exc: # noqa: BLE001 - an unreadable record is not a count + LOG.warning("verify_bigquery_run_record_unreadable error=%s", type(exc).__name__) + landed, info = None, {"source": "none", "note": "the run record could not be read"} + row_count = _row_count_dimension( + count, landed, info, table_id, reference_only=reference_only, engine="BigQuery" + ) + dimensions = result.setdefault("dimensions", {}) + dimensions["row_count"] = row_count + masking = finish_masking(masking_values) + if masking is not None: + dimensions["masking"] = masking + + problems: List[Tuple[str, str]] = [] + if row_count["status"] == "fail": + problems.append( + (row_count["message"], "Re-run the build and check its run record, then verify again") + ) + masking_problem = severity_problem(masking) + if masking_problem is not None: + problems.append(masking_problem) + result["severity"] = _escalate(dict(result.get("severity") or {}), problems, row_count) + if problems: + result["status"] = "mismatch" + metadata = result.setdefault("metadata", {}) + metadata["table_num_rows"] = metadata.get("num_rows") + metadata["num_rows"] = count + metadata["row_count_detail"] = row_count["message"] + result["bigquery"] = query + return result + + +__all__ = ["QUERY_TIMEOUT_SECONDS", "add_data_dimensions"] diff --git a/fluid_build/cli/_verify_bigquery_governance.py b/fluid_build/cli/_verify_bigquery_governance.py new file mode 100644 index 00000000..7dec343c --- /dev/null +++ b/fluid_build/cli/_verify_bigquery_governance.py @@ -0,0 +1,432 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""``fluid verify``: the BigQuery dataset and table apply the governance the contract declares. + +Three dimensions of the BigQuery verifier (``verify.verify_bigquery_table``), each +present only when the expose declares it, read from the live dataset and table: + +* ``retention`` — ``exposes[].lifecycle {retention, expire: true}``. The table must + be partitioned by day (on ``binding.location.partitionBy`` when it names a + column), its partitions must expire after exactly the period, and the table + itself must carry no expiration time, which would delete the product. +* ``encryption`` — ``binding.encryption.kms``. ``kmsKeyName`` of the table, and of + the dataset's default when this product owns the dataset, must be the declared + key (the product key's resource name, or the one written). +* ``columnRestrictions`` — ``exposes[].policy.authz.columnRestrictions``. Every + restricted column must carry a policy tag; the members holding + ``roles/datacatalog.categoryFineGrainedReader`` on it (Data Catalog + ``getIamPolicy``) must be exactly the readers the contract derives, a denied + principal among them fails and so does one granted outside the contract; and the + tag's taxonomy must enforce fine-grained access control. + +What is expected comes from ``iac/providers/gcp_governance.py``, the derivation +``fluid apply`` emits from, so verify checks what apply wrote. A mismatch is CRITICAL +(``fluid verify --strict`` fails); a check that could not run is an error, which +fails ``fluid verify`` with or without ``--strict``. The same shape as the S3 + +Glue storage checks (``_verify_storage_policy.py``). +""" + +from __future__ import annotations + +import logging +import os +from typing import Any, Callable, Dict, List, Mapping, Optional, Tuple + +LOG = logging.getLogger("fluid.cli.verify.bigquery_governance") + +#: The Data Catalog API that holds policy tags and their IAM policies. +DATACATALOG_API = "https://datacatalog.googleapis.com/v1" + +SessionFactory = Callable[[], Any] + +#: The variable python-bigquery reads for an emulator endpoint. No emulator +#: serves Data Catalog, so under it the policy tags' readers cannot be read. +_BIGQUERY_EMULATOR_HOST = "BIGQUERY_EMULATOR_HOST" + + +def _catalog_unreachable() -> Optional[str]: + """Why Data Catalog cannot be asked here, or ``None`` when it can.""" + if os.environ.get(_BIGQUERY_EMULATOR_HOST, "").strip(): + return ( + f"{_BIGQUERY_EMULATOR_HOST} points BigQuery at an emulator, and no emulator " + "serves Data Catalog, where the policy tags' readers are" + ) + return None + + +def default_session() -> Any: + """An authorized HTTP session with Application Default Credentials.""" + import google.auth + from google.auth.transport.requests import AuthorizedSession + + credentials, _ = google.auth.default(scopes=["https://www.googleapis.com/auth/cloud-platform"]) + return AuthorizedSession(credentials) # type: ignore[no-untyped-call] + + +def _key_name(value: Any) -> Optional[str]: + """A ``kmsKeyName`` without the ``/cryptoKeyVersions/`` suffix, or ``None``.""" + text = str(value or "").strip() + if not text: + return None + return text.split("/cryptoKeyVersions/", 1)[0] + + +def _retention_dimension(bq_table: Any, expected: Any) -> Dict[str, Any]: + partitioning = getattr(bq_table, "time_partitioning", None) + expires = getattr(bq_table, "expires", None) + problems: List[str] = [] + actual_ms = getattr(partitioning, "expiration_ms", None) if partitioning else None + if partitioning is None: + problems.append("the table is not partitioned, so no partition ever expires") + else: + kind = str(getattr(partitioning, "type_", "") or "") + if kind.upper() != "DAY": + problems.append(f"the table is partitioned by {kind or 'nothing'}, not by DAY") + field = getattr(partitioning, "field", None) or None + if field != expected.field: + problems.append( + f"the table is partitioned on {field or 'ingestion time'}, not on " + f"{expected.field or 'ingestion time'}" + ) + if actual_ms != expected.expiration_ms: + shown = "never" if not actual_ms else f"after {int(actual_ms) / 86_400_000:g} days" + problems.append( + f"its partitions expire {shown}, but lifecycle.retention is {expected.period} " + f"({expected.days} days)" + ) + if expires is not None: + when = expires.isoformat() if hasattr(expires, "isoformat") else str(expires) + problems.append( + f"the table itself expires at {when}, which deletes the whole product, not only " + "the partitions older than the retention" + ) + return { + "status": "fail" if problems else "pass", + "expected_expiration_ms": expected.expiration_ms, + "actual_expiration_ms": actual_ms, + "message": ( + "Retention: " + "; ".join(problems) + if problems + else f"Retention: daily partitions expire after {expected.days} days" + ), + } + + +def _encryption_dimension( + bq_dataset: Any, bq_table: Any, expected: str, *, dataset_owned: bool +) -> Dict[str, Any]: + table_config = getattr(bq_table, "encryption_configuration", None) + table_key = _key_name(getattr(table_config, "kms_key_name", None)) + problems: List[str] = [] + if table_key != expected: + problems.append(f"the table is encrypted with {table_key or 'a Google-managed key'}") + dataset_key = None + if dataset_owned: + dataset_config = getattr(bq_dataset, "default_encryption_configuration", None) + dataset_key = _key_name(getattr(dataset_config, "kms_key_name", None)) + if dataset_key != expected: + problems.append(f"the dataset's default key is {dataset_key or 'a Google-managed key'}") + return { + "status": "fail" if problems else "pass", + "expected": expected, + "table": table_key, + "dataset": dataset_key, + "message": ( + f"Encryption: {'; '.join(problems)}, not {expected}" + if problems + else f"Encryption: {expected}" + ), + } + + +class _CatalogError(Exception): + pass + + +def _catalog_json(session: Any, method: str, url: str) -> Mapping[str, Any]: + try: + response = ( + session.post(url, json={}) if method == "POST" else session.get(url) + ) # noqa: S113 — AuthorizedSession applies its own timeout + except Exception as exc: # noqa: BLE001 — reported as the dimension's error + raise _CatalogError(f"{method} {url} failed: {type(exc).__name__}") from None + status = int(getattr(response, "status_code", 0) or 0) + if status != 200: + raise _CatalogError(f"{method} {url} returned HTTP {status}") + body = response.json() + return body if isinstance(body, Mapping) else {} + + +def _column_access_dimension( + bq_table: Any, groups: List[Any], session_factory: SessionFactory +) -> Dict[str, Any]: + """Every restricted column's tag, then (through Data Catalog) each tag's readers. + + The tags are on the table, so a column that carries none is reported from + the table alone. The Data Catalog session is opened only for a tag whose + readers must be read: it used to be opened first, so a table with no tags + at all, checked where no credentials were (the BigQuery emulator), was an + error about credentials instead of the failure it is. + """ + fields = {getattr(f, "name", None): f for f in (getattr(bq_table, "schema", None) or [])} + problems: List[str] = [] + tagged: List[Tuple[Any, Dict[str, List[str]]]] = [] + for group in groups: + by_tag: Dict[str, List[str]] = {} + for column in group.columns: + field = fields.get(column) + tags = getattr(field, "policy_tags", None) if field is not None else None + names = list(getattr(tags, "names", None) or ()) + if len(names) != 1: + problems.append( + f"column {column} carries no policy tag, so every reader of the table " + "can read it" + ) + continue + by_tag.setdefault(str(names[0]), []).append(column) + tagged.append((group, by_tag)) + if not any(by_tag for _, by_tag in tagged): + return _column_access_result(problems, []) + + unreachable = _catalog_unreachable() + session: Any = None + if unreachable is None: + try: + session = session_factory() + except Exception as exc: # noqa: BLE001 — reported, with what was found + unreachable = f"no Data Catalog session: {type(exc).__name__}: {exc}" + if unreachable is not None: + note = f"the readers of the tagged columns were not checked: {unreachable}" + if problems: + # What the table shows is conclusive; say what could not be read. + return _column_access_result(problems + [note], []) + if _catalog_unreachable() is not None: + return { + "status": "unsupported", + "tags": [], + "message": ( + "Column restrictions: every restricted column carries a policy tag; " + note + ), + } + raise _CatalogError(unreachable) + try: + return _column_access_readers(tagged, problems, session) + except _CatalogError as exc: + if not problems: + raise + return _column_access_result( + problems + [f"the readers of the tagged columns were not all checked: {exc}"], [] + ) + + +def _column_access_result(problems: List[str], checked: List[Dict[str, Any]]) -> Dict[str, Any]: + return { + "status": "fail" if problems else "pass", + "tags": checked, + "message": ( + "Column restrictions: " + "; ".join(problems) + if problems + else "Column restrictions: every restricted column is readable only by its readers" + ), + } + + +def _column_access_readers( + tagged: List[Tuple[Any, Dict[str, List[str]]]], problems: List[str], session: Any +) -> Dict[str, Any]: + """Each tag's fine-grained readers against the group's, and its taxonomy's enforcement.""" + from fluid_build.iac.providers.gcp_governance import FINE_GRAINED_READER_ROLE + + checked: List[Dict[str, Any]] = [] + taxonomies: Dict[str, bool] = {} + for group, by_tag in tagged: + for tag, columns in by_tag.items(): + policy = _catalog_json(session, "POST", f"{DATACATALOG_API}/{tag}:getIamPolicy") + members: set[str] = set() + for binding in policy.get("bindings") or (): + if isinstance(binding, Mapping) and binding.get("role") == FINE_GRAINED_READER_ROLE: + members.update(str(m) for m in binding.get("members") or ()) + expected = set(group.readers) + extra = sorted(members - expected) + missing = sorted(expected - members) + if extra: + problems.append( + f"{', '.join(extra)} can read {', '.join(columns)} (fine-grained reader on " + "its policy tag), which the contract's column restrictions do not allow" + ) + if missing: + problems.append( + f"{', '.join(missing)} cannot read {', '.join(columns)}, which the " + "contract allows" + ) + taxonomy = tag.split("/policyTags/", 1)[0] + if taxonomy not in taxonomies: + body = _catalog_json(session, "GET", f"{DATACATALOG_API}/{taxonomy}") + taxonomies[taxonomy] = "FINE_GRAINED_ACCESS_CONTROL" in ( + body.get("activatedPolicyTypes") or () + ) + if not taxonomies[taxonomy]: + problems.append( + f"the taxonomy {taxonomy} does not enforce fine-grained access " + "control, so its policy tags restrict no one" + ) + checked.append({"tag": tag, "columns": columns, "readers": sorted(members)}) + return _column_access_result(problems, checked) + + +def governance_dimensions( + expose: Mapping[str, Any], + *, + contract: Mapping[str, Any], + bq_dataset: Any, + bq_table: Any, + project: str, + index: int = 0, + session_factory: Optional[SessionFactory] = None, +) -> Dict[str, Dict[str, Any]]: + """The ``retention``, ``encryption`` and ``columnRestrictions`` dimensions ``expose`` declares. + + Never raises for a GCP failure or a declaration apply would refuse: the reason + is the dimension (``status: error``). + """ + from fluid_build.cli._verify_athena import _resolved_binding + from fluid_build.iac.base import UnsupportedBindingError + from fluid_build.iac.naming import safe_ident + from fluid_build.iac.providers import gcp_governance as gov + from fluid_build.iac.providers.gcp import _bq_table_name + + binding = _resolved_binding(expose.get("binding")) + resolved = {**expose, "binding": binding} + loc = binding.get("location") or {} + dimensions: Dict[str, Dict[str, Any]] = {} + cid = safe_ident(contract.get("id") or contract.get("name") or "product") + dataset = str(loc.get("dataset") or "default") + + def attempt(name: str, check: Callable[[], Optional[Dict[str, Any]]]) -> None: + try: + dimension = check() + except UnsupportedBindingError as exc: + dimension = {"status": "error", "message": f"{name}: {exc}"} + except _CatalogError as exc: + dimension = {"status": "error", "message": f"{name}: {exc}"} + except Exception as exc: # noqa: BLE001 — every GCP failure is reported, not raised + dimension = {"status": "error", "message": f"{name}: {type(exc).__name__}: {exc}"} + if dimension is not None: + dimensions[name] = dimension + + def retention() -> Optional[Dict[str, Any]]: + expected = gov.retention_for(resolved, index) + return None if expected is None else _retention_dimension(bq_table, expected) + + def encryption() -> Optional[Dict[str, Any]]: + expected = gov.encryption_for(binding, gov.dataset_location(loc)) + if expected is None: + return None + key = ( + gov.product_key_name( + str(loc.get("project") or project), + expected.location, + gov.product_key_ring(cid, dataset), + ) + if expected.product_key + else expected.kms + ) + from fluid_build.iac.packaging import resolve_packaging + from fluid_build.iac.providers.gcp import _placement + + # A shared (pool) dataset's default key is its owner's; only the table is ours. + owned = not _placement(resolve_packaging(contract), resolved).dataset_referenced + return _encryption_dimension(bq_dataset, bq_table, key, dataset_owned=owned) + + def columns() -> Optional[Dict[str, Any]]: + groups = gov.tag_groups( + contract, resolved, cid, dataset, _bq_table_name(resolved, loc), index + ) + if not groups: + return None + return _column_access_dimension(bq_table, groups, session_factory or default_session) + + attempt("retention", retention) + attempt("encryption", encryption) + attempt("columnRestrictions", columns) + for name, dimension in dimensions.items(): + LOG.info("verify_bigquery_governance dimension=%s status=%s", name, dimension["status"]) + return dimensions + + +#: What to do about a failed dimension. +_ACTIONS = { + "retention": ( + "Re-apply so the table's partition expiration matches lifecycle.retention (adding " + "partitioning replaces the table: --allow-data-loss), and remove any table " + "expiration set outside the contract" + ), + "encryption": ( + "Re-apply so the dataset's default key and the table's key are the declared key " + "(re-keying a table replaces it: --allow-data-loss)" + ), + "columnRestrictions": ( + "Re-apply so each restricted column carries its policy tag with exactly the " + "contract's fine-grained readers, and revoke roles/datacatalog.categoryFineGrainedReader " + "from any principal granted it outside the contract" + ), +} + + +def governance_problems(dimensions: Mapping[str, Any]) -> List[Tuple[str, str]]: + """``(message, action)`` for every failed governance dimension.""" + return [ + (str(dimension["message"]), _ACTIONS[name]) + for name, dimension in dimensions.items() + if name in _ACTIONS and dimension.get("status") == "fail" + ] + + +def governance_errors(dimensions: Mapping[str, Any]) -> List[str]: + """The message of every governance dimension that could not be checked.""" + return [ + str(dimension["message"]) + for name, dimension in dimensions.items() + if name in _ACTIONS and dimension.get("status") == "error" + ] + + +def with_governance_severity( + severity: Dict[str, Any], dimensions: Mapping[str, Any] +) -> Dict[str, Any]: + """``severity`` raised to CRITICAL with each failed governance dimension's reason.""" + problems = governance_problems(dimensions) + if not problems: + return severity + already = severity.get("level") == "CRITICAL" + return { + "level": "CRITICAL", + "impact": "HIGH", + "symbol": "🔴", + "remediation": "MANUAL_INTERVENTION_REQUIRED", + "reason": "; ".join(([severity["reason"]] if already else []) + [p[0] for p in problems]), + "actions": (list(severity.get("actions") or []) if already else []) + + [p[1] for p in problems], + } + + +__all__ = [ + "DATACATALOG_API", + "default_session", + "governance_dimensions", + "governance_errors", + "governance_problems", + "with_governance_severity", +] diff --git a/fluid_build/cli/_verify_lf_columns.py b/fluid_build/cli/_verify_lf_columns.py new file mode 100644 index 00000000..3f32cadf --- /dev/null +++ b/fluid_build/cli/_verify_lf_columns.py @@ -0,0 +1,199 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""``fluid verify``: Lake Formation lets no principal read a column the contract restricts from it. + +The ``columnRestrictions`` dimension of the S3 + Glue verifier, present only when +the expose declares ``policy.authz.columnRestrictions``. What is expected comes +from ``iac/column_access.py`` (``lf_expected_exclusions``), the derivation the +emitter writes each grant's excluded columns from: for every read grantee, every +principal a ``deny`` names, and ``IAM_ALLOWED_PRINCIPALS``, the columns it must not +read. ``lakeformation:ListPermissions`` on tables then fails the check when any of +that principal's ``SELECT`` (or ``ALL``) permissions on this table reaches one of +them: a table-level grant (every column), a column list naming it, or a column +wildcard that does not exclude it. So a grant made outside the contract is caught +too. A mismatch is CRITICAL; a check that could not run is an error. +""" + +from __future__ import annotations + +import logging +from typing import Any, Callable, Dict, List, Mapping, Optional, Set + +LOG = logging.getLogger("fluid.cli.verify.lf_columns") + +ClientFactory = Callable[[str, str], Any] + + +def _readable(resource: Mapping[str, Any], database: str, table: str) -> Optional[Set[str]]: + """Columns of ``database.table`` this permission's resource reaches; ``None`` for none. + + The empty set with the ``"*"`` member means every column. + """ + whole = {"*"} + if "Table" in resource: + ref = resource.get("Table") or {} + if ref.get("DatabaseName") != database: + return None + if ref.get("Name") == table or "TableWildcard" in ref: + return whole + return None + if "TableWithColumns" in resource: + ref = resource.get("TableWithColumns") or {} + if ref.get("DatabaseName") != database or ref.get("Name") != table: + return None + if ref.get("ColumnNames"): + return set(ref["ColumnNames"]) + wildcard = ref.get("ColumnWildcard") + if wildcard is not None: + return {"*", *(f"-{c}" for c in (wildcard.get("ExcludedColumnNames") or ()))} + return whole + return None + + +def _leaks(readable: Set[str], forbidden: List[str]) -> List[str]: + if "*" in readable: + return [c for c in forbidden if f"-{c}" not in readable] + return [c for c in forbidden if c in readable] + + +def _table_permissions(lf: Any) -> List[Mapping[str, Any]]: + """Every table permission the caller can see (``ResourceType: TABLE``, paginated). + + One listing, filtered here by principal and table. ListPermissions takes a + ``Principal`` only together with a ``Resource`` ("Resource is mandatory if + Principal is set"), and a ``Table`` resource does not return the column-level + grants, which are the ones this check is about; ``IAM_ALLOWED_PRINCIPALS`` is + not an ARN either. The caller must be allowed to list the table's + permissions (a Lake Formation administrator, or a grantable holder). + """ + entries: List[Mapping[str, Any]] = [] + kwargs: Dict[str, Any] = {"ResourceType": "TABLE"} + while True: + page = lf.list_permissions(**kwargs) + entries.extend(page.get("PrincipalResourcePermissions") or []) + token = page.get("NextToken") + if not token: + return entries + kwargs["NextToken"] = token + + +def column_restrictions_dimension( + expose_id: str, + expose: Mapping[str, Any], + binding: Mapping[str, Any], + *, + region: str, + factory: ClientFactory, +) -> Optional[Dict[str, Any]]: + """The ``columnRestrictions`` dimension, or ``None`` when the expose declares none.""" + from fluid_build.iac.base import UnsupportedBindingError + from fluid_build.iac.column_access import LF_READ_PERMISSIONS, lf_expected_exclusions + + exposure = { + "exposeId": expose_id, + "policy": expose.get("policy"), + "contract": expose.get("contract"), + } + try: + expected = lf_expected_exclusions(exposure, binding) + except UnsupportedBindingError as exc: + return {"status": "error", "message": f"Column restrictions: {exc}"} + if not expected: + return None + loc = binding.get("location") or {} + database, table = str(loc.get("database") or ""), str(loc.get("table") or "") + try: + lf = factory("lakeformation", region) + except Exception as exc: # noqa: BLE001 — e.g. boto3 missing + return { + "status": "error", + "message": f"Column restrictions: could not create the Lake Formation client: {exc}", + } + try: + permissions = _table_permissions(lf) + except Exception as exc: # noqa: BLE001 — reported, not raised + from fluid_build.cli._verify_athena import _error_code + + return { + "status": "error", + "message": ( + "Column restrictions: lakeformation:ListPermissions failed " + f"({_error_code(exc) or type(exc).__name__})" + ), + } + # ListPermissions shows only what the caller may see, so a caller that is not a + # Lake Formation administrator gets a partial listing, and an absent grant + # would read as a pass. The grants the contract makes must be visible, or the + # listing cannot be trusted. + grantees = { + str(g.get("principal")) + for g in ((binding.get("governance") or {}).get("lakeFormation") or {}).get("grants") or () + if isinstance(g, Mapping) and g.get("principal") + } + visible = { + (entry.get("Principal") or {}).get("DataLakePrincipalIdentifier") + for entry in permissions + if _readable(entry.get("Resource") or {}, database, table) is not None + } + unseen = sorted(grantees - visible) + if unseen: + return { + "status": "error", + "message": ( + f"Column restrictions: lakeformation:ListPermissions shows no permission on " + f"{database}.{table} for {', '.join(unseen)}, which the contract grants; " + "either the apply has not run or the verifying identity cannot see every " + "permission (it must be a Lake Formation administrator), so the check " + "cannot be trusted" + ), + } + problems: List[str] = [] + checked: Dict[str, List[str]] = {} + for principal, forbidden in expected.items(): + leaked: List[str] = [] + for entry in permissions: + holder = (entry.get("Principal") or {}).get("DataLakePrincipalIdentifier") + if holder != principal: + continue + if not LF_READ_PERMISSIONS & set(entry.get("Permissions") or ()): + continue + readable = _readable(entry.get("Resource") or {}, database, table) + if readable is None: + continue + leaked.extend(c for c in _leaks(readable, list(forbidden)) if c not in leaked) + checked[principal] = list(forbidden) + if leaked: + problems.append( + f"{principal} can read {', '.join(leaked)} of {database}.{table} through " + "Lake Formation, which the contract's column restrictions do not allow" + ) + LOG.info( + "verify_lf_columns expose=%s principals=%d problems=%d", + expose_id, + len(checked), + len(problems), + ) + return { + "status": "fail" if problems else "pass", + "checked": checked, + "message": ( + "Column restrictions: " + "; ".join(problems) + if problems + else "Column restrictions: no principal can read a column restricted from it" + ), + } + + +__all__ = ["column_restrictions_dimension"] diff --git a/fluid_build/cli/_verify_masking.py b/fluid_build/cli/_verify_masking.py index 14e03891..61985753 100644 --- a/fluid_build/cli/_verify_masking.py +++ b/fluid_build/cli/_verify_masking.py @@ -33,7 +33,10 @@ Local files are checked with DuckDB's ``regexp_full_match`` (RE2); S3+Glue tables with Athena's ``regexp_like`` anchored with ``\\A(?:...)\\z``, in the -same query that counts the rows (``_verify_athena._count_rows``). Trino's +same query that counts the rows (``_verify_athena._count_rows``); BigQuery +tables with GoogleSQL's ``REGEXP_CONTAINS`` (RE2, which also finds rather than +matches) under the same anchors, each pattern a query parameter, in the query +that counts the table (``_verify_bigquery``). Trino's ``regexp_like`` finds rather than matches ("the pattern only needs to be contained within string"), and in the Java pattern syntax it documents ``$`` also matches before a final line terminator, so ``^...$`` would pass @@ -154,11 +157,15 @@ def _plan( untreated: Callable[[str, Any], str], ok: str, bad: str, + quote: Callable[[str], str] = _quote, ) -> Tuple[List[str], Finish]: """``(select expressions, finish)``: two per queryable rule, the column's non-null count and ``untreated(quoted column, rule)``, the count of its values without the rule's shape. Rules that cannot be queried get their - entry without a query.""" + entry without a query. ``quote`` is the dialect's identifier quoting: + double quotes for DuckDB and Athena, backticks for GoogleSQL, where a + double-quoted name is a string literal and ``count("msisdn")`` would + count every row.""" rules, problem = declared_rules(expose) present = {c.lower(): c for c in columns} entries: List[Optional[Dict[str, Any]]] = [] @@ -169,7 +176,7 @@ def _plan( entry = _unchecked(rule, actual is not None, where) entries.append(entry) if entry is None and actual is not None: - ident = _quote(actual) + ident = quote(actual) selects += [f"count({ident})", untreated(ident, rule)] queried.append((len(entries) - 1, rule)) @@ -248,6 +255,43 @@ def athena_plan( ) +# ── BigQuery ──────────────────────────────────────────────────────────── + + +def bigquery_plan( + expose: Mapping[str, Any], table_columns: Sequence[str] +) -> Tuple[List[str], List[Tuple[str, str]], Finish]: + """``(select expressions, (parameter, pattern) pairs, finish)`` for the count query. + + As :func:`athena_plan`, in GoogleSQL: backtick-quoted columns, and + ``REGEXP_CONTAINS`` against a named STRING parameter holding the anchored + shape, so no pattern is spliced into the SQL. Column names come from the + live table's schema. + """ + from fluid_build.build_runners._bigquery_load import quote_bq_ident + + params: List[Tuple[str, str]] = [] + + def untreated(ident: str, rule: Any) -> str: + name = f"fluid_mask_shape_{len(params)}" + params.append((name, athena_regex(rule.shape))) + return ( + f"COUNTIF({ident} IS NOT NULL AND NOT " + f"REGEXP_CONTAINS(CAST({ident} AS STRING), @{name}))" + ) + + selects, finish = _plan( + expose, + table_columns, + where="the BigQuery table", + untreated=untreated, + ok=ATHENA_OK, + bad=ATHENA_BAD, + quote=quote_bq_ident, + ) + return selects, params, finish + + def severity_problem(masking: Optional[Mapping[str, Any]]) -> Optional[Tuple[str, str]]: """``(reason, action)`` when the dimension failed, for the CRITICAL grading.""" if not masking or masking.get("status") in (LOCAL_OK, ATHENA_OK): diff --git a/fluid_build/cli/_verify_storage_policy.py b/fluid_build/cli/_verify_storage_policy.py index dfc98c12..fb5268ea 100644 --- a/fluid_build/cli/_verify_storage_policy.py +++ b/fluid_build/cli/_verify_storage_policy.py @@ -401,6 +401,14 @@ def storage_policy( result = StoragePolicy() loc = binding.get("location") or {} bucket = str(loc.get("bucket") or "") + # policy.authz.columnRestrictions, checked against Lake Formation's permissions. + from fluid_build.cli._verify_lf_columns import column_restrictions_dimension + + columns = column_restrictions_dimension( + expose_id, expose, binding, region=region, factory=factory + ) + if columns is not None: + result.dimensions["columnRestrictions"] = columns try: retention = aws_storage.retention_for( {"exposeId": expose_id, "lifecycle": expose.get("lifecycle"), "binding": binding} diff --git a/fluid_build/cli/apply.py b/fluid_build/cli/apply.py index 2924c0ff..197f6539 100644 --- a/fluid_build/cli/apply.py +++ b/fluid_build/cli/apply.py @@ -46,6 +46,8 @@ from fluid_build.cli.console import cprint, success, warning from fluid_build.observability.tracing import traced_stage as _traced_stage +from ._apply_cc_report import reports_apply_run + # Rich imports for enhanced output try: from rich.console import Console @@ -1553,6 +1555,7 @@ def run_in_thread(): @_traced_stage("apply") +@reports_apply_run def run(args, logger: logging.Logger) -> int: """ Main execution function for the apply command diff --git a/fluid_build/cli/bundle.py b/fluid_build/cli/bundle.py index 6fd0e0d6..14138de8 100644 --- a/fluid_build/cli/bundle.py +++ b/fluid_build/cli/bundle.py @@ -228,7 +228,7 @@ def materialize_contract( compiled = _deep_merge(dict(compiled), overlay) logger.info("overlay_applied", extra={"overlay": str(overlay_path)}) else: - note_missing_overlay(contract_path, env, logger) + note_missing_overlay(contract_path, env, logger, contract=compiled) return compiled, overlay_path diff --git a/fluid_build/cli/plan.py b/fluid_build/cli/plan.py index 2295fc4b..56d5cf5a 100644 --- a/fluid_build/cli/plan.py +++ b/fluid_build/cli/plan.py @@ -602,11 +602,16 @@ def _report_sovereignty( return False # No usable provider verdict — fall back to the built-in policy engine. - reason = ( - f"the {pname} provider has no sovereignty hook" - if hook_provider is not None - else "no provider could be built for this contract" - ) + if hook_provider is None: + reason = "no provider could be built for this contract" + else: + from fluid_build.cli.hooks import has_hook + + reason = ( + f"the {pname} provider's sovereignty hook gave no verdict" + if has_hook(hook_provider, "validate_sovereignty") + else f"the {pname} provider has no sovereignty hook" + ) sovereignty = contract.get("sovereignty") or {} if not isinstance(sovereignty, dict) or not sovereignty: diff --git a/fluid_build/cli/schedule_sync.py b/fluid_build/cli/schedule_sync.py index 7d1f8283..b318e37e 100644 --- a/fluid_build/cli/schedule_sync.py +++ b/fluid_build/cli/schedule_sync.py @@ -57,6 +57,13 @@ deletes nothing. This applies to the transports that delete (file, ssh, git+ssh, s3 and gs for airflow; s3 for mwaa); az, scp, composer, astronomer, prefect and dagster never delete and ignore it. +* An env's DAGs are ``__/`` (dag ids + ``____``); forge-cli 0.16.7 and earlier wrote them to + ``/`` as ``__``. Under ``--delete-scope product`` + the first sync after the upgrade retires those old DAGs, the ones rendered + for the same product and env, where the destination can be read here (a + local path, a git+ssh clone). Every other transport prints the one step + left to do, and the report lists each case (``superseded_scopes``). CLI surface:: @@ -76,6 +83,7 @@ from __future__ import annotations import argparse +import ast import json import logging import os @@ -564,6 +572,195 @@ def _under(root: str, suffix: str) -> str: return root if not suffix else root.rstrip("/") + "/" + suffix +# ----------------------------------------------------------------------------- +# Retiring the DAGs an env's scope replaced +# ----------------------------------------------------------------------------- +# +# forge-cli 0.16.7 and earlier wrote a product's DAGs to ``/`` with +# dag_id ``__``, whatever the ``--env``. Now an env's DAGs are +# ``__/`` with dag_id ``____`` +# (``fluid_apply.schedule_scope_for`` / ``dag_id_for``). ``--delete-scope +# product`` mirrors only the new directory, so the first sync after the +# upgrade left the old DAG in place beside the new one: two DAGs applying the +# same product at the same minute (measured on the demo lab, whose Airflow +# unpauses a DAG as it parses it). The old ones are retired where this command +# can read the destination (a local path, and the clone of a git+ssh one), and +# named in a note everywhere else. + +#: The module-level names a rendered DAG file assigns (``fluid_apply.render_dag``). +_DAG_FACT_NAMES = ("PRODUCT_ID", "BUILD_ID", "FLUID_ENV_NAME") +#: A DAG file name the retirement acts on: a plain module name. +_DAG_FILE_RE = re.compile(r"^[A-Za-z0-9_][A-Za-z0-9_.\-]{0,127}\.py$") + + +def _dag_facts(path: Path) -> Optional[Dict[str, str]]: + """``PRODUCT_ID``, ``BUILD_ID``, ``FLUID_ENV_NAME`` and ``dag_id`` of a rendered DAG. + + Read from the file's syntax tree, never by importing it. ``None`` for any + file that is not one ``fluid generate`` renders. + """ + try: + if path.is_symlink() or not path.is_file() or path.stat().st_size > 1_000_000: + return None + tree = ast.parse(path.read_text(encoding="utf-8")) + except (OSError, SyntaxError, ValueError): + return None + facts: Dict[str, str] = {} + for node in tree.body: + if ( + isinstance(node, ast.Assign) + and len(node.targets) == 1 + and isinstance(node.targets[0], ast.Name) + and node.targets[0].id in _DAG_FACT_NAMES + and isinstance(node.value, ast.Constant) + and isinstance(node.value.value, str) + ): + facts[node.targets[0].id] = node.value.value + elif isinstance(node, ast.With): + for item in node.items: + call = item.context_expr + if not (isinstance(call, ast.Call) and getattr(call.func, "id", None) == "DAG"): + continue + for keyword in call.keywords: + value = keyword.value + if ( + keyword.arg == "dag_id" + and isinstance(value, ast.Constant) + and isinstance(value.value, str) + ): + facts["dag_id"] = value.value + return facts if set(_DAG_FACT_NAMES) | {"dag_id"} <= set(facts) else None + + +def _replaced_scopes(dags_dir: Path) -> List[Tuple[str, str, str]]: + """``(env scope, the scope it replaced, env)`` for each ``__/`` directory. + + A directory is an env's scope when every DAG in it was rendered for that + product and env, under the env's directory name and dag id. + """ + out: List[Tuple[str, str, str]] = [] + for scope in sorted(p for p in dags_dir.iterdir() if p.is_dir() and not p.is_symlink()): + dags = [_dag_facts(f) for f in sorted(scope.glob("*.py"))] + if not dags or any(d is None for d in dags): + continue + products = {d["PRODUCT_ID"] for d in dags if d} + envs = {d["FLUID_ENV_NAME"] for d in dags if d} + if len(products) != 1 or len(envs) != 1: + continue + (product,), (env,) = products, envs + if not env or scope.name != f"{product}__{env}": + continue + if all(d and d["dag_id"] == f"{product}__{env}__{d['BUILD_ID']}" for d in dags): + out.append((scope.name, product, env)) + return out + + +def _superseded_dags(directory: Path, product: str, env: str) -> List[str]: + """The files in ``directory`` that are ``product``'s DAGs for ``env`` under the old id. + + Only a DAG rendered for this product and this env, whose dag id carries no + env, qualifies: an env-less DAG (``FLUID_ENV_NAME = ''``), another env's + and any file that is not a rendered DAG stay. + """ + if directory.is_symlink() or not directory.is_dir(): + return [] + out: List[str] = [] + for path in sorted(directory.iterdir()): + if not _DAG_FILE_RE.fullmatch(path.name): + continue + facts = _dag_facts(path) + if ( + facts is not None + and facts["PRODUCT_ID"] == product + and facts["FLUID_ENV_NAME"] == env + and facts["dag_id"] == f"{product}__{facts['BUILD_ID']}" + ): + out.append(path.name) + return out + + +#: The transports whose destination this command can read, and so retire in. +_RETIRING_SCHEMES = ("file", "git+ssh") +#: The transports that mirror with deletion (``--delete-scope destination`` +#: deletes the old directory itself there). +_DELETING_SCHEMES = ("s3", "gs", "file", "ssh", "git+ssh") + + +def _report_superseded_scopes(dags_dir: Path, args: argparse.Namespace) -> List[Dict[str, Any]]: + """Each env directory that replaced an old one, and whether the old DAGs were retired. + + Where this command cannot read the destination, the one step that is left + to do is printed: without it the old DAG keeps running beside the new one. + """ + try: + replaced = _replaced_scopes(dags_dir) + except OSError: + return [] + if not replaced: + return [] + scope = _delete_scope(args) + scheme = None + if args.scheduler in ("airflow", "mwaa") and args.destination: + try: + scheme, _dest = _validate_destination(args.destination, args.scheduler) + except CLIError: + scheme = None + if args.scheduler == "mwaa": + scheme = "s3" + retired_here = ( + args.scheduler == "airflow" and scheme in _RETIRING_SCHEMES and scope == "product" + ) + mirrored = scope == "destination" and scheme in _DELETING_SCHEMES + out: List[Dict[str, Any]] = [] + for current, product, env in replaced: + handled = retired_here or mirrored + out.append({"scope": current, "replaces": product, "env": env, "old_dags_retired": handled}) + if handled: + continue + cprint( + f"[schedule-sync] note: {current}/ now holds {product}'s DAGs for env {env}. " + f"forge-cli 0.16.7 and earlier synced them to {product}/ with dag ids " + f"{product}__, and this destination cannot be read here to retire " + f"those. Delete {product}/'s DAG files for env {env} at the destination once " + f"(each is a DAG whose FLUID_ENV_NAME is {env!r}), or Airflow runs " + f"{product}__ beside {product}__{env}__.", + markup=False, + ) + return out + + +def _retire_argvs( + rsync: str, root: Path, dest_root: str, replaced: List[Tuple[str, str, str]], empty: str +) -> List[List[str]]: + """An ``rsync --delete`` from an empty directory, filtered to the superseded DAGs. + + ``root`` is where the destination can be read (a local path, or the + git+ssh clone); ``dest_root`` is how rsync names it. Only the files + :func:`_superseded_dags` names are deleted: everything else in the old + directory is excluded, and rsync never deletes an excluded file. ``-r``, + not ``-a``: the old directory keeps its own mode and times, not the + empty source's (a private temporary directory). + """ + argvs: List[List[str]] = [] + for _scope, product, env in replaced: + names = _superseded_dags(root / product, product, env) + if not names: + continue + argvs.append( + [ + rsync, + "-rv", + "--delete", + *[f"--include=/{name}" for name in names], + "--exclude=*", + "--", + empty.rstrip("/") + "/", + _under(dest_root, f"{product}/"), + ] + ) + return argvs + + def _run_units(argvs: List[List[str]], args: argparse.Namespace) -> List[Dict[str, Any]]: """Run each argv in turn; stop at the first failure.""" results: List[Dict[str, Any]] = [] @@ -873,8 +1070,16 @@ def _planned(clone_dir: str) -> List[Dict]: if clone_result["exit_code"] != 0: return results + empty = str(Path(tmp) / "empty") + Path(empty).mkdir() + retire = ( + _retire_argvs(rsync, Path(clone_dir), "./", _replaced_scopes(dags_dir), empty) + if _delete_scope(args) == "product" + else [] + ) for argv in ( *rsync_argvs, + *retire, [git, "add", "--", "."], ): result = _run_subprocess_with_cwd( @@ -1025,20 +1230,29 @@ def _planned(clone_dir: str) -> List[Dict]: # future change to _validate_destination that lets a leading-'-' # path slip through still doesn't smuggle an rsync option. root = local_dest.rstrip("/") + "/" - return _run_units( - [ + with tempfile.TemporaryDirectory(prefix="fluid-schedule-sync-empty-") as empty: + # The DAGs an env's directory replaced, once each env's own + # directory is in place (see _replaced_scopes). + retire = ( + _retire_argvs(binary, Path(root), root, _replaced_scopes(dags_dir), empty) + if _delete_scope(args) == "product" + else [] + ) + return _run_units( [ - binary, - "-av", - *(["--delete"] if delete else []), - "--", - source, - _under(root, suffix), + [ + binary, + "-av", + *(["--delete"] if delete else []), + "--", + source, + _under(root, suffix), + ] + for source, suffix, delete in units ] - for source, suffix, delete in units - ], - args, - ) + + retire, + args, + ) elif scheme == "ssh": binary = _which_or_raise("rsync") # rsync over ssh: ssh://user@host/path → user@host:/path @@ -1423,6 +1637,7 @@ def run(args, _logger: Optional[logging.Logger] = None) -> int: ) results = dispatcher(dags_dir, args) + superseded = _report_superseded_scopes(dags_dir, args) # ── Acquisition pattern: emit per-orchestrator artifacts ──────────── # When the contract carries a Bronze ``pattern: acquisition`` build, @@ -1486,6 +1701,7 @@ def run(args, _logger: Optional[logging.Logger] = None) -> int: "delete_scope": delete_scope, "dry_run": args.dry_run, "results": results, + "superseded_scopes": superseded, "overall_exit": overall_exit, } diff --git a/fluid_build/cli/validate.py b/fluid_build/cli/validate.py index df842f44..a07816fa 100644 --- a/fluid_build/cli/validate.py +++ b/fluid_build/cli/validate.py @@ -809,6 +809,23 @@ def _run_contract_rules( if args.verbose: info(logger, f"GCP binding check skipped: {exc}") + # --- Governance a cloud binding cannot apply (refused at apply) ---- + # Unmapped or placeholder principals, column restrictions nothing + # enforces, a key reference for the wrong cloud: the same derivations + # `fluid apply` refuses them with, reported here at stage 2. + try: + from fluid_build.iac.governance_validation import validate_governance + + gov_errors, gov_warnings = validate_governance(contract) + for msg in gov_errors: + validation_result.add_error(msg) + validation_result.is_valid = False + for msg in gov_warnings: + validation_result.add_warning(msg) + except Exception as exc: # pragma: no cover — defensive + if args.verbose: + info(logger, f"Governance binding check skipped: {exc}") + # --- pgvector vector output-port binding checks --------------------- # A pgvector-bound expose (binding.platform: pgvector) needs a vector # dimension and non-colliding embeddings-table names — surface a clean diff --git a/fluid_build/cli/verify.py b/fluid_build/cli/verify.py index d9342614..cd00a492 100644 --- a/fluid_build/cli/verify.py +++ b/fluid_build/cli/verify.py @@ -383,12 +383,78 @@ def _bq_canonical_type(bq_type: Any) -> str: return _BQ_TYPE_SYNONYMS.get(name, name) +def _bigquery_binding_location( + expose_config: Dict[str, Any], +) -> Tuple[Dict[str, Any], Dict[str, Any]]: + """``(binding, location)`` of a BigQuery expose, named as the build's load names it. + + ``{{ env.* }}`` is resolved as ``fluid apply`` resolves it + (``_verify_athena._resolved_binding``), and a location with no project + takes the environment's (``_bigquery_load.bigquery_load_target``), so + verify reads the table the build loaded rather than a literal template. + """ + from fluid_build.build_runners._bigquery_load import bigquery_load_target + from fluid_build.cli._verify_athena import _resolved_binding + + binding = _resolved_binding(expose_config.get("binding", {})) + location = dict(binding.get("location", {}) or {}) + load_target = bigquery_load_target(binding, expose_config) + if load_target is not None and not location.get("project"): + location["project"] = load_target.get("project") or "" + return binding, location + + +def _render_table_rows(metadata: Dict[str, Any]) -> None: + """The row count line; ``num_rows`` is None when the table reports none + (the goccy emulator, a table whose metadata has no count) and nothing + counted it.""" + num_rows = metadata.get("num_rows") + if isinstance(num_rows, int): + cprint(f" 📊 Table Rows: {num_rows:,}") + else: + cprint(" 📊 Table Rows: unknown (the table reports no row count)") + if metadata.get("row_count_detail"): + cprint(f" {metadata['row_count_detail']}", markup=False) + + +def _bigquery_verify_client(bigquery: Any, project: str) -> Tuple[Any, str]: + """``(client, project)``: the binding's project, else the client's own. + + A binding with no project (and no ``GOOGLE_PROJECT`` & co.) is loaded into + the client's own project, ADC's: ``_bigquery_load.load_file`` names the + table from ``client.project``. ``Client(project="")`` keeps the empty + string rather than resolving one, so ``None`` is passed and the project + read back. + """ + from fluid_build.build_runners._bigquery_load import bigquery_client + + client = bigquery_client(bigquery, project or None) + return client, project or str(getattr(client, "project", "") or "") + + +def _no_bigquery_project(dataset: str, table: str) -> Dict[str, Any]: + return { + "status": "error", + "error": ( + f"No project for {dataset}.{table}: the binding names none, and neither " + "GOOGLE_PROJECT / GOOGLE_CLOUD_PROJECT nor the credentials supply one" + ), + "exists": False, + } + + def verify_bigquery_table( project: str, dataset: str, table: str, expected_schema: List[Dict[str, Any]], expected_region: Optional[str] = None, + *, + expose: Optional[Dict[str, Any]] = None, + contract: Optional[Dict[str, Any]] = None, + workdir: Optional[Path] = None, + reference_only: bool = False, + catalog_session_factory: Optional[Any] = None, ) -> Dict[str, Any]: """ Verify BigQuery table with multi-dimensional analysis. @@ -398,11 +464,23 @@ def verify_bigquery_table( 2. Data Types (field types) 3. Constraints (nullable/required modes) 4. Location (region/location) + 5. Retention, encryption and column restrictions, when ``expose`` declares + them (``_verify_bigquery_governance.py``) + + With ``expose`` (what ``fluid verify`` passes), two more from one count + query (``_verify_bigquery``): ``row_count`` held to the build's run + records, and ``masking`` for each ``policy.privacy.masking`` rule. + + The client is ``_bigquery_load.bigquery_client``'s: ADC, or anonymous + against ``BIGQUERY_EMULATOR_HOST``. """ try: - from google.cloud import bigquery + from fluid_build.build_runners._bigquery_load import _bigquery_module - client = bigquery.Client(project=project) + bigquery = _bigquery_module() + client, project = _bigquery_verify_client(bigquery, project) + if not project: + return _no_bigquery_project(dataset, table) table_id = f"{project}.{dataset}.{table}" # Check if table exists @@ -509,7 +587,25 @@ def verify_bigquery_table( missing_fields or extra_fields or type_mismatches or mode_mismatches or not region_match ) - return { + # Retention, encryption and column access, when the expose declares them. + governance: Dict[str, Dict[str, Any]] = {} + # #674 runs these only for an expose with a binding; the row count and + # masking dimensions below (#673) run for every expose. + if expose is not None and expose.get("binding"): + from fluid_build.cli import _verify_bigquery_governance as _bq_gov + + governance = _bq_gov.governance_dimensions( + expose, + contract=contract or {}, + bq_dataset=bq_dataset, + bq_table=bq_table, + project=project, + session_factory=catalog_session_factory, + ) + severity = _bq_gov.with_governance_severity(severity, governance) + has_issues = has_issues or any(d["status"] == "fail" for d in governance.values()) + + result: Dict[str, Any] = { "status": "mismatch" if has_issues else "match", "exists": True, "table_id": table_id, @@ -544,6 +640,28 @@ def verify_bigquery_table( "modified": bq_table.modified.isoformat() if bq_table.modified else None, }, } + if expose is not None: + from fluid_build.cli._verify_bigquery import add_data_dimensions + + add_data_dimensions( + result, + client=client, + bigquery=bigquery, + bq_table=bq_table, + expose=expose, + contract=contract or {}, + workdir=workdir or Path.cwd(), + reference_only=reference_only, + location=bq_dataset.location, + ) + if governance: + result["dimensions"].update(governance) + unchecked = _bq_gov.governance_errors(governance) + if unchecked: + # A declared policy that could not be checked is unproven, not passed. + result["status"] = "error" + result["error"] = "; ".join([e for e in (result.get("error"), *unchecked) if e]) + return result except Exception as e: LOG.error(f"Error verifying table {table}: {e}") @@ -1314,8 +1432,7 @@ def run(args: argparse.Namespace, logger: logging.Logger) -> int: # Get properties from either 'properties' or 'binding.location' properties = expose_config.get("properties", {}) if not properties: - binding = expose_config.get("binding", {}) - location = binding.get("location", {}) + binding, location = _bigquery_binding_location(expose_config) # Build target from binding project = location.get("project", "") dataset = location.get("dataset", "") @@ -1370,6 +1487,10 @@ def run(args: argparse.Namespace, logger: logging.Logger) -> int: table=table, expected_schema=fields, expected_region=region, + expose=expose_config, + contract=contract, + workdir=anchor_dir, + reference_only=reference_only, ) results[expose_name] = result @@ -1613,9 +1734,7 @@ def run(args: argparse.Namespace, logger: logging.Logger) -> int: severity_impact = severity.get("impact", "UNKNOWN") cprint(f"\n {severity_symbol} Severity: {severity_level} (Impact: {severity_impact})") - cprint(f" 📊 Table Rows: {metadata.get('num_rows', 0):,}") - if metadata.get("row_count_detail"): - cprint(f" {metadata['row_count_detail']}", markup=False) + _render_table_rows(metadata) # Dimension 1: Schema Structure cprint("\n 🔍 Dimension 1: Schema Structure") @@ -1669,7 +1788,7 @@ def run(args: argparse.Namespace, logger: logging.Logger) -> int: cprint(f" ❌ FAIL - {location.get('message', 'Location mismatch')}") # Dimension 5: Masking, when the expose declares policy.privacy.masking - # and the verifier checked it (the Glue + Athena one does). + # and the verifier checked it (the Glue + Athena and BigQuery ones do). if dimensions.get("masking") is not None: _render_masking_dimension(dimensions["masking"], "pass", "\n 🔍 Dimension 5: Masking") diff --git a/fluid_build/forge/core/artifact_fanout.py b/fluid_build/forge/core/artifact_fanout.py index a3e129ba..7def0a79 100644 --- a/fluid_build/forge/core/artifact_fanout.py +++ b/fluid_build/forge/core/artifact_fanout.py @@ -32,7 +32,7 @@ ├── odcs/product.odcs..yaml # ODCS v3.1.0 (bitol-io) — one per exposed port ├── odps-bitol/.odps.yaml # ODPS-Bitol v1.0.0 (bitol-io) ├── opds/.opds.json # OPDS v4.1 (LF/ODPI) — schema-validated - ├── schedule// # one directory per product, so + ├── schedule/[__]/ # one directory per product, so │ └── _dag.py # schedule-sync never deletes │ # another product's DAGs (Path A) └── policy/bindings.json # compiled IAM/GRANT bindings @@ -558,7 +558,7 @@ def _emit_schedule( dag_contract_path: Optional[str] = None, warn_no_env: bool = False, ) -> List[Path]: - """DAG/flow emission via ``generate schedule`` into ``/schedule//``. + """DAG/flow emission via ``generate schedule`` into ``/schedule/[__]/``. ``env`` is the ``--env`` a ``fluid apply`` DAG passes on every run. ``overlay_env`` is the overlay applied while rendering: the same env for a @@ -608,11 +608,13 @@ def _emit_schedule( ) dag_contract_path = fluid_apply.DEFAULT_CONTRACT_PATH - # One directory per product: stage 11 (``schedule-sync``, default + # One directory per product and env: stage 11 (``schedule-sync``, default # ``--delete-scope product``) mirrors it into the same-named directory of # the scheduler's DAG root, so deleting stale DAGs never reaches another - # product's files there. - scope_dir = out_dir / product_id + # product's files there, nor the same product's DAGs for another env (an + # aws and a gcp pipeline syncing to one Airflow). ```` with no + # env, ``__`` with one. + scope_dir = out_dir / fluid_apply.schedule_scope_for(product_id, env or None) scope_dir.mkdir(parents=True, exist_ok=True) args = argparse.Namespace( contract=str(contract_path), diff --git a/fluid_build/forge/core/pipeline_systems/_base.py b/fluid_build/forge/core/pipeline_systems/_base.py index 24967322..b65f2fa2 100644 --- a/fluid_build/forge/core/pipeline_systems/_base.py +++ b/fluid_build/forge/core/pipeline_systems/_base.py @@ -520,7 +520,14 @@ def _get_fluid_commands(self, config: Optional["PipelineConfig"] = None) -> Dict "fluid diff ${CONTRACT:-contract.fluid.yaml} --exit-on-drift " "--env ${FLUID_ENV:-dev}" ), - "plan": "fluid plan ${CONTRACT:-contract.fluid.yaml} --out runtime/plan.json", + # --check-sovereignty: the plan stage runs the contract's own + # sovereignty policy (the provider hook, else the policy engine) + # and a strict violation fails it, before stage 7 touches a cloud. + # A contract with no sovereignty block prints NOT CHECKED and passes. + "plan": ( + "fluid plan ${CONTRACT:-contract.fluid.yaml} --out runtime/plan.json " + "--check-sovereignty" + ), # A build id needs `--mode amend-and-build` alongside # `--build-id`: the id only FILTERS, it does not opt into running # builds (`fluid apply --help`). The retired `--build` did both. @@ -1109,10 +1116,12 @@ def _stage_specs(self, config: Optional["PipelineConfig"] = None) -> List["Stage # The plan records the mode it was made for, and stage 7's # ``apply_plan_mode_mismatch`` gate refuses any other. So # stage 6 plans for APPLY_MODE, the one mode stage 7 applies. + # --check-sovereignty makes a strict sovereignty violation fail + # the plan stage, before stage 7 reaches a cloud. command=( f'set -eu; {_needs_bundle(6)}MODE="{p("APPLY_MODE")}"; ' f"fluid plan {BUNDLE_PATH} " - f'--out runtime/plan.json --mode "$MODE" {env}' + f'--out runtime/plan.json --mode "$MODE" {env} --check-sovereignty' ), ), StageSpec( diff --git a/fluid_build/forge/core/pipeline_systems/jenkins.py b/fluid_build/forge/core/pipeline_systems/jenkins.py index e4c58fa0..6f010549 100644 --- a/fluid_build/forge/core/pipeline_systems/jenkins.py +++ b/fluid_build/forge/core/pipeline_systems/jenkins.py @@ -718,9 +718,11 @@ def when(num: int, extra: str = "") -> str: [ "mkdir -p runtime", _needs_bundle(6), + # --check-sovereignty: a strict sovereignty violation fails the + # plan stage, before stage 7 reaches a cloud. ( f"set -- {BUNDLE_PATH} {env_flag} " - f'--mode "{v("APPLY_MODE")}" --out runtime/plan.json' + f'--mode "{v("APPLY_MODE")}" --out runtime/plan.json --check-sovereignty' ), ( f'if [ "{v("PLAN_HTML")}" = "true" ]; then ' diff --git a/fluid_build/iac/backend.py b/fluid_build/iac/backend.py index 0246b051..f16af35c 100644 --- a/fluid_build/iac/backend.py +++ b/fluid_build/iac/backend.py @@ -39,6 +39,18 @@ ``_``, so ``a.b``, ``a-b`` and ``a_b``, all valid ids, would share one state again. An id outside the FLUID identifier grammar cannot key a state and is refused. The variable is new, so no state lives at a key it chose before. + +**The provider is part of the default key.** One contract deployed to two +clouds through overlays (``--env aws`` and ``--env gcp``) is one id, so a key +made of the id alone put the aws and the gcp apply in one state: each plan +then read the other cloud's resources as orphans to destroy, and +``--allow-data-loss`` would have destroyed them. Every per-contract default is +now ``fluid///terraform.tfstate`` (the GCS prefix +``fluid//``) when the caller names the provider, as ``fluid +apply`` always does. The shared legacy key ``fluid/terraform.tfstate`` is +unchanged, and so is an explicit key in the spec. State the previous default +wrote is moved by :mod:`fluid_build.iac.state_migration`, with OpenTofu's own +``init -migrate-state``; :func:`legacy_default_backend` names where it was. """ from __future__ import annotations @@ -68,6 +80,11 @@ #: prints on its state line (a newline there would forge a line of output). _CONTROL_RE = re.compile(r"[\x00-\x1f\x7f]") +#: What a provider name in a state key may hold: the IaC plugin names +#: (``aws``, ``gcp``, ``snowflake``, ``confluent``) and nothing that could add +#: a path segment or a control character to the key. +_PROVIDER_RE = re.compile(r"[a-z][a-z0-9_-]{0,31}") + #: Environment variable ``fluid apply`` reads for the state backend when #: ``--state-backend`` is not on the command line. A CI job sets it once so #: OpenTofu state lives in a bucket rather than the workspace, which CI @@ -101,13 +118,24 @@ def resolve_state_backend_spec( return (None, "default") -def default_state_key(contract: Optional[Mapping[str, Any]], *, per_contract: bool = False) -> str: +def default_state_key( + contract: Optional[Mapping[str, Any]], + *, + per_contract: bool = False, + provider: Optional[str] = None, +) -> str: """The default state key for ``contract`` — legacy unless packaging is declared. Returns :data:`LEGACY_STATE_KEY` when ``contract`` is ``None`` or resolves to the ``packaging.LEGACY`` sentinel, and the per-contract ``fluid//terraform.tfstate`` otherwise. + ``provider`` (``fluid apply`` passes the one it resolved) adds a segment to + every per-contract key, ``fluid///terraform.tfstate``, so the + same contract applied to two clouds keeps two states (see the module + docstring). The legacy shared key never takes it. A provider name outside + ``[a-z][a-z0-9_-]*`` is a ``ValueError``. + ``per_contract=True`` (the spec came from :data:`STATE_BACKEND_ENV`; see the module docstring) skips the packaging test and keys every contract by its id as written, ``fluid//terraform.tfstate``, so two distinct ids @@ -123,8 +151,9 @@ def default_state_key(contract: Optional[Mapping[str, Any]], *, per_contract: bo """ if contract is None: return LEGACY_STATE_KEY + segment = "" if provider is None else f"{_state_provider(provider)}/" if per_contract: - return f"fluid/{_state_id(contract)}/terraform.tfstate" + return f"fluid/{_state_id(contract)}/{segment}terraform.tfstate" try: resolution = resolve_packaging(contract) except PackagingError: @@ -132,7 +161,17 @@ def default_state_key(contract: Optional[Mapping[str, Any]], *, per_contract: bo if resolution is LEGACY: return LEGACY_STATE_KEY cid = safe_ident(contract.get("id") or contract.get("name") or "contract") - return f"fluid/{cid}/terraform.tfstate" + return f"fluid/{cid}/{segment}terraform.tfstate" + + +def _state_provider(provider: str) -> str: + """``provider``, once it is proven to be one safe key segment.""" + if isinstance(provider, str) and _PROVIDER_RE.fullmatch(provider): + return provider + raise ValueError( + f"provider {provider!r} cannot name a state key segment " + "([a-z][a-z0-9_-]*, at most 32 characters)" + ) def _state_id(contract: Mapping[str, Any]) -> str: @@ -153,6 +192,7 @@ def parse_backend( contract: Optional[Mapping[str, Any]] = None, *, per_contract_default: bool = False, + provider: Optional[str] = None, ) -> Optional[Dict[str, Any]]: """Parse a backend spec into a ``terraform.backend`` block. @@ -167,6 +207,8 @@ def parse_backend( contracts, the shared legacy key otherwise. ``per_contract_default`` gives every contract its own default, keyed by its id as written (``fluid apply`` sets it for a spec from :data:`STATE_BACKEND_ENV`). + ``provider`` puts the provider into every per-contract default (see + :func:`default_state_key`); an explicit key or prefix is never changed. The backend block carries no credentials — ``tofu`` reads those from the environment (``AWS_*`` / ``GOOGLE_*``). A bucket name holding @@ -184,7 +226,7 @@ def parse_backend( _check_bucket(bucket, "s3") _check_path(key, "s3", "key") if not key: - key = default_state_key(contract, per_contract=per_contract_default) + key = default_state_key(contract, per_contract=per_contract_default, provider=provider) return {"s3": {"bucket": bucket, "key": key}} if spec.startswith("gcs://"): @@ -199,7 +241,9 @@ def parse_backend( # derive it from the same per-contract default (sans filename) so # both backends isolate identically. Legacy contracts emit no # prefix at all, exactly as before. - default = default_state_key(contract, per_contract=per_contract_default) + default = default_state_key( + contract, per_contract=per_contract_default, provider=provider + ) prefix = default.rsplit("/", 1)[0] if default != LEGACY_STATE_KEY else "" if prefix: block["gcs"]["prefix"] = prefix @@ -212,6 +256,31 @@ def parse_backend( raise ValueError(f"unsupported state backend {named} — use s3:// or gcs://") +def legacy_default_backend( + spec: Optional[str], + contract: Optional[Mapping[str, Any]], + *, + per_contract_default: bool, + provider: str, +) -> Optional[Dict[str, Any]]: + """Where the default state was before the provider joined the key, or None. + + The block :func:`parse_backend` returned for the same spec and contract + without ``provider``: ``fluid//terraform.tfstate`` (GCS prefix + ``fluid/``). ``None`` when there is nothing to migrate from: local + state, an explicit key or prefix (never changed, so never moved), or a + default that did not change (the shared legacy key). Raises + ``ValueError`` exactly where :func:`parse_backend` does. + """ + legacy = parse_backend(spec, contract, per_contract_default=per_contract_default) + current = parse_backend( + spec, contract, per_contract_default=per_contract_default, provider=provider + ) + if legacy is None or current is None or legacy == current: + return None + return legacy + + def _check_bucket(bucket: str, scheme: str) -> None: if not _BUCKET_RE.fullmatch(bucket): # Not echoed: what is not a bucket name may be a credential. diff --git a/fluid_build/iac/column_access.py b/fluid_build/iac/column_access.py new file mode 100644 index 00000000..3b8503f5 --- /dev/null +++ b/fluid_build/iac/column_access.py @@ -0,0 +1,366 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Column restrictions: who may read which column, one derivation for every cloud. + +``exposes[].policy.authz.columnRestrictions[]`` (``{principal, columns, access: +allow|deny}``, in every bundled schema since 0.7.1) was read only by the MCP output +port, so no cloud enforced it. This module turns it into one answer, "the set of +identities that may read column c", which each emitter writes in its own terms and +``fluid verify`` checks against the live platform: + +* GCP (``iac/providers/gcp_governance.py``): each restricted column gets a Data + Catalog policy tag, and the fine-grained reader role on the tag goes to exactly + that set. +* AWS (:func:`lf_exclusions`): each Lake Formation grant's ``excludedColumns`` are the + restricted columns its principal is not in the set for. + +The semantics, the same on both clouds: + +* A column named in any restriction is restricted. +* ``deny``: the principal may not read the columns. +* ``allow``: the columns are readable only by the principals an ``allow`` names + (an allow list per column); a column with no ``allow`` is readable by every reader + of the expose. +* A deny beats an allow. A restriction never grants access: the readers are + intersected with the expose's readers (on GCP the ``accessPolicy`` read grants and + the expose's own ``policy.authz.readers``; on AWS the Lake Formation ``SELECT`` + grants), and an allowed principal that is not a reader is reported, not added. + With no reader at all a restriction is refused on both clouds: it would lock the + columns for everyone, not only for the principals it names. + +Principals are logical and resolve through ``binding.principals`` +(:mod:`fluid_build.iac.principals`), so the same restriction names the analyst role +on AWS and the analysts' group on GCP. A restriction the platform cannot enforce is +refused, never dropped. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import Any, Callable, Dict, FrozenSet, List, Mapping, Optional, Sequence, Tuple + +from .base import UnsupportedBindingError +from .principals import AWS, principal_map, resolve_principal + +LOG = logging.getLogger(__name__) + +ALLOW = "allow" +DENY = "deny" + +#: Lake Formation permissions that let a principal read a table's rows. +LF_READ_PERMISSIONS = frozenset({"SELECT", "ALL"}) + + +@dataclass(frozen=True) +class Restriction: + """One ``columnRestrictions[]`` entry, validated.""" + + principal: str + columns: Tuple[str, ...] + access: str + #: The rule's ``tags`` and ``labels``: descriptive, carried to the cloud object + #: that holds the restriction where it has room for them (GCP's policy tag). + tags: Tuple[str, ...] = () + labels: Tuple[Tuple[str, str], ...] = () + + +def _where(exposure: Mapping[str, Any], index: int) -> str: + expose_id = exposure.get("exposeId") or exposure.get("id") or index + return f"exposes[{expose_id}].policy.authz.columnRestrictions" + + +def restrictions_for(exposure: Mapping[str, Any], index: int = 0) -> Tuple[Restriction, ...]: + """The expose's column restrictions, or ``()``; a malformed one is refused. + + Every column must be one the expose's schema declares: a restriction on a column + the table does not have protects nothing, and on GCP it cannot be attached. + """ + policy = exposure.get("policy") if isinstance(exposure, Mapping) else None + authz = policy.get("authz") if isinstance(policy, Mapping) else None + raw = authz.get("columnRestrictions") if isinstance(authz, Mapping) else None + if not raw: + return () + where = _where(exposure, index) + if not isinstance(raw, list): + raise UnsupportedBindingError( + "column-restriction", f"{where} must be a list of restrictions.", () + ) + schema = (exposure.get("contract") or {}).get("schema") or [] + declared = {col.get("name") for col in schema if isinstance(col, Mapping)} + out: List[Restriction] = [] + for i, entry in enumerate(raw): + at = f"{where}[{i}]" + if not isinstance(entry, Mapping): + raise UnsupportedBindingError("column-restriction", f"{at} must be a mapping.", ()) + principal = entry.get("principal") + columns = entry.get("columns") + access = str(entry.get("access") or "").strip().lower() + if not isinstance(principal, str) or not principal.strip(): + raise UnsupportedBindingError( + "column-restriction", + f"{at} names no principal, so there is no one to restrict.", + ("Set principal to the logical principal the restriction applies to.",), + ) + if ( + not isinstance(columns, list) + or not columns + or not all(isinstance(c, str) and c for c in columns) + ): + raise UnsupportedBindingError( + "column-restriction", + f"{at} must list the columns it restricts.", + ("Set columns to one or more column names of this expose's schema.",), + ) + if access not in (ALLOW, DENY): + raise UnsupportedBindingError( + "column-restriction", + f"{at}.access is {entry.get('access')!r}; it must be 'allow' or 'deny', so " + "the cloud can be told which way the restriction goes.", + ( + "Set access: deny to hide the columns from the principal, or allow to make " + "the principal one of the columns' only readers.", + ), + ) + unknown = [c for c in columns if c not in declared] + if unknown: + raise UnsupportedBindingError( + "column-restriction", + f"{at} restricts {unknown}, which the expose's schema does not declare, so " + "nothing would be protected.", + ("Name columns of exposes[].contract.schema, or add the columns to it.",), + ) + tags = entry.get("tags") or () + labels = entry.get("labels") or {} + out.append( + Restriction( + principal=principal.strip(), + columns=tuple(columns), + access=access, + tags=tuple(str(t) for t in tags) if isinstance(tags, (list, tuple)) else (), + labels=( + tuple(sorted((str(k), str(v)) for k, v in labels.items())) + if isinstance(labels, Mapping) + else () + ), + ) + ) + return tuple(out) + + +def authz_readers(exposure: Mapping[str, Any]) -> Tuple[str, ...]: + """``exposes[].policy.authz.readers``: the expose's own readers, logical principals. + + Readers a column restriction narrows, alongside the ``accessPolicy`` read grants. + No emitter grants them anything: their access to the table is managed elsewhere, + and a restriction must not take a column away from them unless it names them. + """ + policy = exposure.get("policy") if isinstance(exposure, Mapping) else None + authz = policy.get("authz") if isinstance(policy, Mapping) else None + raw = authz.get("readers") if isinstance(authz, Mapping) else None + if not isinstance(raw, list): + return () + return tuple(r.strip() for r in raw if isinstance(r, str) and r.strip()) + + +def restricted_columns( + exposure: Mapping[str, Any], restrictions: Sequence[Restriction] +) -> Tuple[str, ...]: + """The restricted columns, in the schema's order.""" + named = {c for r in restrictions for c in r.columns} + schema = (exposure.get("contract") or {}).get("schema") or [] + return tuple( + col["name"] for col in schema if isinstance(col, Mapping) and col.get("name") in named + ) + + +def column_readers( + exposure: Mapping[str, Any], + restrictions: Sequence[Restriction], + resolve: Callable[[str], Tuple[str, ...]], + readers: FrozenSet[str], + *, + where: str, +) -> Dict[str, FrozenSet[str]]: + """``{restricted column: identities that may read it}`` (see the module docstring). + + ``resolve`` turns a logical principal into its identities on the platform, and + ``readers`` is every identity that reads the expose there. + """ + allowed: Dict[str, set[str]] = {} + denied: Dict[str, set[str]] = {} + for restriction in restrictions: + identities = set(resolve(restriction.principal)) + target = allowed if restriction.access == ALLOW else denied + for column in restriction.columns: + target.setdefault(column, set()).update(identities) + out: Dict[str, FrozenSet[str]] = {} + for column in restricted_columns(exposure, restrictions): + if column in allowed: + not_readers = sorted(allowed[column] - readers) + if not_readers: + LOG.warning( + "column_restriction_allow_not_a_reader %s column=%s identities=%s: a " + "restriction never grants access, so these identities, which have no " + "read grant on the expose, still cannot read it", + where, + column, + not_readers, + ) + base = allowed[column] & readers + else: + base = set(readers) + out[column] = frozenset(base - denied.get(column, set())) + return out + + +# ── AWS: Lake Formation excluded columns ──────────────────────────────── + + +def _lf_grant_where(where: str, index: int) -> str: + return f"{where}: governance.lakeFormation.grants[{index}]" + + +def lf_exclusions( + exposure: Mapping[str, Any], binding: Mapping[str, Any], index: int = 0 +) -> Optional[Dict[int, Tuple[str, ...]]]: + """``{grant index: excluded columns}`` for an AWS binding's Lake Formation grants. + + ``None`` when the expose declares no column restriction: every existing overlay + then emits what it did, its hand-written ``excludedColumns`` included. With + restrictions, each read grant's excluded columns are the restricted ones its + principal may not read, and a grant that already declares ``excludedColumns`` + (or a ``columns`` projection) must agree, or the emit is refused rather than + left with two policies that disagree. + """ + restrictions = restrictions_for(exposure, index) + if not restrictions: + return None + where = _where(exposure, index) + gov = (binding.get("governance") or {}).get("lakeFormation") or {} + grants = gov.get("grants") if isinstance(gov, Mapping) else None + if not grants: + raise UnsupportedBindingError( + "column-restriction-unenforceable", + f"{where} restricts columns, but this aws binding declares no " + "governance.lakeFormation.grants, the only column-level control the AWS emitter " + "writes. Nothing would enforce the restriction.", + ( + "Add governance.lakeFormation to the aws overlay's binding: registerLocation: " + "true and a grant for each reader; forge-cli then writes the restricted " + "columns into each grant's excluded columns.", + ), + ) + mapping = principal_map(binding) + + def resolve(principal: str) -> Tuple[str, ...]: + return resolve_principal(principal, mapping, platform=AWS, where=where) + + readers = frozenset( + str(g.get("principal")) + for g in grants + if isinstance(g, Mapping) + and g.get("principal") + and LF_READ_PERMISSIONS & set(g.get("permissions") or ()) + ) + by_column = column_readers(exposure, restrictions, resolve, readers, where=where) + out: Dict[int, Tuple[str, ...]] = {} + for i, grant in enumerate(grants): + if not isinstance(grant, Mapping): + continue + principal = str(grant.get("principal") or "") + if not principal or not LF_READ_PERMISSIONS & set(grant.get("permissions") or ()): + continue + excluded = tuple(c for c, allowed in by_column.items() if principal not in allowed) + declared_excluded = grant.get("excludedColumns") + declared_columns = grant.get("columns") + if declared_columns: + leaked = [c for c in declared_columns if c in excluded] + if leaked: + raise UnsupportedBindingError( + "column-restriction-conflict", + f"{_lf_grant_where(where, i)} grants {principal} the columns {leaked}, " + "which the contract's column restrictions do not let it read.", + ("Drop those columns from the grant's columns, or change the restriction.",), + ) + continue + if declared_excluded is not None and set(declared_excluded) != set(excluded): + raise UnsupportedBindingError( + "column-restriction-conflict", + f"{_lf_grant_where(where, i)} excludes {sorted(declared_excluded)} for " + f"{principal}, but the contract's column restrictions exclude " + f"{sorted(excluded)}. The two must agree.", + ( + "Remove excludedColumns from the grant (forge-cli writes it from the " + "restrictions), or change it or the restrictions until they match.", + ), + ) + if excluded: + out[i] = excluded + return out + + +def lf_expected_exclusions( + exposure: Mapping[str, Any], binding: Mapping[str, Any], index: int = 0 +) -> Dict[str, Tuple[str, ...]]: + """``{principal ARN: columns it must not be able to read}``, for ``fluid verify``. + + Every read grantee whose restricted columns are not all readable, and every + identity a ``deny`` names, whether or not the contract grants it anything: a + grant made outside the contract must not reach a denied column either. + ``IAM_ALLOWED_PRINCIPALS`` is held to every restricted column, since a table that + still grants it (Lake Formation's default for a new table) lets any IAM principal + with S3 access read every column. + """ + restrictions = restrictions_for(exposure, index) + if not restrictions: + return {} + exclusions = lf_exclusions(exposure, binding, index) or {} + grants = ((binding.get("governance") or {}).get("lakeFormation") or {}).get("grants") or [] + where = _where(exposure, index) + mapping = principal_map(binding) + expected: Dict[str, set[str]] = {} + for i, columns in exclusions.items(): + expected.setdefault(str(grants[i].get("principal")), set()).update(columns) + for restriction in restrictions: + if restriction.access != DENY: + continue + for identity in resolve_principal( + restriction.principal, mapping, platform=AWS, where=where + ): + expected.setdefault(identity, set()).update(restriction.columns) + expected.setdefault("IAM_ALLOWED_PRINCIPALS", set()).update( + restricted_columns(exposure, restrictions) + ) + order = restricted_columns(exposure, restrictions) + return { + principal: tuple(c for c in order if c in columns) + for principal, columns in sorted(expected.items()) + if columns + } + + +__all__ = [ + "ALLOW", + "DENY", + "LF_READ_PERMISSIONS", + "Restriction", + "authz_readers", + "column_readers", + "lf_exclusions", + "lf_expected_exclusions", + "restricted_columns", + "restrictions_for", +] diff --git a/fluid_build/iac/governance_validation.py b/fluid_build/iac/governance_validation.py new file mode 100644 index 00000000..4a68db08 --- /dev/null +++ b/fluid_build/iac/governance_validation.py @@ -0,0 +1,109 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Validate-time gate for the governance a cloud binding cannot apply. + +``fluid generate iac``, ``fluid plan`` and ``fluid apply`` refuse a principal that +is unmapped or a placeholder, a column restriction nothing on the binding enforces, +a Cloud KMS key on AWS or an AWS key on GCP, and retention or encryption on a GCP +target that is not a BigQuery table. This runs the SAME derivations +(``iac/providers/gcp_governance.py``, ``iac/column_access.py``, +``iac/principals.py``) so ``fluid validate`` reports each refusal at stage 2, with +the same message, instead of at apply. The same shape as ``validate_gcp_binding``. +""" + +from __future__ import annotations + +from typing import Any, List, Mapping, Tuple + +from .access import normalize_access_grants +from .base import UnsupportedBindingError +from .principals import gcp_grants, principal_map +from .provider_match import is_cloud + + +def _message(exc: UnsupportedBindingError) -> str: + remedy = " ".join(exc.remediation) + return f"{exc} {remedy}".strip() + + +def _lf_grants(binding: Mapping[str, Any]) -> List[Any]: + governance = binding.get("governance") if isinstance(binding, Mapping) else None + lake = governance.get("lakeFormation") if isinstance(governance, Mapping) else None + grants = lake.get("grants") if isinstance(lake, Mapping) else None + return list(grants) if isinstance(grants, list) else [] + + +def validate_governance(contract: Mapping[str, Any]) -> Tuple[List[str], List[str]]: + """``(errors, warnings)`` for the contract's cloud bindings. + + Each binding is dispatched on its platform first: an ``aws`` binding is checked + against what the AWS emitter writes, whatever its format resolves to on GCP + (``resolve_gcp_target`` resolves an Iceberg format with a bucket to GCP Iceberg + storage on any platform). + """ + from .providers import gcp as _gcp + from .providers import gcp_governance as _gov + + errors: List[str] = [] + warnings: List[str] = [] + grants = normalize_access_grants(contract) + unenforced: List[str] = [] + try: + _gov.refuse_mixed_dataset_encryption(contract) + except UnsupportedBindingError as exc: + errors.append(_message(exc)) + for index, exposure in enumerate(contract.get("exposes") or []): + if not isinstance(exposure, Mapping): + continue + binding = exposure.get("binding") or {} + if not isinstance(binding, Mapping): + continue + try: + principal_map(binding) + if is_cloud(binding, "aws"): + from .providers.aws import lf_column_exclusions + + lf_column_exclusions(exposure, binding, index) + if not _lf_grants(binding): + unenforced.append(str(exposure.get("exposeId") or index)) + continue + if not _gov.gcp_owned(binding): + continue + target = _gcp.resolve_gcp_target(binding) + if target in (_gcp.BIGQUERY_TABLE, _gcp.BIGQUERY_VIEW): + _gov.validate_bigquery_governance( + contract, exposure, index, is_view=(target == _gcp.BIGQUERY_VIEW) + ) + elif target is not None: + _gov.refuse_unsupported_target(exposure, index, target) + if target in (_gcp.GCS_BUCKET, _gcp.ICEBERG_STORAGE): + gcp_grants(grants, binding, where=f"exposes[{index}] accessPolicy") + except UnsupportedBindingError as exc: + errors.append(_message(exc)) + if grants and unenforced: + # Only here is the contract's access intent unenforced on AWS: a binding + # with Lake Formation grants is how the aws overlay says who reads, and + # warning for it too failed `fluid validate --strict` for every contract + # that carries accessPolicy for its gcp deployment. + warnings.append( + f"accessPolicy.grants are not enforced on aws binding(s) {', '.join(unenforced)}: " + "the AWS emitter does not write accessPolicy, and these bindings declare no " + "governance.lakeFormation.grants, the AWS form of who may read the table. Add " + "them to the aws overlay's binding (column restrictions then narrow them)." + ) + return errors, warnings + + +__all__ = ["validate_governance"] diff --git a/fluid_build/iac/principals.py b/fluid_build/iac/principals.py new file mode 100644 index 00000000..862a30d9 --- /dev/null +++ b/fluid_build/iac/principals.py @@ -0,0 +1,337 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Logical principals, and the cloud identities an environment's binding maps them to. + +A base contract names principals in ``accessPolicy.grants[].principal`` and +``exposes[].policy.authz.columnRestrictions[].principal``. One contract deploys +to several clouds, so those names are LOGICAL: ``group:data-platform@northwind.example`` +stands for "the data platform readers", which is an IAM role ARN on AWS and a +Google group or service account on GCP. The binding, the one part of an expose an +environment's overlay patches, says which: ``binding.principals`` (fluid-schema +0.7.6) maps each logical principal to one identity or a list of them on the +binding's platform. ``[]`` says, explicitly, that the principal has no identity +on that cloud, so nothing is granted to it there. + +Rules, the same on every cloud: + +* With ``binding.principals`` present, every principal the contract names for the + expose must be a key in it. An unmapped one is refused (``principal-unmapped``), + never emitted as written: a placeholder in an access list is either rejected by + the cloud or, worse, granted to whoever owns that name. +* Without it, the principal is used as written (every contract before this field + keeps emitting what it emitted, when that was a real identity), except that on + GCP a placeholder is refused (``principal-placeholder``): a principal in a + reserved top-level domain (``.example``, ``.test``, ``.invalid``, ``.localhost``: + RFC 2606 section 2 and RFC 6761), which no real identity can have, and one that + is not an IAM member at all (no domain, as ``group:data-platform`` or a bare + ``analysts``, or a prefix IAM does not know, as ``role:analyst``). BigQuery and + Cloud Storage refuse an access entry for either, so it was never a working grant. +* Every mapped identity is checked for the platform's shape: an IAM member + (``user:``, ``group:``, ``serviceAccount:``, ``domain:``) on GCP, an IAM ARN on + AWS (``principal-invalid``). On AWS an unmapped principal must already be an ARN, + since it can only be matched against the Lake Formation grants' ARNs. + +Prior art, adapted rather than depended on (none is a library for this): + +* ODCS v3 keeps ``roles[]`` at the top of a contract and ``servers[].roles`` on + each server, so a role is declared once and bound per deployment; the server is + the part that knows the platform, as the binding is here. +* dbt ``grants`` resolve the grantee per target (``{{ target.name }}``), and the + Terraform modules for BigQuery (terraform-google-modules/bigquery ``access``, + cloud-foundation-fabric's ``iam`` maps) take the real identities as per-environment + inputs while the module stays the same: the neutral part and the binding part are + separate, which is the split ``accessPolicy`` / ``binding.principals`` makes. +""" + +from __future__ import annotations + +import re +from typing import Any, Dict, Mapping, Optional, Tuple + +from .access import AccessGrant, _split_principal +from .base import UnsupportedBindingError + +#: Top-level domains reserved for documentation and testing: RFC 2606 section 2 +#: and RFC 6761 section 6. No registrar issues them, so no identity has one. +RESERVED_TLDS = frozenset({"example", "test", "invalid", "localhost"}) + +#: IAM member prefixes, lower-cased, to the spelling IAM expects. +_GCP_PREFIXES = { + "user": "user", + "group": "group", + "serviceaccount": "serviceAccount", + "domain": "domain", +} + +#: An email-shaped IAM member, and a ``domain:`` member. +_GCP_EMAIL_MEMBER_RE = re.compile(r"(user|group|serviceAccount):[^@\s:]+@[A-Za-z0-9.-]+") +_GCP_DOMAIN_MEMBER_RE = re.compile(r"domain:[A-Za-z0-9.-]+") + +#: An IAM principal ARN, as ``governance.lakeFormation.grants[].principal`` takes it +#: (the schema's pattern; the rest may carry an ``{{ env.* }}`` template). +_AWS_PRINCIPAL_RE = re.compile(r"arn:aws[a-z0-9-]*:iam::\S+") + +#: An ``{{ env.NAME }}`` template, which ``fluid apply`` resolves before the emitter +#: runs and ``fluid validate`` does not: an identity is checked with each template +#: standing in for one plain segment, so ``serviceAccount:x@{{ env.P }}.iam...`` +#: validates as it will emit. +_ENV_TEMPLATE_RE = re.compile(r"\{\{\s*env\.[A-Za-z_][A-Za-z0-9_]*\s*\}\}") + +#: Platforms this module resolves identities for. +GCP = "gcp" +AWS = "aws" + +PrincipalMap = Dict[str, Tuple[str, ...]] + + +def logical_key(raw: Any) -> str: + """A principal as written, with its type prefix spelled one way. + + ``serviceaccount:x`` and ``serviceAccount:x`` name the same principal, as + ``accessPolicy`` has always read them (``iac/access.py``); the rest of the + string is kept exactly, since the mapping is looked up by it. + """ + text = str(raw or "").strip() + head, sep, rest = text.partition(":") + canonical = _GCP_PREFIXES.get(head.strip().lower()) + if sep and canonical and rest.strip(): + return f"{canonical}:{rest.strip()}" + return text + + +def grant_key(grant: AccessGrant) -> str: + """The logical key of an ``accessPolicy`` grant (its declared or inferred type).""" + return f"{grant.principal_type}:{grant.principal}" + + +def principal_map(binding: Mapping[str, Any]) -> Optional[PrincipalMap]: + """``binding.principals`` as ``{logical key: identities}``, or ``None`` when absent.""" + raw = binding.get("principals") if isinstance(binding, Mapping) else None + if raw is None: + return None + if not isinstance(raw, Mapping): + raise UnsupportedBindingError( + "principal-map", + f"binding.principals must be a mapping of logical principal to identity, got " + f"{type(raw).__name__}.", + ("Write binding.principals as {: }.",), + ) + out: PrincipalMap = {} + for key, value in raw.items(): + name = logical_key(key) + if not name: + raise UnsupportedBindingError( + "principal-map", "binding.principals has an empty key.", () + ) + if isinstance(value, str): + identities: Tuple[str, ...] = (value.strip(),) + elif isinstance(value, (list, tuple)): + identities = tuple(str(item).strip() for item in value if isinstance(item, str)) + if len(identities) != len(value): + identities = identities + ("",) + else: + identities = ("",) + if any(not identity for identity in identities): + raise UnsupportedBindingError( + "principal-map", + f"binding.principals[{name!r}] must be an identity, a list of identities, or " + "[] for none on this cloud.", + (), + ) + out[name] = identities + return out + + +def _reserved_tld(identity: str) -> Optional[str]: + """The reserved top-level domain ``identity`` ends in, if any.""" + _, _, member = identity.partition(":") + domain = member.rsplit("@", 1)[-1].strip().rstrip(".").lower() + tld = domain.rsplit(".", 1)[-1] if domain else "" + return tld if tld in RESERVED_TLDS else None + + +def _shape(identity: str) -> str: + """``identity`` with each ``{{ env.* }}`` template replaced by a plain segment.""" + return _ENV_TEMPLATE_RE.sub("env", identity) + + +def _gcp_identity(identity: str, *, raw: str, where: str, mapped: bool = True) -> str: + member = logical_key(identity) + shape = _shape(member) + iam_shaped = bool( + _GCP_EMAIL_MEMBER_RE.fullmatch(shape) or _GCP_DOMAIN_MEMBER_RE.fullmatch(shape) + ) + if not mapped and not iam_shaped: + # Unmapped and not an IAM member (no domain, or a prefix IAM does not + # know, such as ``role:``): it can only be a logical name, and BigQuery + # and Cloud Storage refuse it at apply. Refused here instead, like a + # reserved-TLD placeholder. + raise UnsupportedBindingError( + "principal-placeholder", + f"{where}: principal {raw!r} is not a GCP IAM member (user:, group: or " + "serviceAccount: with an email address, or domain:), and this binding's " + "binding.principals does not map it, so it can only be a logical name. " + f"Emitted as written it would be {member!r}, which the cloud refuses.", + ( + "Map the logical principal in this environment's binding.principals to the " + "real group or service account it stands for, e.g. " + f"principals: {{'{raw}': 'group:data-platform@yourcompany.com'}}.", + "Or write it in accessPolicy as the IAM member itself, e.g. " + "group:data-platform@yourcompany.com.", + ), + ) + if not iam_shaped: + raise UnsupportedBindingError( + "principal-invalid", + f"{where}: principal {raw!r} resolves to {identity!r}, which is not a GCP IAM " + "member (user:, group:, serviceAccount: or domain:).", + ( + "Map it in binding.principals to an IAM member such as " + "group:readers@yourcompany.com or " + "serviceAccount:pipeline@your-project.iam.gserviceaccount.com.", + ), + ) + tld = _reserved_tld(shape) + if tld: + raise UnsupportedBindingError( + "principal-placeholder", + f"{where}: principal {raw!r} resolves to {member!r}, a placeholder: .{tld} is a " + "reserved top-level domain (RFC 2606), so no real identity has it, and BigQuery " + "refuses an access entry for an identity that does not exist.", + ( + "Map the logical principal in this environment's binding.principals to the " + "real group or service account it stands for, e.g. " + f"principals: {{'{raw}': 'group:data-platform@yourcompany.com'}}.", + "Map it to [] if it has no identity on this cloud; nothing is granted to it.", + ), + ) + return member + + +def _aws_identity(identity: str, *, raw: str, where: str, mapped: bool = True) -> str: + if not _AWS_PRINCIPAL_RE.fullmatch(_shape(identity)): + raise UnsupportedBindingError( + "principal-invalid", + f"{where}: principal {raw!r} resolves to {identity!r}, which is not an IAM " + "principal ARN (arn:aws:iam:::role/ or :user/).", + ("Map it in binding.principals to the IAM role or user ARN it stands for.",), + ) + return identity + + +def resolve_principal( + raw: Any, + mapping: Optional[PrincipalMap], + *, + platform: str, + where: str, + fallback: Optional[str] = None, +) -> Tuple[str, ...]: + """The ``platform`` identities logical principal ``raw`` stands for. + + ``mapping`` is :func:`principal_map` of the binding. ``fallback`` is another + spelling of the same principal to look it up by (a legacy unprefixed grant's + bare address). Raises :class:`UnsupportedBindingError` for an unmapped + principal, a placeholder or an identity of the wrong shape; returns ``()`` + for one mapped to ``[]``. + """ + key = logical_key(raw) + if mapping is not None: + identities = mapping.get(key) + if identities is None and fallback is not None: + identities = mapping.get(logical_key(fallback)) + if identities is None: + example = ( + "group:data-platform@yourcompany.com" + if platform == GCP + else "arn:aws:iam:::role/" + ) + raise UnsupportedBindingError( + "principal-unmapped", + f"{where} names principal {key!r}, and this {platform} binding's " + "binding.principals does not map it to an identity. forge-cli never emits a " + "logical principal as written once a binding maps principals.", + ( + f"Add it to binding.principals in the {platform} overlay: " + f"{{'{key}': '{example}'}}.", + "Map it to [] if it has no identity on this cloud; nothing is granted to it.", + ), + ) + elif platform == GCP: + # Unmapped: the principal as written, exactly as every contract before + # ``binding.principals`` emitted it; a placeholder (a reserved TLD, or no + # IAM member shape at all) is refused. + split = _split_principal(key) + identities = (f"{split[1]}:{split[0]}",) if split else (key,) + else: + identities = (key,) + check = _gcp_identity if platform == GCP else _aws_identity + mapped = mapping is not None + return tuple( + check(identity, raw=key, where=where, mapped=mapped or platform == AWS) + for identity in identities + ) + + +def member_grant(member: str, template: AccessGrant) -> AccessGrant: + """``template``'s permissions and scope, granted to IAM member ``member``.""" + head, _, rest = member.partition(":") + return AccessGrant( + principal=rest, + principal_type=_GCP_PREFIXES.get(head.lower(), head), + permissions=template.permissions, + resources=template.resources, + ) + + +def gcp_grants( + grants: Tuple[AccessGrant, ...], binding: Mapping[str, Any], *, where: str +) -> Tuple[AccessGrant, ...]: + """``accessPolicy`` grants with each logical principal replaced by its GCP identities. + + Order is kept and duplicates collapse, so the emit stays deterministic. + """ + mapping = principal_map(binding) + out: list[AccessGrant] = [] + seen: set[Tuple[str, str, Tuple[str, ...], Tuple[str, ...]]] = set() + for grant in grants: + members = resolve_principal( + grant_key(grant), + mapping, + platform=GCP, + where=where, + fallback=grant.principal, + ) + for member in members: + mapped = member_grant(member, grant) + dedupe = (mapped.principal, mapped.principal_type, mapped.permissions, mapped.resources) + if dedupe in seen: + continue + seen.add(dedupe) + out.append(mapped) + return tuple(out) + + +__all__ = [ + "AWS", + "GCP", + "RESERVED_TLDS", + "gcp_grants", + "grant_key", + "logical_key", + "member_grant", + "principal_map", + "resolve_principal", +] diff --git a/fluid_build/iac/providers/aws.py b/fluid_build/iac/providers/aws.py index 7ac76dc3..a958f9ef 100644 --- a/fluid_build/iac/providers/aws.py +++ b/fluid_build/iac/providers/aws.py @@ -55,6 +55,7 @@ from ...providers._sql_safety import quote_string_literal, validate_ident from ...providers.aws.util import warehouse as _warehouse +from .. import column_access from ..base import UnsupportedBindingError from ..importer import ImportBlock from ..naming import TofuExpr, safe_ident, tofu_ref @@ -370,13 +371,17 @@ def emit( # resource_lf_tags association references them. _emit_lf_account_settings(resources, contract, cid, base_tags) - for exposure in contract.get("exposes") or []: + for index, exposure in enumerate(contract.get("exposes") or []): binding = exposure.get("binding") or {} if not is_cloud(binding, "aws"): continue loc = binding.get("location") or {} fmt = binding.get("format") or "parquet" schema = (exposure.get("contract") or {}).get("schema") or [] + # The contract's column restrictions, as each Lake Formation grant's + # excluded columns (``iac/column_access.py``); refused when nothing + # on this binding could enforce them. + exclusions = lf_column_exclusions(exposure, binding, index) placement = _placement(packaging, exposure) tags = _tags_for(base_tags, placement) _emit_glue( @@ -386,12 +391,21 @@ def emit( _emit_kinesis(resources, loc, cid, tags) _emit_redshift_serverless(resources, loc, cid, tags) _emit_redshift_external_schema(resources, loc, cid, tags) - _check_lf_grant_columns(binding, loc, fmt, schema) + _check_lf_grant_columns(binding, loc, fmt, schema, exclusions or {}) # Per-exposure Lake Formation: location registration, # principal grants, LF-tag associations, row/column filters. # Only fires when the binding carries a governance.lakeFormation # block — every existing AWS contract is unaffected. - _emit_lakeformation(resources, binding, loc, fmt, cid, tags, placement=placement) + _emit_lakeformation( + resources, + binding, + loc, + fmt, + cid, + tags, + placement=placement, + exclusions=exclusions or {}, + ) # Retention (exposes[].lifecycle) and encryption at rest # (binding.encryption), per bucket this product owns. Nothing is # added for a contract that declares neither. @@ -1859,6 +1873,73 @@ def _emit_lf_bucket_policy_data( } +def lf_column_exclusions( + exposure: Mapping[str, Any], binding: Mapping[str, Any], index: int = 0 +) -> Optional[Dict[int, Tuple[str, ...]]]: + """``column_access.lf_exclusions``, refused where no Lake Formation grant can carry it. + + The exclusions reach a grant only as ``table_with_columns`` on a Glue-catalog + table, so a restriction on another format, or on a binding that names no + database and table, would be dropped by :func:`_emit_lakeformation`. + """ + exclusions = column_access.lf_exclusions(exposure, binding, index) + loc = binding.get("location") or {} + fmt = str(binding.get("format") or "parquet") + if exclusions is not None and ( + fmt.lower() not in _GLUE_CATALOG_FORMATS or not loc.get("database") or not loc.get("table") + ): + raise UnsupportedBindingError( + "column-restriction-unenforceable", + f"exposes[{exposure.get('exposeId') or index}] restricts columns, but its {fmt} " + "binding names no Glue table; the AWS emitter enforces column restrictions " + "through Lake Formation grants on a Glue-catalog table (location.database and " + "location.table), and without one they would not reach the grants.", + ("Restrict the columns of the Glue table expose instead.",), + ) + if exclusions: + _refuse_grants_left_no_column(exposure, binding, exclusions, index) + return exclusions + + +def _refuse_grants_left_no_column( + exposure: Mapping[str, Any], + binding: Mapping[str, Any], + exclusions: Mapping[int, Tuple[str, ...]], + index: int, +) -> None: + """Refuse a read grant whose principal the column restrictions leave no column. + + The grant's excluded columns would be every column of the table, and + :func:`_emit_lakeformation` would write a column wildcard that excludes all + of them: the same grant :func:`_check_lf_grant_columns` refuses when it is + written by hand as ``excludedColumns``. Run here, where ``fluid validate`` + and the emitter both derive the exclusions, so stage 2 refuses it too. + """ + declared = { + str(col.get("name")) + for col in ((exposure.get("contract") or {}).get("schema") or []) + if isinstance(col, Mapping) and col.get("name") + } + if not declared: + return + grants = ((binding.get("governance") or {}).get("lakeFormation") or {}).get("grants") or [] + for idx, excluded in sorted(exclusions.items()): + if not declared <= set(excluded): + continue + principal = grants[idx].get("principal") if idx < len(grants) else None + raise UnsupportedBindingError( + "lakeformation-grant-columns", + f"exposes[{exposure.get('exposeId') or index}] " + f"governance.lakeFormation.grants[{idx}] gives {principal} read access, but " + "the contract's column restrictions let it read no column of the table, so " + "the grant would give it nothing to read.", + ( + "Remove the grant if the principal should read nothing.", + "Allow it at least one column in policy.authz.columnRestrictions.", + ), + ) + + def _emit_lakeformation( resources: Dict[str, Any], binding: Mapping[str, Any], @@ -1868,10 +1949,16 @@ def _emit_lakeformation( tags: Dict[str, str], *, placement: _Placement = _LEGACY_PLACEMENT, + exclusions: Optional[Mapping[int, Tuple[str, ...]]] = None, ) -> None: """Emit per-exposure LF resources. No-op when the binding has no ``governance.lakeFormation`` block. + ``exclusions`` is ``column_access.lf_exclusions``: for each grant index, the + columns the contract's ``policy.authz.columnRestrictions`` do not let that + grant's principal read. They become the grant's ``excluded_column_names``; + a hand-written ``excludedColumns`` was checked to agree with them. + Under a REFERENCED bucket the grants narrow to the binding's ``location.path`` prefix rather than the bucket root (RFC §Security — "LF registers the ``path`` prefix, not the bucket"), and every Glue / @@ -1926,7 +2013,7 @@ def _emit_lakeformation( if gp: body["permissions_with_grant_option"] = list(gp) cols = grant.get("columns") - excluded = grant.get("excludedColumns") + excluded = (exclusions or {}).get(idx) or grant.get("excludedColumns") if (cols or excluded) and table_key: if cols and excluded: # One block cannot hold both: Lake Formation takes either a column @@ -2074,6 +2161,7 @@ def _check_lf_grant_columns( loc: Mapping[str, Any], fmt: str, schema: List[Mapping[str, Any]], + exclusions: Optional[Mapping[int, Tuple[str, ...]]] = None, ) -> None: """Refuse a Lake Formation grant whose ``columns`` / ``excludedColumns`` name a column the table does not have, or exclude every column it has, or that sits @@ -2085,6 +2173,8 @@ def _check_lf_grant_columns( plan`` passes any name. A misspelt exclusion is the dangerous one: the grant becomes a column wildcard that still includes the column it meant to hide. Only a binding whose grants are emitted against a Glue table is checked. + ``exclusions`` are the ones the column restrictions derive + (:func:`lf_column_exclusions`): what the emitter writes, so what is checked. """ gov = (binding.get("governance") or {}).get("lakeFormation") or {} if not gov or str(fmt or "").lower() not in _GLUE_CATALOG_FORMATS: @@ -2095,7 +2185,7 @@ def _check_lf_grant_columns( limited: List[int] = [] for idx, grant in enumerate(gov.get("grants") or []): cols = list(grant.get("columns") or []) - excluded = list(grant.get("excludedColumns") or []) + excluded = list((exclusions or {}).get(idx) or grant.get("excludedColumns") or []) unknown = [c for c in cols + excluded if c not in declared] if unknown: raise UnsupportedBindingError( diff --git a/fluid_build/iac/providers/gcp.py b/fluid_build/iac/providers/gcp.py index 4a0ef302..bb545412 100644 --- a/fluid_build/iac/providers/gcp.py +++ b/fluid_build/iac/providers/gcp.py @@ -40,6 +40,7 @@ from __future__ import annotations +import hashlib import json from dataclasses import dataclass from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple @@ -52,15 +53,17 @@ role_grants, ) from ..importer import ImportBlock -from ..naming import safe_ident, tofu_ref +from ..naming import TofuExpr, safe_ident, tofu_ref from ..packaging import ( ContainerDecision, PackagingError, PackagingResolution, resolve_packaging, ) +from ..principals import gcp_grants from ..provider_match import is_cloud from ..versions import required_providers +from . import gcp_governance as _gov # FLUID column type → BigQuery type (best-effort; unknown types upper-cased). _BQ_TYPES = { @@ -151,9 +154,9 @@ def _bq_type(raw: Any) -> str: return str(raw).upper() if raw else "STRING" -def _bq_schema(schema: List[Mapping[str, Any]]) -> str: - """FLUID contract schema → BigQuery schema JSON string.""" - fields = [ +def _bq_fields(schema: List[Mapping[str, Any]]) -> List[Dict[str, Any]]: + """FLUID contract schema → BigQuery schema fields.""" + return [ { "name": col.get("name"), "type": _bq_type(col.get("type")), @@ -162,12 +165,48 @@ def _bq_schema(schema: List[Mapping[str, Any]]) -> str: } for col in schema or [] ] - return json.dumps(fields, sort_keys=True) + + +def _bq_schema(schema: List[Mapping[str, Any]]) -> str: + """FLUID contract schema → BigQuery schema JSON string.""" + return json.dumps(_bq_fields(schema), sort_keys=True) + + +def _literal(value: Any) -> Any: + """``value`` with OpenTofu interpolation neutralised, as the renderer does it.""" + if isinstance(value, str): + return value.replace("${", "$${").replace("%{", "%%{") + return value + + +def _bq_schema_with_policy_tags( + schema: List[Mapping[str, Any]], tags: Mapping[str, Any] +) -> TofuExpr: + """The schema JSON with each restricted column's ``policyTags`` (dbt's ``policy_tags``). + + The tag's name is only known at apply, so it is an interpolation inside the + JSON string, and the string must reach OpenTofu unescaped. Every string that + came from the contract is escaped here instead, exactly as the renderer would + escape the whole value, so a column description still cannot inject one. + """ + fields = [] + for field in _bq_fields(schema): + escaped = {key: _literal(value) for key, value in field.items()} + tag = tags.get(field.get("name") or "") + if tag is not None: + escaped["policyTags"] = {"names": [str(tag)]} + fields.append(escaped) + return TofuExpr(json.dumps(fields, sort_keys=True)) def _bq_access_entries(grants: Sequence[AccessGrant]) -> List[Dict[str, str]]: """Normalized grants → a ``google_bigquery_dataset`` ``access`` block. + No longer emitted: that block is AUTHORITATIVE for the dataset's whole ACL, so + the grants are ``google_bigquery_dataset_iam_member`` resources now + (:func:`_emit_dataset_iam`). Kept, with its field choice, for callers outside + the emitter. + The BigQuery field is chosen from the grant's **declared** principal type rather than guessed from the string. The previous heuristic (``"@" in principal`` → user, else group) mis-filed every group as @@ -329,8 +368,18 @@ class GcpIacPlugin: ) def emit( - self, contract: Mapping[str, Any], actions: Iterable[Mapping[str, Any]] = () + self, + contract: Mapping[str, Any], + actions: Iterable[Mapping[str, Any]] = (), + *, + enforce_sovereignty: bool = True, ) -> Dict[str, Any]: + """The contract's GCP resources, refused when they land outside its sovereignty. + + ``enforce_sovereignty=False`` returns them unchecked, for a caller that + runs the same check itself and reports it (``GcpProvider.validate_sovereignty``, + what ``fluid plan --check-sovereignty`` reads). + """ resources: Dict[str, Dict[str, Any]] = {} cid = safe_ident(contract.get("id") or contract.get("name") or "product") base_labels = {"managed_by": "fluid", "fluid_contract": cid} @@ -338,16 +387,31 @@ def emit( # resource. Read from the schema-valid `accessPolicy` surface, with # the deprecated (schema-invalid) `metadata.policies` appended for # back-compat — see `iac/access.py` for why that split exists. - grants = normalize_access_grants(contract) + contract_grants = normalize_access_grants(contract) packaging = resolve_packaging(contract) + # One key per dataset: an unkeyed table in a keyed dataset would be + # replaced on every apply (``gcp_governance.refuse_mixed_dataset_encryption``). + _gov.refuse_mixed_dataset_encryption(contract) - for exposure in contract.get("exposes") or []: + for index, exposure in enumerate(contract.get("exposes") or []): binding = exposure.get("binding") or {} target = resolve_gcp_target(binding) loc = binding.get("location") or {} schema = (exposure.get("contract") or {}).get("schema") or [] placement = _placement(packaging, exposure) labels = _labels_for(base_labels, placement) + # The contract's principals are logical: each GCP expose's binding maps + # them to real IAM members (``binding.principals``), and an unmapped or + # placeholder principal is refused rather than written into IAM. + grants = ( + gcp_grants( + contract_grants, + binding, + where=f"exposes[{_expose_id(exposure) or index}] accessPolicy", + ) + if target in _GRANTED_TARGETS + else contract_grants + ) if target in (BIGQUERY_TABLE, BIGQUERY_VIEW): _emit_bigquery( resources, @@ -359,8 +423,17 @@ def emit( is_view=(target == BIGQUERY_VIEW), grants=grants, placement=placement, + governance=_bigquery_governance( + contract, exposure, index, cid, is_view=(target == BIGQUERY_VIEW) + ), ) - elif target == GCS_BUCKET: + continue + if target is not None: + # Whatever the binding's platform: this emitter is about to write + # the resource, so a policy it would drop is refused here. + # (``fluid validate`` dispatches an aws binding to the AWS checks.) + _gov.refuse_unsupported_target(exposure, index, target) + if target == GCS_BUCKET: # An expose that NAMES the bucket as its port (``format: # gcs_bucket``) owns it. One resolved from the location shape # merely cites the container its files live in, so a declared @@ -388,6 +461,18 @@ def emit( # planner already interpreted the loose `execution.trigger` # surface into structured `run.*` / `scheduler.*` / `ps.*` ops. _emit_from_actions(resources, actions, cid) + # The GCP sovereignty hook, where the data lands: every location an + # emitted resource carries (a region the binding left to a default + # included) and each gcp expose that names no region. A refusal is + # raised before any module exists, for `fluid apply` and `fluid + # generate iac` alike (providers/gcp/util/sovereignty.py). + from ...providers.gcp.util.sovereignty import ( + enforce_gcp_sovereignty, + resource_placements, + ) + + if enforce_sovereignty: + enforce_gcp_sovereignty(contract, resource_placements(resources)) return resources def emit_data( @@ -404,9 +489,11 @@ def emit_data( """ cid = safe_ident(contract.get("id") or contract.get("name") or "product") packaging = resolve_packaging(contract) + # The BigQuery service agent a product key is granted to, whatever the + # packaging: looked up by the same derivation ``_emit_bigquery_key`` uses. + data: Dict[str, Dict[str, Any]] = _service_agent_lookups(contract, cid) if packaging.is_legacy: - return {} - data: Dict[str, Dict[str, Any]] = {} + return data for exposure in contract.get("exposes") or []: binding = exposure.get("binding") or {} # Resolve through the same chokepoint :meth:`emit` uses — the two @@ -560,6 +647,28 @@ def _add(address: str, resource_id: str) -> None: topic_id = f"projects/{project}/topics/{topic}" if project else topic _add(f"google_pubsub_topic.{topic_key}", topic_id) + if dataset and target in (BIGQUERY_TABLE, BIGQUERY_VIEW): + # A Cloud KMS key ring and key cannot be deleted: after a destroy + # they still exist, so a re-apply must adopt them, not 409. + encryption = _gov.encryption_for(binding, _gov.dataset_location(loc)) + if encryption is not None and encryption.product_key: + kms_project = loc.get("project") or project + ident = _kms_ident(cid, dataset) + ring = _gov.product_key_ring(cid, dataset) + # The provider's documented import ids; without a project it + # takes the provider's own. + if kms_project: + ring_id = ( + f"projects/{kms_project}/locations/{encryption.location}" + f"/keyRings/{ring}" + ) + key_id = f"{ring_id}/cryptoKeys/{_gov.PRODUCT_KEY_NAME}" + else: + ring_id = f"{encryption.location}/{ring}" + key_id = f"{ring_id}/{_gov.PRODUCT_KEY_NAME}" + _add(f"google_kms_key_ring.{ident}", ring_id) + _add(f"google_kms_crypto_key.{ident}", key_id) + return blocks def provider_block(self) -> Dict[str, Any]: @@ -567,6 +676,382 @@ def provider_block(self) -> Dict[str, Any]: provider self-configures from the environment.""" return {} + def reconcile_state( + self, module: Dict[str, Any], state: Sequence[Mapping[str, Any]] + ) -> List[Dict[str, Any]]: + """Patch ``module`` for datasets an older forge-cli applied; see + :func:`reconcile_legacy_dataset_access`.""" + return reconcile_legacy_dataset_access(module, state) + + +#: BigQuery's basic dataset roles by the legacy name its access list reports. +_LEGACY_DATASET_ROLES = { + "roles/bigquery.dataOwner": "OWNER", + "roles/bigquery.dataEditor": "WRITER", + "roles/bigquery.dataViewer": "READER", +} + +#: IAM member prefix → the ``access`` entry field BigQuery files it under. +_ACCESS_FIELDS = { + "user": "user_by_email", + "serviceAccount": "user_by_email", + "group": "group_by_email", + "domain": "domain", +} + +#: ``access`` entry fields that name something other than one principal. +_NON_PRINCIPAL_ACCESS_FIELDS = ("special_group", "iam_member", "view", "dataset", "routine") + + +def _access_principal(entry: Mapping[str, Any]) -> Optional[Tuple[str, str, str]]: + """``(role, field, identity)`` for an entry the old emitter could have written.""" + if any(entry.get(field) for field in _NON_PRINCIPAL_ACCESS_FIELDS): + return None + named = [(f, entry.get(f)) for f in ("user_by_email", "group_by_email", "domain")] + named = [(f, v) for f, v in named if v] + if len(named) != 1 or not entry.get("role"): + return None + role = str(entry["role"]) + field, identity = named[0] + return _LEGACY_DATASET_ROLES.get(role, role), field, str(identity).lower() + + +def _module_access_entry(entry: Mapping[str, Any]) -> Dict[str, Any]: + """A state ``access`` entry as module JSON: its non-empty fields and blocks. + + Read from state, so every string is escaped as the renderer escapes contract + text: nothing read back from the cloud can become an interpolation. + """ + + def escaped(value: Any) -> Any: + if isinstance(value, Mapping): + return {k: escaped(v) for k, v in value.items()} + if isinstance(value, list): + return [escaped(v) for v in value] + return _literal(value) + + return {k: escaped(v) for k, v in entry.items() if v not in (None, "", [], {})} + + +def reconcile_legacy_dataset_access( + module: Dict[str, Any], state: Sequence[Mapping[str, Any]] +) -> List[Dict[str, Any]]: + """Revoke, once, the grants a dataset's old authoritative access list held. + + forge-cli 0.16.6 and earlier wrote a dataset's grants as its ``access`` list, which is + authoritative: the provider replaced the dataset's whole ACL with it. Grants are + ``google_bigquery_dataset_iam_member`` resources now, and the module no longer + sets ``access``, which the provider keeps as Computed: an entry the old list held + and no member resource covers (a principal removed in the same change, or a + logical principal ``binding.principals`` now maps to another identity) would stay + on the dataset, unmanaged, and no later plan would show it (measured with + provider 6.50.0; terraform-provider-google issue 8165: removing every ``access`` + block plans nothing). + + For a dataset in ``state`` with a non-empty ``access`` and no member resource yet + (the module the old emitter applied), this sets the module's ``access`` for this + one apply to the state's entries less those stale ones, so the provider revokes + them; the member resources, which depend on the dataset, are created after it. + The next run finds the member resources in state and leaves ``access`` unset + again. Only entries of the old emitter's shape (a role and one user, group or + domain) can be stale; special groups, views and routines are kept as they are. + For that one apply the list is authoritative, as it was on every apply before: + it is what state recorded at the last apply, so an entry added by hand since + then is removed, as the old module removed it. + + Returns one record per dataset patched, with the entries revoked. A dataset whose + every entry is stale cannot be reconciled this way (an empty ``access`` plans + nothing), and is returned with ``"blocked": True`` for the caller to refuse. + """ + resources = module.get("resource") or {} + datasets = resources.get("google_bigquery_dataset") or {} + members = resources.get("google_bigquery_dataset_iam_member") or {} + in_state = { + str(r.get("name")): r + for r in state + if r.get("type") == "google_bigquery_dataset" and r.get("mode", "managed") == "managed" + } + # A dataset any member resource already names has been applied by this emitter. + migrated = { + (r.get("values") or {}).get("dataset_id") + for r in state + if r.get("type") == "google_bigquery_dataset_iam_member" + } + reports: List[Dict[str, Any]] = [] + for name, body in datasets.items(): + if not isinstance(body, dict) or "access" in body or name not in in_state: + continue + values = in_state[name].get("values") or {} + access = [e for e in values.get("access") or [] if isinstance(e, Mapping)] + dataset_id = values.get("dataset_id") + if not access or dataset_id in migrated: + continue + ref = tofu_ref(f"google_bigquery_dataset.{name}.dataset_id") + desired: set[Tuple[str, str, str]] = set() + for member in members.values(): + if not isinstance(member, Mapping) or member.get("dataset_id") != ref: + continue + head, _, identity = str(member.get("member") or "").partition(":") + field = _ACCESS_FIELDS.get(head) + role = _LEGACY_DATASET_ROLES.get(str(member.get("role")), str(member.get("role"))) + if field: + desired.add((role, field, identity.lower())) + stale = [] + for entry in access: + principal = _access_principal(entry) + if principal is not None and principal not in desired: + stale.append(entry) + if not stale: + continue + kept = [_module_access_entry(e) for e in access if e not in stale] + prefix = {"user_by_email": "user", "group_by_email": "group", "domain": "domain"} + revoked = [ + f"{p[0]} {prefix[p[1]]}:{p[2]}" + for p in (_access_principal(e) for e in stale) + if p is not None + ] + report = {"dataset": dataset_id or name, "revoked": revoked, "blocked": not kept} + if kept: + body["access"] = kept + reports.append(report) + return reports + + +@dataclass(frozen=True) +class _BqGovernance: + """What one BigQuery expose asks of retention, encryption and column access. + + Derived by ``gcp_governance`` (shared with ``fluid verify``); empty for every + contract that declares none of them, whose emit is then what it was before. + """ + + retention: Optional[_gov.BqRetention] = None + encryption: Optional[_gov.BqEncryption] = None + tags: Tuple[_gov.TagGroup, ...] = () + + +_NO_GOVERNANCE = _BqGovernance() + +#: Targets whose emitters write the contract's access grants. +_GRANTED_TARGETS = frozenset({"bigquery_table", "bigquery_view", "gcs_bucket", "iceberg"}) + + +def _bigquery_governance( + contract: Mapping[str, Any], + exposure: Mapping[str, Any], + index: int, + cid: str, + *, + is_view: bool, +) -> _BqGovernance: + """The governance one BigQuery expose declares; refuses what cannot be applied.""" + _gov.validate_bigquery_governance(contract, exposure, index, is_view=is_view) + binding = exposure.get("binding") or {} + loc = binding.get("location") or {} + where = f"exposes[{_expose_id(exposure) or index}].binding" + return _BqGovernance( + retention=_gov.retention_for(exposure, index, is_view=is_view), + encryption=_gov.encryption_for(binding, _gov.dataset_location(loc), where=where), + tags=( + () + if is_view + else tuple( + _gov.tag_groups( + contract, + exposure, + cid, + str(loc.get("dataset") or "default"), + _bq_table_name(exposure, loc), + index, + ) + ) + ), + ) + + +def _service_agent_lookups(contract: Mapping[str, Any], cid: str) -> Dict[str, Dict[str, Any]]: + """``data.google_bigquery_default_service_account`` for each product key. + + Mirrors :func:`_emit_bigquery_key` (same targets, same derivation), so every + ``${data...email}`` the key's IAM member names is declared, and nothing else is. + """ + lookups: Dict[str, Any] = {} + for exposure in contract.get("exposes") or []: + binding = exposure.get("binding") or {} + if resolve_gcp_target(binding) not in (BIGQUERY_TABLE, BIGQUERY_VIEW): + continue + loc = binding.get("location") or {} + try: + encryption = _gov.encryption_for(binding, _gov.dataset_location(loc)) + except Exception: # noqa: BLE001 — ``emit`` refuses it with the reason + continue + if encryption is None or not encryption.product_key: + continue + body: Dict[str, Any] = {} + if loc.get("project"): + body["project"] = loc["project"] + lookups.setdefault(_kms_ident(cid, str(loc.get("dataset") or "default")), body) + return {"google_bigquery_default_service_account": lookups} if lookups else {} + + +def _kms_ident(cid: str, dataset: str) -> str: + """The resource-name stem of the product key for ``dataset`` (ring, key, IAM, lookup).""" + return safe_ident(f"{cid}_{dataset}_kms") + + +def _emit_bigquery_key( + resources: Dict[str, Any], + encryption: Optional[_gov.BqEncryption], + cid: str, + dataset: str, + loc: Mapping[str, Any], + labels: Dict[str, str], +) -> Tuple[Optional[Any], List[str]]: + """``(kms_key_name, depends_on)`` for the dataset and table; ``(None, [])`` for none. + + ``product``: a key ring and a crypto key in the dataset's location, with 90-day + rotation, and ``roles/cloudkms.cryptoKeyEncrypterDecrypter`` on the key for the + project's BigQuery service agent (``data.google_bigquery_default_service_account``, + the provider's documented way to name it), which BigQuery encrypts and decrypts + as. The dataset and table wait for that grant: BigQuery checks it can use the key + when the table is created. An existing key is named as given; its owner grants + the service agent. + """ + if encryption is None: + return None, [] + if not encryption.product_key: + return encryption.kms, [] + ident = _kms_ident(cid, dataset) + ring: Dict[str, Any] = { + "name": _gov.product_key_ring(cid, dataset), + "location": encryption.location, + } + if loc.get("project"): + ring["project"] = loc["project"] + resources.setdefault("google_kms_key_ring", {}).setdefault(ident, ring) + resources.setdefault("google_kms_crypto_key", {}).setdefault( + ident, + { + "name": _gov.PRODUCT_KEY_NAME, + "key_ring": tofu_ref(f"google_kms_key_ring.{ident}.id"), + "purpose": "ENCRYPT_DECRYPT", + "rotation_period": _gov.KEY_ROTATION_PERIOD, + "labels": labels, + }, + ) + resources.setdefault("google_kms_crypto_key_iam_member", {}).setdefault( + ident, + { + "crypto_key_id": tofu_ref(f"google_kms_crypto_key.{ident}.id"), + "role": _gov.KMS_ENCRYPTER_ROLE, + "member": TofuExpr( + "serviceAccount:" + + tofu_ref(f"data.google_bigquery_default_service_account.{ident}.email") + ), + }, + ) + return tofu_ref(f"google_kms_crypto_key.{ident}.id"), [ + f"google_kms_crypto_key_iam_member.{ident}" + ] + + +def _emit_policy_tags( + resources: Dict[str, Any], + tags: Sequence[_gov.TagGroup], + cid: str, + dataset: str, + table: str, + loc: Mapping[str, Any], +) -> Dict[str, Any]: + """The taxonomy, one policy tag per group and its readers; ``{column: tag name}``. + + The taxonomy (one per product and dataset, in the dataset's location, since a + tag applies only to tables in its own location) has fine-grained access control + on, so a column carrying one of its tags is readable only by a principal with + ``roles/datacatalog.categoryFineGrainedReader`` on the tag. That role is granted, + one ``google_data_catalog_policy_tag_iam_member`` each, to exactly the readers + ``column_access`` derives; a denied principal gets an access error on the column + (``SELECT * EXCEPT (...)`` still works for it). + """ + if not tags: + return {} + taxonomy = safe_ident(f"{cid}_{dataset}_taxonomy") + body: Dict[str, Any] = { + "display_name": _gov.taxonomy_display_name(cid, dataset), + "description": ( + f"Column-level access for the data product {cid} in dataset {dataset}, " + "written by fluid apply from its column restrictions." + ), + "activated_policy_types": ["FINE_GRAINED_ACCESS_CONTROL"], + "region": _gov.taxonomy_region(_gov.dataset_location(loc)), + } + if loc.get("project"): + body["project"] = loc["project"] + resources.setdefault("google_data_catalog_taxonomy", {}).setdefault(taxonomy, body) + refs: Dict[str, Any] = {} + for group in tags: + resources.setdefault("google_data_catalog_policy_tag", {})[group.key] = { + "taxonomy": tofu_ref(f"google_data_catalog_taxonomy.{taxonomy}.id"), + "display_name": group.display_name, + # The restrictions' own tags and labels ride here: a policy tag has no + # labels of its own, and they must not be dropped. + "description": f"{table}: {group.description}", + } + name = tofu_ref(f"google_data_catalog_policy_tag.{group.key}.name") + for member in group.readers: + resources.setdefault("google_data_catalog_policy_tag_iam_member", {})[ + _iam_key(group.key, member, role=_gov.FINE_GRAINED_READER_ROLE) + ] = { + "policy_tag": name, + "role": _gov.FINE_GRAINED_READER_ROLE, + "member": member, + } + for column in group.columns: + refs[column] = name + return refs + + +def _iam_key(stem: str, member: str, *, role: str) -> str: + """A resource name for one role and member that no other member can share. + + ``safe_ident`` folds every character it cannot keep into ``_``, so + ``group:data.eng@x``, ``group:data-eng@x`` and ``group:data_eng@x`` shared one + name and the last written silently replaced the other two grants. The readable + stem is kept and a hash of the exact role and member made unique, the pattern + Terraform modules use for ``for_each`` keys over IAM lists (an md5 of + member/resource/role; sha256 here). + """ + digest = hashlib.sha256(f"{role}\n{member}".encode("utf-8")).hexdigest()[:10] + return f"{safe_ident(f'{stem}_{member}')}_{digest}" + + +def _emit_dataset_iam( + resources: Dict[str, Any], + grants: Sequence[AccessGrant], + ds_ref: Any, + cid: str, + dataset: str, + loc: Mapping[str, Any], +) -> None: + """One non-authoritative ``google_bigquery_dataset_iam_member`` per role and member. + + This was an authoritative ``access`` list on the dataset: it replaced every + entry the dataset had (its default owners, the pipeline's own, any grant made + elsewhere) with exactly the contract's, and it carried the contract's + principals as written. A member resource adds its binding and leaves the rest, + and its principal is the binding's mapped identity (``iac/principals.py``). The + provider names the caveat: dataset IAM resources rewrite the access list without + authorized-view entries, which forge-cli does not emit. + """ + for role, grant in role_grants(grants, _BQ_TABLE_IAM_ROLES): + member = _gcs_member(grant) + body: Dict[str, Any] = {"dataset_id": ds_ref, "role": role, "member": member} + if loc.get("project"): + body["project"] = loc["project"] + resources.setdefault("google_bigquery_dataset_iam_member", {})[ + _iam_key(f"{cid}_{dataset}_{role}", member, role=role) + ] = body + def _emit_bigquery( resources: Dict[str, Any], @@ -579,23 +1064,24 @@ def _emit_bigquery( is_view: bool, grants: Sequence[AccessGrant], placement: _Placement = _LEGACY_PLACEMENT, + governance: Optional[_BqGovernance] = None, ) -> None: + gov = governance or _NO_GOVERNANCE dataset = loc.get("dataset") or "default" table = _bq_table_name(exposure, loc) ds_name = safe_ident(f"{cid}_{dataset}") tbl_name = safe_ident(f"{cid}_{table}") + kms_key, kms_deps = _emit_bigquery_key(resources, gov.encryption, cid, dataset, loc, labels) if placement.dataset_referenced: # Shared pool dataset: looked up (see ``emit_data``), never created. - # The dataset-level ``access[]`` block is deliberately NOT emitted — - # it is authoritative in the BigQuery API, so a tenant writing it - # would replace the pool's entire ACL and evict its other tenants. - # The same grants move down to table level instead. + # The grants move down to table level, and its default key stays its + # owner's: only the table this product creates carries the key. ds_ref: Any = tofu_ref(f"data.google_bigquery_dataset.{ds_name}.dataset_id") else: dataset_body: Dict[str, Any] = { "dataset_id": dataset, - "location": loc.get("region") or loc.get("location") or "US", + "location": _gov.dataset_location(loc), "labels": labels, } # The binding's project, when it names one, goes on the resource. It was @@ -603,13 +1089,20 @@ def _emit_bigquery( # binding for one project could provision into another. if loc.get("project"): dataset_body["project"] = loc["project"] - # Access grants → the dataset ACL (mirrors the retired native - # `iam.bind_bq_dataset`, which appended BigQuery access entries). - access = _bq_access_entries(grants) - if access: - dataset_body["access"] = access - resources.setdefault("google_bigquery_dataset", {}).setdefault(ds_name, dataset_body) + if kms_key is not None: + dataset_body["default_encryption_configuration"] = {"kms_key_name": kms_key} + dataset_body["depends_on"] = list(kms_deps) + existing = resources.setdefault("google_bigquery_dataset", {}).setdefault( + ds_name, dataset_body + ) + if existing is not dataset_body and kms_key is not None: + # A dataset another expose created first (an unkeyed view, say) still + # takes the key: the default key must not depend on expose order. + # ``refuse_mixed_dataset_encryption`` has made every key here agree. + existing.setdefault("default_encryption_configuration", {"kms_key_name": kms_key}) + existing.setdefault("depends_on", list(kms_deps)) ds_ref = tofu_ref(f"google_bigquery_dataset.{ds_name}.dataset_id") + _emit_dataset_iam(resources, grants, ds_ref, cid, dataset, loc) body: Dict[str, Any] = { "dataset_id": ds_ref, @@ -622,11 +1115,22 @@ def _emit_bigquery( body["project"] = loc["project"] if is_view: body["view"] = {"query": loc.get("query", ""), "use_legacy_sql": False} + elif gov.tags: + body["schema"] = _bq_schema_with_policy_tags( + schema, _emit_policy_tags(resources, gov.tags, cid, dataset, table, loc) + ) elif schema: body["schema"] = _bq_schema(schema) + if kms_key is not None and not is_view: + # ForceNew in the provider: adding a key to a live table replaces it, which + # the data-loss gate refuses without --allow-data-loss. + body["encryption_configuration"] = {"kms_key_name": kms_key} + body["depends_on"] = list(kms_deps) + if gov.retention is not None: + _partition_table(resources, body, tbl_name, gov.retention) resources.setdefault("google_bigquery_table", {})[tbl_name] = body - # Table-level IAM replaces the suppressed dataset ACL under shared mode: + # Table-level IAM replaces the dataset grants under shared mode: # the same principals, the same intent, the narrowest scope that still # grants it (RFC §Security — "GCP grants move to table level"). if placement.dataset_referenced: @@ -643,11 +1147,38 @@ def _emit_bigquery( # Cross-project access needs no new schema fields: declare the consumer # in ``accessPolicy.grants[]`` as - # ``serviceAccount:consumer@other-project.iam.gserviceaccount.com`` and - # the email lands in BQ's ``user_by_email`` field on the dataset's - # ``access[]`` block. (``accessPolicy`` is the schema-valid surface; - # ``metadata.policies`` also emits but fails ``fluid validate`` — see - # ``iac/access.py``.) + # ``serviceAccount:consumer@other-project.iam.gserviceaccount.com`` (or map + # a logical principal to it in ``binding.principals``) and it becomes a + # ``google_bigquery_dataset_iam_member`` on the dataset. (``accessPolicy`` is + # the schema-valid surface; ``metadata.policies`` also emits but fails + # ``fluid validate`` — see ``iac/access.py``.) + + +def _partition_table( + resources: Dict[str, Any], + body: Dict[str, Any], + tbl_name: str, + retention: _gov.BqRetention, +) -> None: + """Daily partitions expiring ``retention`` after their day, and the replace trigger. + + Partition expiration, not ``expiration_time``: a table TTL would delete the + product. See ``gcp_governance.partition_trigger_input`` for why the table's + replacement is triggered by a ``terraform_data`` (the OpenTofu documentation's + ``replace_triggered_by`` pattern for a plain value). + """ + partitioning: Dict[str, Any] = { + "type": _gov.PARTITION_TYPE, + "expiration_ms": retention.expiration_ms, + } + if retention.field: + partitioning["field"] = retention.field + body["time_partitioning"] = partitioning + trigger = safe_ident(f"{tbl_name}_partitioning") + resources.setdefault("terraform_data", {})[trigger] = { + "input": _gov.partition_trigger_input(retention) + } + body["lifecycle"] = {"replace_triggered_by": [f"terraform_data.{trigger}"]} #: ``binding.format`` values marking an Iceberg-table expose. Matches the @@ -1012,10 +1543,17 @@ def _emit_pubsub( ) -> None: topic = loc.get("topic") or f"{cid}-topic" topic_res = safe_ident(f"{cid}_{topic}") - resources.setdefault("google_pubsub_topic", {})[topic_res] = { - "name": topic, - "labels": labels, - } + body: Dict[str, Any] = {"name": topic, "labels": labels} + # The binding's region is where the topic's messages may be stored: + # hashicorp/google ``google_pubsub_topic.message_storage_policy`` + # (``allowed_persistence_regions``). It was dropped, so a topic bound to + # europe-west1 under an EU-only sovereignty block passed validate and + # stored messages wherever Pub/Sub chose. The GCP sovereignty hook reads + # the same field back (``resource_placements``). + region = loc.get("region") or loc.get("location") + if region: + body["message_storage_policy"] = {"allowed_persistence_regions": [str(region)]} + resources.setdefault("google_pubsub_topic", {})[topic_res] = body subscription = loc.get("subscription") if subscription: resources.setdefault("google_pubsub_subscription", {})[ diff --git a/fluid_build/iac/providers/gcp_governance.py b/fluid_build/iac/providers/gcp_governance.py new file mode 100644 index 00000000..e0e0d0e2 --- /dev/null +++ b/fluid_build/iac/providers/gcp_governance.py @@ -0,0 +1,627 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Retention, encryption at rest and column access for a GCP BigQuery binding. + +THE one derivation, read by the OpenTofu emitter (``iac/providers/gcp.py``, which +turns it into ``hashicorp/google`` resources) and by ``fluid verify`` +(``cli/_verify_bigquery_governance.py``, which checks the live dataset and table +against it), so the two cannot disagree. The AWS counterpart is ``aws_storage.py``; +the contract fields are the same, so one contract is governed alike on both clouds. + +**Retention** is ``exposes[].lifecycle {retention, expire: true}`` (fluid-schema +0.7.6). On BigQuery it is partition expiration, never a whole-table TTL +(``expiration_time``), which would delete the product itself. The table is +partitioned by day, on the column ``binding.location.partitionBy`` names (one +DATE, TIMESTAMP or DATETIME column) or, without one, by ingestion time; each +partition is deleted ``retention`` after its day ends (BigQuery counts from the +partition boundary, not from when the rows were written). Without ``partitionBy`` +the partition day is the day the rows landed, so no row is deleted sooner than +``retention`` after it was written: the S3 rule's semantics, and the parity default. +With ``partitionBy`` the day is the column's value, so retention is the age of the +event, not of the load: a backfill of rows whose date is already older than +``retention`` lands in expired partitions and BigQuery deletes it at once, which +the S3 rule (counting from the write) never does. Naming the column is the opt-in +to that; the emitter logs it. dbt-bigquery's ``partition_by`` + +``partition_expiration_days`` is the same pair. BigQuery cannot partition an +existing table, so adding it replaces the table (see :func:`partition_trigger_input`). + +**Encryption** is ``binding.encryption.kms``: ``product`` (the default when the block +is present) is a Cloud KMS key this product creates in the dataset's location +(``google_kms_key_ring`` + ``google_kms_crypto_key``, 90-day rotation), usable by the +project's BigQuery service agent, and made the dataset's default and the table's +key; ``projects/.../cryptoKeys/...`` names an existing key; ``none`` leaves +Google-managed encryption, which BigQuery applies to all data by default. An AWS +alias or ARN is refused here, as the AWS emitter refuses a Cloud KMS name. +dbt-bigquery's ``kms_key_name`` names the same key. + +**Column access** is ``exposes[].policy.authz.columnRestrictions`` through +``iac/column_access.py``: a Data Catalog taxonomy per product and dataset with +fine-grained access control on, a policy tag per set of restricted columns that +share their readers, attached through the table schema's ``policyTags`` (dbt's +``policy_tags``), and ``roles/datacatalog.categoryFineGrainedReader`` for exactly +those readers. The resource shapes follow cloud-foundation-fabric's +``data-catalog-policy-tag`` module. +""" + +from __future__ import annotations + +import hashlib +import logging +import math +import re +from dataclasses import dataclass +from typing import Any, Dict, FrozenSet, List, Mapping, Optional, Tuple + +from ..access import normalize_access_grants +from ..base import UnsupportedBindingError +from ..column_access import authz_readers, column_readers, restrictions_for +from ..naming import safe_ident +from ..principals import GCP, gcp_grants, principal_map, resolve_principal +from ..provider_match import canonical_cloud + +LOG = logging.getLogger(__name__) + +#: ``binding.encryption.kms`` values that are not a key reference. +KMS_PRODUCT = "product" +KMS_NONE = "none" + +#: A Cloud KMS crypto key's resource name. The schema carries the same pattern. +_GCP_KEY_RE = re.compile( + r"projects/(?P[a-z0-9.:-]{1,100})/locations/(?P[a-z0-9-]{1,63})/" + r"keyRings/(?P[A-Za-z0-9_-]{1,63})/cryptoKeys/(?P[A-Za-z0-9_-]{1,63})" +) + +#: A Cloud KMS / BigQuery location id (``europe-west1``, ``us``, ``europe``). +_LOCATION_RE = re.compile(r"[a-z0-9-]{1,63}") + +#: A product key rotates to a new primary version every 90 days, the period CIS +#: GCP 1.10 and terraform-google-modules/kms's examples use. +KEY_ROTATION_PERIOD = "7776000s" + +#: The crypto key's name inside the product's key ring. +PRODUCT_KEY_NAME = "bigquery" + +#: What the BigQuery service agent needs on the key (BigQuery CMEK guide). +KMS_ENCRYPTER_ROLE = "roles/cloudkms.cryptoKeyEncrypterDecrypter" + +#: What a principal needs on a policy tag to read the columns it is attached to. +FINE_GRAINED_READER_ROLE = "roles/datacatalog.categoryFineGrainedReader" + +#: Only DAY partitions are emitted: the one granularity retention in days maps onto. +PARTITION_TYPE = "DAY" + +#: The column types BigQuery time-unit partitioning accepts. +_PARTITION_COLUMN_TYPES = frozenset({"DATE", "TIMESTAMP", "DATETIME"}) + +#: ``accessPolicy`` verbs that read the rows; their principals are the expose's readers. +READ_VERBS = frozenset({"read", "select", "query", "admin", "owner"}) + +_MS_PER_DAY = 86_400_000 + + +@dataclass(frozen=True) +class BqRetention: + """One table's partition expiration.""" + + period: str + days: int + #: The partition column, or ``None`` for ingestion-time partitioning. + field: Optional[str] + + @property + def expiration_ms(self) -> int: + return self.days * _MS_PER_DAY + + +@dataclass(frozen=True) +class BqEncryption: + """One binding's customer-managed key.""" + + #: ``product``, or the existing key's resource name. + kms: str + #: The Cloud KMS location the key lives in (the dataset's). + location: str + + @property + def product_key(self) -> bool: + return self.kms == KMS_PRODUCT + + +@dataclass(frozen=True) +class TagGroup: + """Restricted columns of one table that share their readers: one policy tag.""" + + #: Resource-name stem, unique in the module. + key: str + display_name: str + columns: Tuple[str, ...] + #: IAM members granted the fine-grained reader role on the tag. + readers: Tuple[str, ...] + #: The ``tags`` and ``labels`` of the restrictions naming these columns, for the + #: policy tag's description (a policy tag has no labels of its own). + rule_tags: Tuple[str, ...] = () + rule_labels: Tuple[Tuple[str, str], ...] = () + + @property + def description(self) -> str: + """The policy tag's description: what it restricts, and the rules' tags.""" + text = ( + f"Restricted columns {', '.join(self.columns)}: readable only by the principals " + "granted the fine-grained reader role on this tag." + ) + if self.rule_tags: + text += f" Tags: {', '.join(self.rule_tags)}." + if self.rule_labels: + text += " Labels: " + ", ".join(f"{k}={v}" for k, v in self.rule_labels) + "." + # Data Catalog allows at most 2000 bytes. + return _bounded(text, 1900) + + +def _where(exposure: Mapping[str, Any], index: int) -> str: + return f"exposes[{exposure.get('exposeId') or exposure.get('id') or index}]" + + +def dataset_location(loc: Mapping[str, Any]) -> str: + """The dataset location ``_emit_bigquery`` writes (``US`` when the binding names none).""" + return str(loc.get("region") or loc.get("location") or "US") + + +def kms_location(bq_location: str) -> str: + """The Cloud KMS location a key for a dataset in ``bq_location`` must be in. + + A regional dataset uses the same region; the ``EU`` and ``US`` multi-regions use + the ``europe`` and ``us`` key locations (BigQuery CMEK guide). + """ + lowered = bq_location.strip().lower() + return {"eu": "europe", "us": "us"}.get(lowered, lowered) + + +def taxonomy_region(bq_location: str) -> str: + """The Data Catalog location a taxonomy for a dataset in ``bq_location`` must be in. + + Policy tags apply only to tables in the same location as their taxonomy; the + multi-regions are ``eu`` and ``us``. + """ + return bq_location.strip().lower() + + +def _bounded(text: str, limit: int) -> str: + """``text`` cut to ``limit`` characters, a short hash keeping cut names distinct.""" + if len(text) <= limit: + return text + digest = hashlib.sha256(text.encode("utf-8")).hexdigest()[:8] + return f"{text[: limit - 9]}-{digest}" + + +def product_key_ring(cid: str, dataset: str) -> str: + """The key ring this product creates for ``dataset``. + + One per product and dataset: two environments of one product in the same project + and location (whose overlays name different datasets) do not share a key, as the + AWS emitter keys one alias per bucket. Key ring names allow letters, digits, + ``_`` and ``-``, at most 63. A key ring cannot be deleted on GCP, so the name is + stable and a re-apply adopts it (``GcpIacPlugin.discover_imports``). + """ + return _bounded(re.sub(r"[^A-Za-z0-9_-]", "-", f"fluid-{cid}-{dataset}"), 63) + + +def product_key_name(project: str, location: str, ring: str) -> str: + """The resource name of the product key, as BigQuery reports ``kmsKeyName``.""" + return f"projects/{project}/locations/{location}/keyRings/{ring}/cryptoKeys/{PRODUCT_KEY_NAME}" + + +def _display(text: str, limit: int = 200) -> str: + """A Data Catalog display name: letters, digits, ``_``, ``-`` and spaces.""" + cleaned = re.sub(r"[^A-Za-z0-9_\- ]", "_", text).strip() or "fluid" + return _bounded(cleaned, limit) + + +def taxonomy_display_name(cid: str, dataset: str) -> str: + """The taxonomy's display name, unique per project and location as Data Catalog requires.""" + return _display(f"fluid {cid} {dataset}") + + +def _period_days(period: Any, *, where: str) -> int: + from fluid_build.build_runners._retention import parse_iso_duration + + try: + span = parse_iso_duration(str(period)) + except ValueError: + raise UnsupportedBindingError( + "retention-period", + f"{where}.lifecycle.retention is {period!r}, which is not an ISO-8601 duration.", + ("Write the period as an ISO-8601 duration, e.g. P30D, P12W or P7Y.",), + ) from None + days = math.ceil(span.total_seconds() / 86400) + if days < 1: + raise UnsupportedBindingError( + "retention-period", + f"{where}.lifecycle.retention is {period!r}: BigQuery partition expiration on " + "daily partitions needs a period of at least one day.", + ("Set a period of at least P1D, or drop expire: true.",), + ) + return days + + +def _expire_requested(exposure: Mapping[str, Any]) -> bool: + lifecycle = exposure.get("lifecycle") + return isinstance(lifecycle, Mapping) and lifecycle.get("expire") is True + + +def retention_for( + exposure: Mapping[str, Any], index: int = 0, *, is_view: bool = False +) -> Optional[BqRetention]: + """The partition expiration ``exposure`` asks for, or ``None``. + + ``None`` unless ``lifecycle.expire`` is exactly ``true``. A missing or invalid + period, a view (it stores nothing to expire) and an unusable partition column + are refused. + """ + if not _expire_requested(exposure): + return None + where = _where(exposure, index) + if is_view: + raise UnsupportedBindingError( + "retention-view", + f"{where}.lifecycle.expire is true on a BigQuery view, which stores no rows to " + "expire.", + ("Set expire: true on the expose of the table the view reads, or drop it here.",), + ) + period = (exposure.get("lifecycle") or {}).get("retention") + if not period: + raise UnsupportedBindingError( + "retention-period", + f"{where}.lifecycle.expire is true but lifecycle.retention is not set, so there " + "is no period to expire partitions after.", + ("Set lifecycle.retention (e.g. P30D), or drop expire: true.",), + ) + days = _period_days(period, where=where) + loc = (exposure.get("binding") or {}).get("location") or {} + partition_by = loc.get("partitionBy") if isinstance(loc, Mapping) else None + field: Optional[str] = None + if partition_by: + columns = list(partition_by) if isinstance(partition_by, (list, tuple)) else [] + if len(columns) != 1 or not isinstance(columns[0], str): + raise UnsupportedBindingError( + "retention-partition-column", + f"{where}.binding.location.partitionBy is {partition_by!r}; a BigQuery table " + "is partitioned by at most one column.", + ( + "Name one DATE, TIMESTAMP or DATETIME column, or drop partitionBy to " + "partition by ingestion time.", + ), + ) + from .gcp import _bq_type + + schema = (exposure.get("contract") or {}).get("schema") or [] + declared = {c.get("name"): c for c in schema if isinstance(c, Mapping)} + column = declared.get(columns[0]) + if column is None or _bq_type(column.get("type")) not in _PARTITION_COLUMN_TYPES: + raise UnsupportedBindingError( + "retention-partition-column", + f"{where}.binding.location.partitionBy names {columns[0]!r}, which is not a " + "DATE, TIMESTAMP or DATETIME column of the expose's schema, so BigQuery " + "cannot partition by it.", + ( + "Name a date or timestamp column, or drop partitionBy to partition by " + "ingestion time.", + ), + ) + field = columns[0] + LOG.info( + "bigquery_retention_event_time %s: partitioned by %s, so each row expires %s " + "after the date in %s, not after it was written (a backfill of older rows is " + "deleted at once). Drop binding.location.partitionBy to count from landing, " + "as the S3 rule does.", + where, + field, + period, + field, + ) + return BqRetention(period=str(period), days=days, field=field) + + +def partition_trigger_input(retention: BqRetention) -> Dict[str, str]: + """What forces the table's replacement: the partitioning's shape, not its expiry. + + BigQuery cannot partition an existing table, and the provider plans adding + ``time_partitioning`` (without a column) as an in-place update the API then + refuses. A ``terraform_data`` holding this value is the table's + ``replace_triggered_by`` (the OpenTofu docs' pattern for a plain value): created + for a table that exists, or changed, it plans the table's replacement, which the + data-loss gate refuses without ``--allow-data-loss``. The period is left out, so a + new retention is an in-place change of ``expiration_ms``. + """ + return {"type": PARTITION_TYPE, "field": retention.field or ""} + + +def encryption_for( + binding: Mapping[str, Any], bq_location: str, *, where: str = "binding" +) -> Optional[BqEncryption]: + """The key ``binding`` asks for; ``None`` for no block or ``kms: none``.""" + block = binding.get("encryption") + if block is None: + return None + if not isinstance(block, Mapping): + raise UnsupportedBindingError( + "encryption-kms", + f"{where}.encryption must be a mapping, got {type(block).__name__}.", + ("Write binding.encryption as {kms: product}.",), + ) + kms = block.get("kms", KMS_PRODUCT) + location = kms_location(bq_location) + if kms == KMS_NONE: + return None + if not _LOCATION_RE.fullmatch(location): + # The location becomes a path segment of the key ring's import id, so a + # value that is not a location must not reach it (it could name another + # project's key ring for the apply to adopt). + raise UnsupportedBindingError( + "encryption-kms-location", + f"{where}.location.region is {bq_location!r}, which is not a BigQuery " + "location, so no Cloud KMS location can be derived for its key.", + ("Set binding.location.region to the dataset's location, e.g. europe-west1.",), + ) + if kms == KMS_PRODUCT: + return BqEncryption(kms=KMS_PRODUCT, location=location) + text = kms if isinstance(kms, str) else "" + match = _GCP_KEY_RE.fullmatch(text) + if not match: + aws_shaped = text.startswith("alias/") or text.startswith("arn:") + raise UnsupportedBindingError( + "encryption-kms", + f"{where}.encryption.kms is {kms!r}" + + (", an AWS KMS key, on a gcp binding" if aws_shaped else "") + + "; on GCP it must be 'product', 'none' or a Cloud KMS key name " + "(projects/

/locations//keyRings//cryptoKeys/).", + ( + "Omit kms (or set 'product') to have this product create its own key.", + "Name an existing Cloud KMS key by its resource name.", + ), + ) + if match.group("location") != location: + raise UnsupportedBindingError( + "encryption-kms-location", + f"{where}.encryption.kms is in location {match.group('location')!r}, but the " + f"dataset is in {bq_location!r}; BigQuery uses only a key in the dataset's " + f"location ({location!r}).", + (f"Name a key in {location!r}, or use kms: product.",), + ) + return BqEncryption(kms=text, location=location) + + +def expose_readers( + contract: Mapping[str, Any], + binding: Mapping[str, Any], + *, + where: str, + exposure: Optional[Mapping[str, Any]] = None, + readers_where: str = "policy.authz.readers", +) -> FrozenSet[str]: + """Every IAM member that reads the expose, mapped through ``binding.principals``. + + The ``accessPolicy`` read grants (which the emitter grants on the dataset) and + the expose's own ``policy.authz.readers`` (which it does not: their table access + is managed elsewhere). Both are readers a restriction narrows; leaving the second + out gave a restricted column no reader at all, so a deny for one group locked it + for everyone. + """ + members: set[str] = set() + for grant in gcp_grants(normalize_access_grants(contract), binding, where=where): + if READ_VERBS & set(grant.permissions): + members.add(f"{grant.principal_type}:{grant.principal}") + if exposure is not None: + mapping = principal_map(binding) + for reader in authz_readers(exposure): + members.update(resolve_principal(reader, mapping, platform=GCP, where=readers_where)) + return frozenset(members) + + +def tag_groups( + contract: Mapping[str, Any], + exposure: Mapping[str, Any], + cid: str, + dataset: str, + table: str, + index: int = 0, +) -> List[TagGroup]: + """The policy tags one table's column restrictions become, in schema order.""" + restrictions = restrictions_for(exposure, index) + if not restrictions: + return [] + binding = exposure.get("binding") or {} + where = f"{_where(exposure, index)}.policy.authz.columnRestrictions" + mapping = principal_map(binding) + + def resolve(principal: str) -> Tuple[str, ...]: + return resolve_principal(principal, mapping, platform=GCP, where=where) + + readers = expose_readers( + contract, + binding, + where=f"{_where(exposure, index)} accessPolicy", + exposure=exposure, + readers_where=f"{_where(exposure, index)}.policy.authz.readers", + ) + if not readers: + # The AWS emitter refuses a restriction with no Lake Formation grant to + # narrow; the GCP one would attach a policy tag nobody may read, so a deny + # for one principal would lock the columns for every principal. + raise UnsupportedBindingError( + "column-restriction-no-readers", + f"{where} restricts columns, but the expose has no reader on this gcp binding: " + "no accessPolicy grant with read, select or query and no policy.authz.readers. " + "A policy tag is readable only by the readers it names, so the restricted " + "columns would be unreadable by everyone, not only by the principals the " + "restrictions name.", + ( + "Add the expose's readers: accessPolicy grants with read (forge-cli grants " + "them on the dataset), or policy.authz.readers for readers whose table " + "access is managed elsewhere.", + "Or remove the restriction.", + ), + ) + by_column = column_readers(exposure, restrictions, resolve, readers, where=where) + grouped: Dict[FrozenSet[str], List[str]] = {} + for column, allowed in by_column.items(): + grouped.setdefault(allowed, []).append(column) + groups: List[TagGroup] = [] + for allowed, columns in grouped.items(): + rules = [r for r in restrictions if set(r.columns) & set(columns)] + rule_tags = tuple(dict.fromkeys(t for r in rules for t in r.tags)) + rule_labels = tuple(sorted({pair for r in rules for pair in r.labels})) + groups.append( + TagGroup( + key=safe_ident(f"{cid}_{dataset}_{table}_{columns[0]}"), + display_name=_display(f"{table} {' '.join(columns)}"), + columns=tuple(columns), + readers=tuple(sorted(allowed)), + rule_tags=rule_tags, + rule_labels=rule_labels, + ) + ) + return groups + + +def validate_bigquery_governance( + contract: Mapping[str, Any], exposure: Mapping[str, Any], index: int, *, is_view: bool +) -> None: + """Every derivation above for one BigQuery expose; raises what the emitter would.""" + binding = exposure.get("binding") or {} + loc = binding.get("location") or {} + where = _where(exposure, index) + retention_for(exposure, index, is_view=is_view) + encryption_for(binding, dataset_location(loc), where=f"{where}.binding") + gcp_grants(normalize_access_grants(contract), binding, where=f"{where} accessPolicy") + if is_view and restrictions_for(exposure, index): + raise UnsupportedBindingError( + "column-restriction-view", + f"{where}.policy.authz.columnRestrictions is set on a BigQuery view; policy tags " + "attach to table columns, so restrict the columns of the table the view reads.", + (), + ) + tag_groups(contract, exposure, "c", "d", "t", index) + + +def gcp_owned(binding: Mapping[str, Any]) -> bool: + """Is ``binding`` one the GCP emitter owns: platform gcp, or no cloud named at all? + + ``resolve_gcp_target`` resolves an explicit format (``iceberg``, say) whatever the + platform, so an AWS Iceberg binding resolves to GCP Iceberg storage. Governance + is dispatched on the platform first: an ``aws`` binding is the AWS emitter's, + and its policies are checked against what that emitter writes. + """ + if not isinstance(binding, Mapping): + return False + cloud = canonical_cloud(binding.get("platform")) or canonical_cloud(binding.get("provider")) + return cloud in ("", GCP) + + +def refuse_mixed_dataset_encryption(contract: Mapping[str, Any]) -> None: + """Refuse BigQuery tables of one dataset that declare different keys. + + The dataset's default key is the key of the tables in it: BigQuery gives a table + created without one the dataset's default. A table declared unkeyed in a keyed + dataset therefore gets the key, the provider then plans removing it, and + ``encryption_configuration`` is ForceNew, so every later plan replaces the table + (terraform-provider-google issue 26193, whose workaround is the same key on the + table). With the unkeyed table first, the dataset got no default at all and + ``fluid verify`` failed it on every run. One dataset, one key. A view stores no + rows and carries no key, so an unkeyed view is left out; a keyed one sets the + dataset's default and must agree too. + """ + from .gcp import BIGQUERY_TABLE, BIGQUERY_VIEW, resolve_gcp_target + + seen: Dict[Tuple[str, str], Tuple[str, str]] = {} + for index, exposure in enumerate(contract.get("exposes") or []): + if not isinstance(exposure, Mapping): + continue + binding = exposure.get("binding") or {} + target = resolve_gcp_target(binding) if gcp_owned(binding) else None + if target not in (BIGQUERY_TABLE, BIGQUERY_VIEW): + continue + loc = binding.get("location") or {} + where = _where(exposure, index) + encryption = encryption_for(binding, dataset_location(loc), where=f"{where}.binding") + key = encryption.kms if encryption is not None else KMS_NONE + if target == BIGQUERY_VIEW and encryption is None: + continue + dataset = (str(loc.get("project") or ""), str(loc.get("dataset") or "default")) + first = seen.setdefault(dataset, (key, where)) + if first[0] != key: + raise UnsupportedBindingError( + "encryption-kms-mixed-dataset", + f"{where} and {first[1]} are BigQuery tables in dataset {dataset[1]!r} with " + f"different encryption (kms: {key!r} and {first[0]!r}). The dataset's " + "default key is the key of every table in it: BigQuery gives an unkeyed " + "table the default, and the provider would then replace that table on " + "every apply.", + ( + "Declare the same binding.encryption on every table of the dataset.", + "Or put the tables with a different key in a dataset of their own.", + ), + ) + + +def refuse_unsupported_target(exposure: Mapping[str, Any], index: int, target: str) -> None: + """A GCP expose that is not a BigQuery table must not carry a policy it would drop. + + The GCS, Pub/Sub and Iceberg-storage emitters write no lifecycle rule, key or column + control, so a contract asking for one there is refused rather than applied without it. + """ + where = _where(exposure, index) + binding = exposure.get("binding") or {} + asked = [] + if _expire_requested(exposure): + asked.append("lifecycle.expire") + block = binding.get("encryption") + if isinstance(block, Mapping) and block.get("kms", KMS_PRODUCT) != KMS_NONE: + asked.append("binding.encryption") + if restrictions_for(exposure, index): + asked.append("policy.authz.columnRestrictions") + if asked: + raise UnsupportedBindingError( + "gcp-governance-unsupported-target", + f"{where} is a GCP {target} binding and declares {', '.join(asked)}, which the " + "GCP emitter applies only to BigQuery tables. It would not be enforced.", + ("Move the policy to a BigQuery table expose, or remove it from this one.",), + ) + + +__all__ = [ + "BqEncryption", + "BqRetention", + "FINE_GRAINED_READER_ROLE", + "KEY_ROTATION_PERIOD", + "KMS_ENCRYPTER_ROLE", + "PARTITION_TYPE", + "PRODUCT_KEY_NAME", + "TagGroup", + "dataset_location", + "encryption_for", + "expose_readers", + "gcp_owned", + "kms_location", + "partition_trigger_input", + "product_key_name", + "product_key_ring", + "refuse_mixed_dataset_encryption", + "refuse_unsupported_target", + "retention_for", + "tag_groups", + "taxonomy_display_name", + "taxonomy_region", + "validate_bigquery_governance", +] diff --git a/fluid_build/iac/runner.py b/fluid_build/iac/runner.py index ab7da5a9..2cc7b52c 100644 --- a/fluid_build/iac/runner.py +++ b/fluid_build/iac/runner.py @@ -31,7 +31,7 @@ import shutil import subprocess from dataclasses import dataclass, field -from typing import Any, Dict, List, Mapping, Optional +from typing import Any, Dict, List, Mapping, Optional, Tuple # Per-command wall-clock cap. A hung ``tofu`` (e.g., an unauthenticated # interactive auth prompt that ``-input=false`` did not catch, or a cloud @@ -199,14 +199,49 @@ def _run( ) -def tofu_init(workdir: str, *, backend: bool = True, env: Optional[Mapping[str, str]] = None): - """``tofu init`` — install providers (and initialise the backend).""" +def tofu_init( + workdir: str, + *, + backend: bool = True, + env: Optional[Mapping[str, str]] = None, + reconfigure: bool = False, + force_copy: bool = False, + plugin_dir: Optional[str] = None, +): + """``tofu init`` — install providers (and initialise the backend). + + ``reconfigure`` passes ``-reconfigure``: take the module's backend as + given and ignore the one ``.terraform/`` recorded, moving no state. + ``force_copy`` passes ``-force-copy``, which implies ``-migrate-state``: + copy the recorded backend's state into the module's backend without a + prompt, overwriting whatever the destination holds (OpenTofu + ``backendMigrateState_s_s``), so a caller checks the destination first. + ``plugin_dir`` passes ``-plugin-dir``: providers come only from that + directory, "as if it had been configured as a ``filesystem_mirror``" + (opentofu.org/docs/cli/commands/init), so nothing is downloaded. + """ args = ["init", "-input=false", "-no-color"] if not backend: args.append("-backend=false") + if reconfigure: + args.append("-reconfigure") + if force_copy: + args.append("-force-copy") + if plugin_dir: + args.append(f"-plugin-dir={plugin_dir}") return _run(args, workdir=workdir, env=env, command="init") +def tofu_state_pull(workdir: str, *, env: Optional[Mapping[str, str]] = None) -> TofuResult: + """``tofu state pull`` — the backend's current state document, as JSON text. + + A key that holds no state prints a document with an empty ``lineage`` + and serial 0 (measured on OpenTofu 1.12 against an S3 backend), not an + error; see :mod:`fluid_build.iac.state_migration` for how that is read. + """ + return _run(["state", "pull"], workdir=workdir, env=env, command="state-pull") + + def tofu_validate(workdir: str, *, env: Optional[Mapping[str, str]] = None) -> TofuResult: """``tofu validate`` — check config syntax + provider-schema correctness.""" return _run(["validate", "-no-color"], workdir=workdir, env=env, command="validate") @@ -374,6 +409,26 @@ def tofu_import( ) +def planned_removals(result: TofuResult) -> List[Tuple[str, str]]: + """``(address, resource type)`` of every resource a plan deletes or replaces. + + Read from the ``planned_change`` events of ``tofu plan -json`` (OpenTofu's + machine-readable UI: ``change.action`` is ``delete`` or ``replace`` for the two + that remove an object). The ``change_summary`` event's ``remove`` counts the + same resources, so a caller can tell which of them hold data. + """ + out: List[Tuple[str, str]] = [] + for event in result.events: + if event.get("type") != "planned_change": + continue + change = event.get("change") or {} + if change.get("action") not in ("delete", "replace"): + continue + resource = change.get("resource") or {} + out.append((str(resource.get("addr") or ""), str(resource.get("resource_type") or ""))) + return out + + def change_summary(result: TofuResult) -> Dict[str, int]: """Extract the ``{add, change, remove}`` counts from a plan/apply result.""" for event in result.events: diff --git a/fluid_build/iac/state_migration.py b/fluid_build/iac/state_migration.py new file mode 100644 index 00000000..6ebed551 --- /dev/null +++ b/fluid_build/iac/state_migration.py @@ -0,0 +1,500 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Move a contract's state to its per-provider key, with OpenTofu's own migration. + +The default remote state key gained the provider +(``fluid///terraform.tfstate``, see :mod:`.backend`), so a +contract applied to aws and to gcp keeps two states instead of one that each +cloud's plan would read as the other's orphans. State a previous release +wrote at ``fluid//terraform.tfstate`` has to follow, or the first apply +after the upgrade plans every resource as new. + +**The mechanism is OpenTofu's.** Nothing here reads, edits or writes a state +document's content. The copy is ``tofu init -force-copy`` (which implies +``-migrate-state``) in a scratch directory whose ``.terraform/`` recorded the +old backend and whose module names the new one: the backend-reconfiguration +path ``terraform init -migrate-state`` has always had, which locks both +states where the backend supports locking and leaves the source untouched +(OpenTofu ``internal/command/meta_backend_migrate.go``, +``backendMigrateState_s_s``). The copy is checked afterwards by reading it +back: the same resources, since OpenTofu gives a copy into an empty +destination a fresh lineage. Terragrunt's ``backend migrate`` wraps the same +step for a renamed unit; this is that idea without the wrapper. + +**What the scratch ``tofu init`` installs.** OpenTofu's init installs every +provider the *state* names, not only the module's: a ``{"terraform": {}}`` +module beside a state naming ``hashicorp/null`` installed the latest +``hashicorp/null`` (measured, tofu 1.12), and for an aws state that is the +latest ``hashicorp/aws``, not the pinned ``~> 5.0``. The probe runs on every +apply whose new key is still empty (every ``--dry-run`` while a move is +pending, every gcp run while the old key holds the aws state), so it installs +nothing: ``-plugin-dir`` names an empty directory, the init stops at its +provider step after its backend step recorded the old backend, and the state +is pulled from exactly that recorded backend (checked, never assumed). +Attribution needs the document, not the providers. Should a future OpenTofu +not record the backend first, the probe falls back to a plain init, which +may install, rather than fail. The apply's own ``.terraform/providers`` is +never the plugin directory: with a plugin cache configured its entries link +into the cache, and an init reading them as a mirror broke the workdir +(measured: "no package for hashicorp/aws 5.100.0 cached"). The copy itself, +once per contract, installs what the old state names at the plugin's own +pins (``required_providers`` of the IaC plugin), never the latest. + +**Never lose it, never guess.** ``-force-copy`` overwrites a destination that +holds state (the same OpenTofu function skips its confirmation), so the copy +runs only when the new key holds none, checked once before deciding and again +just before copying. The old object is left where it was. Which provider a +state belongs to is read from its resources' own ``provider`` addresses +(``provider["registry.opentofu.org/hashicorp/aws"]``) against the provider +sources each IaC plugin requires: + +* every resource under this provider's plugin → migrate it; +* every resource under exactly one other plugin (the gcp apply finding the + aws state at the old key) → leave it, it is not this apply's; +* anything else (both clouds in one state, a provider no plugin emits, only + shared utility providers) → refuse with a typed error naming both keys, so + an operator decides. Moving the wrong state is the one mistake that cannot + be undone by the next apply. + +OpenTofu's built-in provider (``terraform.io/builtin/terraform``, which +``terraform_data`` belongs to) is left out of the attribution: it names no +cloud, so a state is attributed by its other resources. + +A read-only caller (``fluid diff``, ``fluid verify --state-drift``) asks with +``migrate=False`` and reads the old key while the move is pending, so a drift +gate that runs before the first upgraded apply still sees the real state. +""" + +from __future__ import annotations + +import json +import logging +import re +import tempfile +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Dict, FrozenSet, Iterable, List, Mapping, Optional, Set, Tuple + +from . import runner +from .backend import backend_location + +#: The new key already holds state: it is used and nothing moves. +CURRENT = "current" +#: Neither key holds a state with resources in it. +NOTHING = "nothing" +#: The old key holds another provider's state, which is left alone. +OTHER_PROVIDER = "other_provider" +#: The old key holds this provider's state and the caller asked not to move it. +PENDING = "pending" +#: The old key's state was copied to the new key. +MIGRATED = "migrated" + +#: Provider sources a previous release wrote under a name the pins no longer +#: use: the Snowflake provider moved from ``Snowflake-Labs`` to ``snowflakedb`` +#: at v2 (see ``versions.PROVIDER_PINS``). +_SOURCE_ALIASES: Dict[str, Tuple[str, ...]] = { + "snowflake": ("snowflake-labs/snowflake",), +} + +#: ``provider["registry.opentofu.org/hashicorp/aws"]`` (optionally ``.alias``). +_PROVIDER_ADDR_RE = re.compile(r'provider\["([^"]+)"\]') + +_LOG = logging.getLogger(__name__) + + +class StateMigrationError(RuntimeError): + """The state could not be moved, or whose it is could not be told.""" + + def __init__(self, code: str, message: str) -> None: + super().__init__(message) + self.code = code + + +@dataclass(frozen=True) +class StateDoc: + """What ``tofu state pull`` said is at one key.""" + + lineage: str + serial: int + resources: Tuple[Mapping[str, Any], ...] + + @property + def exists(self) -> bool: + """A key with no state pulls as an empty lineage (measured, tofu 1.12).""" + return bool(self.lineage) + + +@dataclass(frozen=True) +class StateReconciliation: + """What :func:`reconcile_state_key` found and did.""" + + outcome: str + #: Where the state was before the provider joined the key. + legacy: Mapping[str, Any] + #: Where it is now (the key the apply uses). + current: Mapping[str, Any] + resources: int = 0 + detail: str = "" + + @property + def read_from(self) -> Mapping[str, Any]: + """The backend a read-only caller should read: the old one while pending.""" + return self.legacy if self.outcome == PENDING else self.current + + def summary(self) -> str: + """One line for the apply's output.""" + old = backend_location(self.legacy) + new = backend_location(self.current) + if self.outcome == MIGRATED: + return ( + f"moved {self.resources} resource(s) from {old} to {new} with " + "`tofu init -migrate-state`; the old object is left in place" + ) + if self.outcome == PENDING: + return f"read from {old}: `fluid apply` moves it to {new}" + if self.outcome == OTHER_PROVIDER: + return f"{old} holds {self.detail}, not this provider's; left in place" + return "" + + +def plugin_sources(provider: str) -> FrozenSet[str]: + """The provider sources (``namespace/type``, lower case) ``provider`` emits.""" + from .registry import get_iac_plugin + + plugin = get_iac_plugin(provider) + required = getattr(plugin, "required_providers", None) or {} + sources = {str(spec.get("source", "")).lower() for spec in required.values()} + sources.update(_SOURCE_ALIASES.get(provider, ())) + return frozenset(s for s in sources if s) + + +def state_sources(resources: Iterable[Mapping[str, Any]]) -> FrozenSet[str]: + """``namespace/type`` of every provider the state's resources name. + + OpenTofu's built-in provider (``terraform.io/builtin/terraform``: the + ``terraform_data`` resource, the ``terraform_remote_state`` data source) + is not one: it ships inside the binary, any module can use it, and it + says nothing about which cloud a state is. The GCP plugin writes a + ``terraform_data`` beside a partitioned table (the trigger that replaces + it when its partitioning changes), and counted as a provider no plugin + emits it made every such gcp state "ambiguous", refusing the move and + with it every apply, dry-run and diff on the product. + """ + found = set() + for resource in resources: + match = _PROVIDER_ADDR_RE.search(str(resource.get("provider") or "")) + if not match: + # A resource with no readable provider address is itself a reason + # not to decide: keep it visible to the classification. + found.add("") + continue + parts = match.group(1).lower().split("/") + if _is_builtin(parts): + continue + found.add("/".join(parts[-2:])) + return frozenset(found) + + +def _is_builtin(parts: List[str]) -> bool: + """``terraform.io/builtin/``: a provider compiled into OpenTofu itself.""" + return len(parts) >= 3 and parts[-3] == "terraform.io" and parts[-2] == "builtin" + + +def classify(resources: Iterable[Mapping[str, Any]], provider: str) -> Tuple[str, str]: + """``(verdict, detail)``: ``"mine"``, ``"other"`` or ``"ambiguous"``. + + ``detail`` names the owner for ``"other"`` and the reason for + ``"ambiguous"``. See the module docstring for the rule. + """ + resources = list(resources) + sources = state_sources(resources) + if resources and not sources: + return "ambiguous", "only OpenTofu built-in resources, which name no cloud" + by_plugin = _sources_by_plugin(provider) + known = frozenset().union(*by_plugin.values()) + unknown = sources - known + if unknown: + return "ambiguous", "providers no forge-cli IaC plugin emits: " + ", ".join(sorted(unknown)) + owners = _owners(sources, by_plugin) + if owners == {provider} and sources <= by_plugin[provider]: + return "mine", provider + if len(owners) == 1 and provider not in owners: + (owner,) = owners + if sources <= by_plugin[owner]: + return "other", f"the {owner} provider's state ({', '.join(sorted(sources))})" + if len(owners) > 1: + return "ambiguous", "resources of several clouds (" + ", ".join(sorted(owners)) + ")" + return "ambiguous", "only providers no single cloud owns (" + ", ".join(sorted(sources)) + ")" + + +def other_clouds(resources: Iterable[Mapping[str, Any]], provider: str) -> FrozenSet[str]: + """The IaC plugins other than ``provider`` whose own resources the state holds. + + A resource counts for a plugin when its provider source is one no other + plugin emits (``hashicorp/google`` is gcp's; ``hashicorp/null`` is no + one's), the rule :func:`classify` attributes a state by. + """ + by_plugin = _sources_by_plugin(provider) + return frozenset(_owners(state_sources(resources), by_plugin) - {provider}) + + +def _sources_by_plugin(provider: str) -> Dict[str, FrozenSet[str]]: + from .registry import IAC_PLUGINS + + by_plugin = {name: plugin_sources(name) for name in IAC_PLUGINS} + by_plugin.setdefault(provider, plugin_sources(provider)) + return by_plugin + + +def _owners(sources: FrozenSet[str], by_plugin: Mapping[str, FrozenSet[str]]) -> Set[str]: + """Plugins owning a source in ``sources`` that no other plugin emits.""" + owners: Set[str] = set() + for name, mine in by_plugin.items(): + exclusive = mine - frozenset().union(*(s for n, s in by_plugin.items() if n != name)) + if sources & exclusive: + owners.add(name) + return owners + + +def reconcile_state_key( + *, + workdir: Path, + current: Mapping[str, Any], + legacy: Mapping[str, Any], + provider: str, + env: Mapping[str, str], + migrate: bool, + logger: Optional[logging.Logger] = None, +) -> StateReconciliation: + """Make sure ``current`` holds this provider's state, moving it from ``legacy``. + + ``workdir`` is the apply's own workdir, already ``tofu init``-ed on + ``current``: the new key is read there, so an apply whose state is + already at the new key pays one ``tofu state pull`` and nothing else. + Only a new key with no state brings the scratch directory (and its + ``tofu init -plugin-dir``, which installs nothing) in; only a move + installs, at the plugin's pins. + ``current`` and ``legacy`` are ``terraform.backend`` blocks + (:func:`.backend.parse_backend`). Returns what was found; raises + :class:`StateMigrationError` when the old state cannot be attributed or + the copy fails or does not verify. ``migrate=False`` reports + :data:`PENDING` instead of copying. + """ + log = logger or _LOG + current_dir = Path(workdir) + now = _pull(current_dir, env) + if now.exists: + return StateReconciliation(CURRENT, legacy, current) + with tempfile.TemporaryDirectory(prefix="fluid-state-") as tmp: + old = _probe(Path(tmp), legacy, env) + if not old.exists or not old.resources: + return StateReconciliation(NOTHING, legacy, current) + verdict, detail = classify(old.resources, provider) + if verdict == "other": + return StateReconciliation(OTHER_PROVIDER, legacy, current, len(old.resources), detail) + if verdict != "mine": + raise StateMigrationError( + "state_migration_ambiguous", + f"{backend_location(legacy)} holds {detail}, and {backend_location(current)} " + f"holds no state, so it cannot be told whether it is the {provider} apply's " + "and nothing was moved. Move it yourself (`tofu init -migrate-state` from a " + "directory configured with the old key), or name the key this apply should " + "use explicitly in --state-backend / FLUID_STATE_BACKEND", + ) + if not migrate: + return StateReconciliation(PENDING, legacy, current, len(old.resources)) + + # Looked at again right before the copy: -force-copy would overwrite + # a state another job wrote since the first probe. + again = _pull(current_dir, env) + if again.exists: + if again.resources == old.resources: + return StateReconciliation(CURRENT, legacy, current) + raise StateMigrationError( + "state_migration_raced", + f"{backend_location(current)} received a different state while this apply " + f"was about to move {backend_location(legacy)} there; nothing was moved", + ) + # The copy starts from a directory initialised on the old key, whose + # module pins the providers the old state names to the plugin's own + # versions (see the module docstring). + move_dir = Path(tmp) / "move" + pins = plugin_pins(provider, state_sources(old.resources)) + _write_backend(move_dir, legacy, pins) + first = runner.tofu_init(str(move_dir), env=env) + if not first.ok: + raise StateMigrationError( + "state_migration_failed", + f"could not initialise on {backend_location(legacy)} to move it: " + + _tail(first.stderr or first.stdout), + ) + _write_backend(move_dir, current, pins) + init = runner.tofu_init(str(move_dir), env=env, force_copy=True) + if not init.ok: + raise StateMigrationError( + "state_migration_failed", + "`tofu init -force-copy` could not copy the state: " + + _tail(init.stderr or init.stdout), + ) + # Verified on content, not lineage: OpenTofu 1.12 writes the copy into + # an empty destination under a fresh lineage and serial 1 (measured + # against an S3 backend), so only the resources can be compared. + moved = _pull(current_dir, env) + if moved.resources != old.resources: + raise StateMigrationError( + "state_migration_unverified", + f"after the copy {backend_location(current)} holds " + f"{len(moved.resources)} resource(s) that are not the " + f"{len(old.resources)} copied from {backend_location(legacy)}; the old " + "object is untouched", + ) + result = StateReconciliation(MIGRATED, legacy, current, len(old.resources)) + log.warning("state_migrated: %s", result.summary()) + return result + + +def _write_backend( + workdir: Path, + backend: Mapping[str, Any], + required_providers: Optional[Mapping[str, Any]] = None, +) -> None: + workdir.mkdir(parents=True, exist_ok=True) + terraform: Dict[str, Any] = {"backend": dict(backend)} + if required_providers: + terraform["required_providers"] = dict(required_providers) + doc = {"terraform": terraform} + (workdir / "main.tf.json").write_text(json.dumps(doc, indent=2), encoding="utf-8") + + +def plugin_pins(provider: str, sources: Iterable[str]) -> Dict[str, Dict[str, str]]: + """``provider``'s ``required_providers`` entries for the given ``namespace/type`` sources.""" + from .registry import get_iac_plugin + + wanted = {str(s).lower() for s in sources} + required = getattr(get_iac_plugin(provider), "required_providers", None) or {} + return { + name: dict(spec) + for name, spec in required.items() + if str(spec.get("source", "")).lower() in wanted + } + + +def _probe(tmp: Path, backend: Mapping[str, Any], env: Mapping[str, str]) -> StateDoc: + """``backend``'s state, read with nothing installed (see the module docstring). + + ``tofu init -plugin-dir`` with an empty directory: a state that names a + provider stops the init after the backend step, and the state is read + only when the init is shown to have recorded ``backend``. Otherwise a + plain init in a fresh directory is tried, as before; its failure is the + error. + """ + empty = tmp / "no-providers" + empty.mkdir(parents=True, exist_ok=True) + probe_dir = tmp / "legacy" + _write_backend(probe_dir, backend) + init = runner.tofu_init(str(probe_dir), env=env, plugin_dir=str(empty)) + if init.ok or records_backend(probe_dir, backend): + return _pull(probe_dir, env) + plain_dir = tmp / "legacy-plain" + _write_backend(plain_dir, backend) + init = runner.tofu_init(str(plain_dir), env=env) + if not init.ok: + raise StateMigrationError( + "state_migration_probe_failed", + f"could not read {backend_location(backend)}: " + _tail(init.stderr or init.stdout), + ) + return _pull(plain_dir, env) + + +def read_state(workdir: Path, env: Mapping[str, str]) -> StateDoc: + """The state ``workdir``'s initialised backend holds (``tofu state pull``). + + Raises :class:`StateMigrationError` (``state_migration_probe_failed``) + when it cannot be read; the error carries stderr only, never the + document. + """ + return _pull(Path(workdir), env) + + +def _pull(workdir: Path, env: Mapping[str, str]) -> StateDoc: + result = runner.tofu_state_pull(str(workdir), env=env) + if not result.ok: + # stderr only: stdout of `state pull` is the state document, whose + # attributes can hold secrets, and must never reach an error message. + raise StateMigrationError( + "state_migration_probe_failed", + "`tofu state pull` failed: " + (_tail(result.stderr) or f"exit {result.returncode}"), + ) + return parse_state(result.stdout) + + +def parse_state(text: str) -> StateDoc: + """A ``tofu state pull`` document. Empty output is no state.""" + if not (text or "").strip(): + return StateDoc("", 0, ()) + try: + doc = json.loads(text) + except json.JSONDecodeError as exc: + raise StateMigrationError( + "state_migration_probe_failed", f"`tofu state pull` printed no JSON ({exc.msg})" + ) from None + if not isinstance(doc, dict): + raise StateMigrationError( + "state_migration_probe_failed", "`tofu state pull` printed JSON that is not a state" + ) + resources = doc.get("resources") or [] + return StateDoc( + lineage=str(doc.get("lineage") or ""), + serial=int(doc.get("serial") or 0), + resources=tuple(r for r in resources if isinstance(r, dict)), + ) + + +def recorded_backend(workdir: Path) -> Optional[Dict[str, Any]]: + """The backend ``tofu init`` last recorded in ``workdir``, as ``{type: config}``. + + ``.terraform/terraform.tfstate`` keeps it as ``{"backend": {"type": ..., + "config": {...}}}``. ``None`` when there is none or it cannot be read. + """ + path = workdir / ".terraform" / "terraform.tfstate" + try: + doc = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + backend = doc.get("backend") if isinstance(doc, dict) else None + if not isinstance(backend, dict) or not backend.get("type"): + return None + config = backend.get("config") if isinstance(backend.get("config"), dict) else {} + return {str(backend["type"]): config} + + +def records_backend(workdir: Path, backend: Mapping[str, Any]) -> bool: + """True when ``workdir``'s recorded backend is ``backend`` (bucket and key/prefix). + + Only the fields forge-cli writes are compared: OpenTofu records every + backend attribute, most of them null. + """ + recorded = recorded_backend(workdir) + if not recorded or set(recorded) != set(backend): + return False + (kind,) = tuple(backend) + want = backend[kind] or {} + have = recorded[kind] or {} + return all(have.get(k) == v for k, v in want.items()) + + +def _tail(text: str, limit: int = 600) -> str: + text = (text or "").strip() + return text[-limit:] if len(text) > limit else text diff --git a/fluid_build/loader.py b/fluid_build/loader.py index 9e144572..2417e3fc 100644 --- a/fluid_build/loader.py +++ b/fluid_build/loader.py @@ -19,7 +19,7 @@ import logging import threading from pathlib import Path -from typing import Any, Dict, List, Optional, Set, Tuple, Union +from typing import Any, Dict, List, Mapping, Optional, Set, Tuple, Union try: import yaml # type: ignore @@ -481,18 +481,147 @@ def available_overlay_envs(contract_path: str | Path) -> List[str]: _NOTED_MISSING_OVERLAYS_LOCK = threading.Lock() +#: The ``fluid.workspace.yaml`` key that names, per product, the environments +#: it is deployed to (``{product: [env, ...]}``, the product being the +#: contract's directory name or its id). Written by workspaces whose own gate +#: checks one overlay per environment (fluid-demo-env's ``make targets``); +#: read here so ``--env`` for a declared environment cannot fall back to the +#: base contract. +EXPECTED_ENVIRONMENTS_KEY = "expected-environments" + + +def _base_platforms(contract: Mapping[str, Any]) -> List[str]: + out: List[str] = [] + for expose in contract.get("exposes") or []: + binding = expose.get("binding") if isinstance(expose, dict) else None + platform = binding.get("platform") if isinstance(binding, dict) else None + if platform and str(platform) not in out: + out.append(str(platform)) + return out + + +#: How :func:`declared_environments` names the contract's own block. +CONTRACT_ENVIRONMENTS_BLOCK = "the contract's environments block" + + +def declared_environments( + contract_path: str | Path, contract: Mapping[str, Any] +) -> List[Tuple[str, List[str]]]: + """``[(source, envs)]``: where this product's environments are declared. + + Two declarations are read: the contract's own ``environments`` block + (schema ``$defs.environmentConfig``, one key per environment), and the + workspace's ``expected-environments`` entry for this product, looked up + by the contract's directory name, then by its id. Unreadable or absent + declarations are simply not listed. + """ + found: List[Tuple[str, List[str]]] = [] + environments = contract.get("environments") + if isinstance(environments, dict) and environments: + found.append((CONTRACT_ENVIRONMENTS_BLOCK, [str(k) for k in environments])) + try: + from .util.workspace_root import WORKSPACE_CONFIG_FILENAME, find_workspace_root + + base = Path(contract_path).resolve() + root = find_workspace_root(base.parent) + if root is None: + return found + workspace = _parse_file(root / WORKSPACE_CONFIG_FILENAME) + except Exception: # noqa: BLE001 - an unreadable workspace declares nothing + return found + if not isinstance(workspace, dict): + return found + block = workspace.get(EXPECTED_ENVIRONMENTS_KEY) + if not isinstance(block, dict): + return found + for key in (base.parent.name, contract.get("id")): + envs = block.get(key) if isinstance(key, str) else None + if isinstance(envs, list): + found.append( + ( + f"{WORKSPACE_CONFIG_FILENAME} {EXPECTED_ENVIRONMENTS_KEY} ({key})", + [str(e) for e in envs], + ) + ) + break + return found + + +def refuse_declared_missing_overlay( + contract_path: str | Path, env: str, contract: Mapping[str, Any] +) -> None: + """Refuse ``--env `` with no overlay when the workspace expects ``env``. + + Without an overlay the base contract is used unchanged, which for a + declared environment means deploying it as if it were that environment + (measured: silver ``--env gcp`` validated and planned the local base, + rc=0). The declaration that refuses is the workspace's + ``expected-environments``: a statement that this product has one overlay + per environment. The contract's own ``environments`` block does not + refuse. It is schema-valid, forge-cli applies nothing from it, and + refusing on it broke contracts that validated before, so + :func:`note_missing_overlay` names it in its warning instead. The + base-by-convention ``dev`` and an env the base contract is already bound + to (``local`` for a local base) are the base, and pass. + """ + if env == BASE_ENV_BY_CONVENTION: + return + platforms = _base_platforms(contract) + if env in platforms: + return + for source, envs in declared_environments(contract_path, contract): + if source == CONTRACT_ENVIRONMENTS_BLOCK: + continue + if env in envs: + from ._contract_loader import CLIError + + refusal = CLIError( + 1, + "overlay_declared_but_missing", + { + "env": env, + "contract": str(Path(contract_path)), + "declared_by": source, + "base_platforms": platforms, + "available_envs": available_overlay_envs(contract_path), + "error": ( + f"--env {env!r} has no overlay, but {source} declares {env!r} an " + f"environment of this product, so the base contract (bound to " + f"{', '.join(platforms) or 'nothing'}) would be used as if it were " + f"{env!r}. Add overlays/{env}.yaml, or remove {env!r} from {source}" + ), + }, + ) + # ``str()`` of the error is its sentence, not just the event: most + # callers wrap a load failure as ``{"error": str(e)}``. + refusal.args = (refusal.context["error"],) + raise refusal + + def note_missing_overlay( - contract_path: str | Path, env: str, logger: Optional[logging.Logger] = None + contract_path: str | Path, + env: str, + logger: Optional[logging.Logger] = None, + *, + contract: Optional[Mapping[str, Any]] = None, ) -> None: """Report that ``env`` matched no overlay for ``contract_path``, once. - WARNING for any env but :data:`BASE_ENV_BY_CONVENTION`, naming the env and - the overlays that do exist, because the caller is about to use the base - contract where it asked for an environment. INFO for ``dev``, which is - the base by convention. Emitted once per (contract, env) per process — - one command loads the same contract several times. + WARNING for any env but :data:`BASE_ENV_BY_CONVENTION`, naming the env, + the overlays that do exist and the platforms the base binds to, because + the caller is about to use the base contract where it asked for an + environment. INFO for ``dev``, which is the base by convention. Emitted + once per (contract, env) per process — one command loads the same + contract several times. + + ``contract`` (the base) enables the refusal: an env the workspace + expects (:func:`refuse_declared_missing_overlay`) is an error, every + time. An env only the contract's ``environments`` block names is said in + the warning. """ log = logger or LOG + if contract is not None: + refuse_declared_missing_overlay(contract_path, env, contract) contract_key = str(Path(contract_path).resolve()) with _NOTED_MISSING_OVERLAYS_LOCK: if (contract_key, env) in _NOTED_MISSING_OVERLAYS: @@ -509,14 +638,30 @@ def note_missing_overlay( ) return existing = available_overlay_envs(contract_key) + platforms = _base_platforms(contract) if contract is not None else [] + environments = contract.get("environments") if contract is not None else None + in_block = isinstance(environments, dict) and env in environments log.warning( "overlay_not_found: --env %r matched no overlay for %s, so the BASE contract is " - "used unchanged. Overlays that exist: %s. Add an overlay for it under overlays/ " + "used unchanged%s.%s Overlays that exist: %s. Add an overlay for it under overlays/ " "or pass one of the existing environments.", env, contract_key, + f" (it binds to {', '.join(platforms)}, not to {env!r})" if platforms else "", + ( + f" The contract's environments block names {env!r}, but forge-cli applies " + "nothing from that block; an overlay is what changes a binding." + if in_block + else "" + ), ", ".join(existing) if existing else "none", - extra={"event": "overlay_not_found", "env": env, "available_envs": existing}, + extra={ + "event": "overlay_not_found", + "env": env, + "available_envs": existing, + "base_platforms": platforms, + "declared_in_environments_block": in_block, + }, ) @@ -603,7 +748,7 @@ def load_with_overlay( # No overlay found. This used to be a DEBUG line, so ``--env prod`` # with a typo'd or missing overlay silently deployed the BASE # contract at the default log level. Say so, once per contract/env. - note_missing_overlay(base_path, env, log) + note_missing_overlay(base_path, env, log, contract=base) return base # No env → return base as-is diff --git a/fluid_build/observability/apply_run.py b/fluid_build/observability/apply_run.py new file mode 100644 index 00000000..e5d31836 --- /dev/null +++ b/fluid_build/observability/apply_run.py @@ -0,0 +1,47 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The ``fluid apply`` run in progress in this context, for code below the CLI. + +``cli/_apply_cc_report.py`` opens one Command Center run report per ``fluid +apply`` and sets it here. ``build_runners`` records each build on it +(``record_build`` / ``build_failed``) without importing the CLI, which the +layering in ``tests/observability/test_import_hygiene.py`` forbids. +""" + +from __future__ import annotations + +import contextvars +from typing import Any, Optional + +_CURRENT: contextvars.ContextVar[Optional[Any]] = contextvars.ContextVar( + "fluid_apply_run", default=None +) + + +def current_apply_run() -> Optional[Any]: + """The report of the ``fluid apply`` running in this context, or ``None``.""" + return _CURRENT.get() + + +def set_current_apply_run(report: Optional[Any]) -> "contextvars.Token[Optional[Any]]": + """Make ``report`` the current run; pass the token to :func:`reset_current_apply_run`.""" + return _CURRENT.set(report) + + +def reset_current_apply_run(token: "contextvars.Token[Optional[Any]]") -> None: + _CURRENT.reset(token) + + +__all__ = ["current_apply_run", "reset_current_apply_run", "set_current_apply_run"] diff --git a/fluid_build/observability/config.py b/fluid_build/observability/config.py index ff004214..e57a1605 100644 --- a/fluid_build/observability/config.py +++ b/fluid_build/observability/config.py @@ -17,9 +17,9 @@ """ import os -from dataclasses import dataclass +from dataclasses import dataclass, field from pathlib import Path -from typing import Optional +from typing import Dict, Optional import yaml @@ -53,6 +53,11 @@ class CommandCenterConfig: retry_attempts: int = 3 batch_size: int = 100 # logs/metrics per batch flush_interval: int = 5 # seconds + #: Extra request headers: the ``Authorization`` a bearer credential needs + #: and the ``X-Organization-Id`` the Command Center scopes a run to. Set + #: by callers that authenticate the way ``fluid publish`` does + #: (``cli/_apply_cc_report.py``); never read from a file here. + headers: Dict[str, str] = field(default_factory=dict) @classmethod def from_environment(cls) -> "CommandCenterConfig": @@ -128,7 +133,8 @@ def is_configured(self) -> bool: Returns: True if URL and API key are set, False otherwise """ - return bool(self.enabled and self.url and self.api_key) + credential = self.api_key or self.headers.get("Authorization") + return bool(self.enabled and self.url and credential) def __repr__(self) -> str: # Mask API key for security @@ -136,7 +142,9 @@ def __repr__(self) -> str: if self.api_key: visible_prefix = self.api_key[:4] masked_key = f"{visible_prefix}***REDACTED***" + # Header names only: an Authorization value is a credential. return ( f"CommandCenterConfig(url={self.url}, api_key={masked_key}, " - f"enabled={self.enabled}, timeout={self.timeout})" + f"enabled={self.enabled}, timeout={self.timeout}, " + f"headers={sorted(self.headers)})" ) diff --git a/fluid_build/observability/reporter.py b/fluid_build/observability/reporter.py index 0b3eb7d8..203cb55e 100644 --- a/fluid_build/observability/reporter.py +++ b/fluid_build/observability/reporter.py @@ -225,13 +225,19 @@ def __init__(self, config: CommandCenterConfig): # Circuit breaker self.circuit_breaker = CircuitBreaker(failure_threshold=5, timeout=60, success_threshold=1) + #: What the worker managed to deliver, for a caller that has to say + #: whether the run reached the Command Center (the queue is async). + self.stats: Dict[str, int] = {"sent": 0, "failed": 0} + # Session for connection pooling self.session: Optional[Any] = None if self.enabled and requests: self.session = requests.Session() - self.session.headers.update( - {"X-API-Key": config.api_key, "Content-Type": "application/json"} - ) + headers = {"Content-Type": "application/json"} + if config.api_key: + headers["X-API-Key"] = config.api_key + headers.update(config.headers or {}) + self.session.headers.update(headers) def start(self): """Start background worker thread.""" @@ -477,7 +483,17 @@ def _send_event(self, event: Dict[str, Any]): response.raise_for_status() logger.debug(f"Command Center: {method} {endpoint} → {response.status_code}") + self.stats["sent"] += 1 except Exception as e: - logger.warning(f"Failed to send event to Command Center: {e}") + self.stats["failed"] += 1 + # The type and the status only: a requests exception's text + # repeats the URL, and nothing here needs the response body. + status = getattr(getattr(e, "response", None), "status_code", None) + logger.warning( + "Failed to send event to Command Center: %s %s -> %s", + method, + endpoint, + status or type(e).__name__, + ) raise diff --git a/fluid_build/policy/sovereignty.py b/fluid_build/policy/sovereignty.py index 6caa7f5d..f5a453bc 100644 --- a/fluid_build/policy/sovereignty.py +++ b/fluid_build/policy/sovereignty.py @@ -27,7 +27,7 @@ from enum import Enum from pathlib import Path from types import MappingProxyType -from typing import Any, Dict, List, Mapping, Optional, Tuple +from typing import AbstractSet, Any, Dict, FrozenSet, List, Mapping, Optional, Sequence, Set, Tuple from ._common import iter_exposes @@ -182,6 +182,8 @@ def region_jurisdiction_map() -> Mapping[str, str]: for provider in ("aws", "gcp", "azure"): table.update(_load_vendored(provider)) table.update(SovereigntyValidator.VENDORED_CORRECTIONS) + table.update(_gcp_locations_not_vendored()) + table.update(_MULTI_REGION_JURISDICTIONS) table.update(_load_botocore_aws()) # Read-only: the cached object is shared process-wide (and re-exported by # providers/aws/util/sovereignty.py), so a stray mutation anywhere would @@ -370,12 +372,38 @@ def validate(self, contract: Dict[str, Any]) -> Tuple[bool, List[SovereigntyViol Returns: (is_valid, violations) - is_valid=False means BLOCK deployment in strict mode """ - violations = [] - # Extract sovereignty config (optional in 0.7.1) sovereignty = contract.get("sovereignty") if not sovereignty: return True, [] # No sovereignty constraints = always valid + return self.check_placements( + sovereignty, + contract_placements(contract), + region_placed=region_placed_exposes(contract), + ) + + def check_placements( + self, + sovereignty: Mapping[str, Any], + placements: Sequence[Tuple[str, Optional[str]]], + *, + region_placed: AbstractSet[str] = frozenset(), + ) -> Tuple[bool, List[SovereigntyViolation]]: + """Evaluate ``sovereignty`` against each ``(where, region)`` placement. + + :meth:`validate` passes the contract's own bindings + (:func:`contract_placements`); a provider hook passes the places its + emitted resources actually land (the GCP plugin passes every resource + ``location``), so a region the provider filled in by default is + checked where it is used. A placement whose region is ``None`` is a + binding on a region-placed platform that names none. + + ``region_placed`` names the placements (their ``where``) that sit on a + :data:`REGION_PLACED_PLATFORMS` cloud. For those, a strict + jurisdiction refuses a region whose jurisdiction cannot be resolved + (check 3); elsewhere that stays a warning. + """ + violations = [] # Defaults MUST mirror the JSON schema's declared ``default`` keys # (``$defs.sovereignty`` in fluid-schema-0.7.x.json). They previously @@ -397,16 +425,36 @@ def validate(self, contract: Dict[str, Any]) -> Tuple[bool, List[SovereigntyViol "crossBorderTransfer", DEFAULT_CROSS_BORDER_TRANSFER ) - # Validate each expose's binding location - for expose in iter_exposes(contract): - binding = expose.get("binding", {}) - location = binding.get("location", {}) - region = location.get("region") - + # Validate each place the contract puts data + for expose_id, region in placements: + # Check 0: a region-placed binding that names no region. It used + # to be skipped ("no region, nothing to check"), which failed OPEN: + # validate and ``plan --check-sovereignty`` printed PASS while the + # platform picked the region itself (BigQuery: the US multi-region, + # measured on an EU-only contract). The mode decides, like check 2: + # strict refuses, advisory warns, audit logs. if not region: - continue # No region specified, skip validation - - expose_id = expose.get("exposeId", "unknown") + violations.append( + SovereigntyViolation( + severity=severity_for(enforcement_mode), + message=( + "Binding declares no region, so where its data lives cannot " + "be checked against the sovereignty policy (the platform " + "would choose)" + ), + expose_id=expose_id, + region_expected=allowed_regions or None, + suggestion=( + "Set binding.location.region" + + ( + f" to one of: {', '.join(allowed_regions)}" + if allowed_regions + else "" + ) + ), + ) + ) + continue # Check 1: Denied regions — deliberately an error in EVERY mode. # @@ -448,25 +496,42 @@ def validate(self, contract: Dict[str, Any]) -> Tuple[bool, List[SovereigntyViol if jurisdiction and jurisdiction not in UNCONSTRAINED_JURISDICTIONS: region_jurisdiction = region_jurisdiction_map().get(region, "Unknown") if region_jurisdiction != jurisdiction and region_jurisdiction != "Global": - # "Unknown" is an inability to evaluate, not a violation, and - # the two must not be conflated: a region the vendored table - # does not carry says nothing about where it actually is. - # Escalating it under strict would fail closed on every - # contract using an unmapped region — defensible for a - # sovereignty control, but a separate decision with its own - # blast radius, not a side effect of making enforcementMode - # mean what the schema says. Check 4 already draws this exact - # line and refuses to let one Unknown agree with another. + # "Unknown" is an inability to evaluate, not a violation: a + # region the table does not carry says nothing about where + # it is. On a cloud region (aws / gcp / azure) under + # strict, though, a jurisdiction the policy cannot show is + # a place it cannot allow: the GCP table lagged Google by + # nine regions and the ASIA multi-region, and each went + # through validate and `generate iac` with a warning + # (me-central2 on an EU-only contract, measured). So + # strict refuses it there, unless the operator vouched for + # the region by naming it in allowedRegions. Advisory and + # audit, and every other platform, keep the warning. + # Check 4 still refuses to let one Unknown agree with + # another. unresolvable = region_jurisdiction == "Unknown" + fail_closed = ( + unresolvable + and enforcement_mode is EnforcementMode.STRICT + and expose_id in region_placed + and region not in allowed_regions + ) violations.append( SovereigntyViolation( severity=( - "warning" if unresolvable else severity_for(enforcement_mode) + "warning" + if unresolvable and not fail_closed + else severity_for(enforcement_mode) ), message=f"Region '{region}' (jurisdiction: {region_jurisdiction}) " f"does not match required jurisdiction: {jurisdiction}", expose_id=expose_id, - suggestion=f"Consider using regions in {jurisdiction} jurisdiction", + suggestion=( + f"Use a region in the {jurisdiction} jurisdiction; if " + f"'{region}' is one, name it in sovereignty.allowedRegions" + if fail_closed + else f"Consider using regions in {jurisdiction} jurisdiction" + ), ) ) @@ -489,11 +554,9 @@ def validate(self, contract: Dict[str, Any]) -> Tuple[bool, List[SovereigntyViol # than being silently folded into a jurisdiction comparison. if data_residency and not cross_border_transfer: baseline: Any = _UNSET - for exp in iter_exposes(contract): - exp_region = exp.get("binding", {}).get("location", {}).get("region") + for exp_id, exp_region in placements: if not exp_region: continue - exp_id = exp.get("exposeId", "unknown") exp_jurisdiction = region_jurisdiction_map().get(exp_region, "Unknown") if exp_jurisdiction == "Unknown": @@ -552,6 +615,111 @@ def validate(self, contract: Dict[str, Any]) -> Tuple[bool, List[SovereigntyViol return is_valid, violations +#: Platforms whose bindings put data in a cloud region: a binding on one of +#: these with a sovereignty block and no region is a finding (check 0), not a +#: skip. Other platforms (``local`` above all, and those whose region lives +#: outside the binding) keep the old behaviour: no region, nothing checked. +REGION_PLACED_PLATFORMS = frozenset({"aws", "gcp", "azure"}) + +#: Multi-region locations the vendored region table does not carry. BigQuery +#: and Cloud Storage both name their multi-regions ``US`` (data centres in the +#: United States) and ``EU`` (data centres in EU member states); left unmapped +#: they resolved "Unknown", so the ``US`` a GCP binding with no region used to +#: land in could not fail a ``jurisdiction: EU`` check. +_MULTI_REGION_JURISDICTIONS = {"US": "US", "EU": "EU", "us": "US", "eu": "EU"} + + +#: GCP locations the vendored dataset (dgl/cloud-regions) does not carry, by +#: the countries their data centres are in, from Google's own location lists +#: (docs.cloud.google.com/bigquery/docs/locations and +#: /storage/docs/locations, read 2026-09-28). Kept here, not patched into the +#: ODbL csv, for the reason ``VENDORED_CORRECTIONS`` gives. A location +#: resolves only when every country it spans is in one jurisdiction: the +#: dual-regions EUR5 (Belgium + London), EUR7 (London + Frankfurt) and EUR8 +#: (Frankfurt + Zürich) span two and stay Unknown, as does the ASIA +#: multi-region ("data centres in Asia", several countries), which a strict +#: jurisdiction then refuses on a cloud region (check 3). +_GCP_LOCATION_COUNTRIES: Dict[str, Tuple[str, ...]] = { + "africa-south1": ("za",), # Johannesburg + "asia-southeast3": ("th",), # Bangkok + "europe-north2": ("se",), # Stockholm + "europe-west10": ("de",), # Berlin + "europe-west12": ("it",), # Turin + "me-central1": ("qa",), # Doha + "me-central2": ("sa",), # Dammam + "me-west1": ("il",), # Tel Aviv + "northamerica-south1": ("mx",), # Mexico + # Cloud Storage predefined dual-regions (either case is accepted). + "asia1": ("jp",), # Tokyo + Osaka + "eur4": ("fi", "nl"), # Finland + Netherlands + "eur5": ("be", "uk"), # Belgium + London + "eur7": ("uk", "de"), # London + Frankfurt + "eur8": ("de", "ch"), # Frankfurt + Zürich + "nam4": ("us",), # Iowa + South Carolina +} + + +def _gcp_locations_not_vendored() -> Dict[str, str]: + """``location -> jurisdiction`` for :data:`_GCP_LOCATION_COUNTRIES`.""" + resolved: Dict[str, str] = {} + for location, countries in _GCP_LOCATION_COUNTRIES.items(): + found = {SovereigntyValidator.COUNTRY_JURISDICTIONS.get(c) for c in countries} + jurisdiction = found.pop() if len(found) == 1 else None + if jurisdiction is None: + continue + resolved[location] = jurisdiction + if location.isalnum(): # a dual-region code: EUR4 and eur4 alike + resolved[location.upper()] = jurisdiction + return resolved + + +def region_placed_exposes(contract: Mapping[str, Any]) -> FrozenSet[str]: + """``exposeId`` of every expose bound to a :data:`REGION_PLACED_PLATFORMS` cloud.""" + out: Set[str] = set() + for expose in iter_exposes(dict(contract)): + binding = expose.get("binding") or {} + if isinstance(binding, Mapping): + if str(binding.get("platform") or "").lower() in REGION_PLACED_PLATFORMS: + out.add(str(expose.get("exposeId", "unknown"))) + return frozenset(out) + + +def binding_region(binding: Mapping[str, Any]) -> Optional[str]: + """The region a binding places its data in, read where its emitter reads it. + + ``location.region``, and for GCP also ``location.location``: the GCP + emitter falls back to it (``iac/providers/gcp.py``), so a check that read + ``region`` alone never saw a BigQuery dataset placed through ``location``. + """ + location = binding.get("location") if isinstance(binding, Mapping) else None + if not isinstance(location, Mapping): + return None + region = location.get("region") + if not region and str(binding.get("platform") or "").lower() == "gcp": + region = location.get("location") + return str(region) if region else None + + +def contract_placements(contract: Mapping[str, Any]) -> List[Tuple[str, Optional[str]]]: + """``(exposeId, region)`` for each expose the sovereignty checks evaluate. + + An expose with a region is always listed. One without is listed with + ``None`` only on a :data:`REGION_PLACED_PLATFORMS` platform, where the + region is the platform's to pick; any other binding with no region is + left out, as before. + """ + out: List[Tuple[str, Optional[str]]] = [] + for expose in iter_exposes(dict(contract)): + binding = expose.get("binding") or {} + if not isinstance(binding, Mapping): + continue + region = binding_region(binding) + platform = str(binding.get("platform") or "").lower() + if region or platform in REGION_PLACED_PLATFORMS: + out.append((str(expose.get("exposeId", "unknown")), region)) + return out + + def validate_sovereignty(contract: Dict[str, Any]) -> Tuple[bool, List[str]]: """ Convenience function for CLI integration. diff --git a/fluid_build/providers/gcp/provider.py b/fluid_build/providers/gcp/provider.py index a7c24ce1..3ebb00fc 100644 --- a/fluid_build/providers/gcp/provider.py +++ b/fluid_build/providers/gcp/provider.py @@ -186,6 +186,12 @@ def plan( # Older planner signature without mode kwarg. actions = plan_actions(contract, self.project, self.region, self.logger) + # Sovereignty, the way AwsProvider.plan does it: a refusal raised + # here reaches `fluid apply` / `fluid generate iac` through + # native_actions, which re-raises a sovereignty veto instead of + # treating it as "planner unavailable". + self._validate_sovereignty(contract, actions) + self.info_kv( event="plan_completed", contract_id=contract.get("id"), @@ -195,10 +201,87 @@ def plan( return actions + except ProviderError: + raise except Exception as e: self.err_kv(event="plan_failed", contract_id=contract.get("id"), error=str(e)) raise ProviderError(f"Failed to plan GCP deployment: {e}") from e + def _validate_sovereignty( + self, contract: Mapping[str, Any], actions: List[Dict[str, Any]] + ) -> None: + """Refuse a planned placement outside ``contract.sovereignty``. + + Checks each gcp expose's binding region (none is a finding under + strict) and every ``location`` / ``region`` a planned action carries: + where the planner fell back to a default (``US`` for a dataset, this + provider's region for a scheduler job or a staging bucket), that + default is what gets checked. See ``util/sovereignty.py``. + """ + if not contract.get("sovereignty"): + return + from fluid_build._errors import ResidencyViolationError, SovereigntyViolationError + + from .util.sovereignty import action_placements, enforce_gcp_sovereignty + + try: + enforce_gcp_sovereignty(contract, action_placements(actions), logger=self.logger) + except (SovereigntyViolationError, ResidencyViolationError) as e: + self.err_kv(event="sovereignty_violation", error=str(e)) + raise ProviderError(str(e)) from e + self.info_kv(event="sovereignty_validated", region=self.region) + + def validate_sovereignty(self, contract: Mapping[str, Any]) -> Optional[List[str]]: + """``fluid plan --check-sovereignty``: what ``fluid apply`` would refuse, and why. + + Stage 7 checks sovereignty twice, both where the data lands: this + provider's planned actions (:meth:`_validate_sovereignty`) and every + resource the OpenTofu plugin emits (``GcpIacPlugin.emit``: the dataset, + and the KMS key ring and Data Catalog taxonomy governance adds). The + plan stage used to run only the policy engine over the bindings, so a + placement the emitter derived passed stage 6 and was refused at stage + 7. This runs the same placements through the same engine, plus the + engine's own contract-level checks, and returns every error-severity + finding (``where: message``); a warning or an info finding is logged, + as apply logs it, and does not block. + + ``None`` (no verdict) for a contract with no ``sovereignty`` block, so + the plan reports NOT CHECKED rather than a pass. A planner or emitter + that cannot run raises; ``run_validate_sovereignty`` reads that as no + verdict too, and the plan falls back to the policy engine. + """ + if not contract.get("sovereignty"): + return None + from fluid_build.iac import get_iac_plugin + from fluid_build.iac.plan_packaging import filter_referenced_container_actions + from fluid_build.policy.sovereignty import SovereigntyValidator + + from .util.sovereignty import ( + action_placements, + gcp_sovereignty_violations, + resource_placements, + ) + + actions = plan_actions(contract, self.project, self.region, self.logger) + # What apply emits from: the planned actions less the creations a + # shared pool already holds (``generate_iac.native_actions``). + kept, _dropped = filter_referenced_container_actions(contract, list(actions)) + resources = get_iac_plugin("gcp").emit(contract, kept, enforce_sovereignty=False) + + _ok, found = SovereigntyValidator().validate(dict(contract)) + found = list(found) + found += gcp_sovereignty_violations(contract, action_placements(actions)) + found += gcp_sovereignty_violations(contract, resource_placements(resources)) + errors: List[str] = [] + for v in found: + line = f"{v.expose_id}: {v.message}" + if v.severity == "error": + if line not in errors: + errors.append(line) + else: + self.warn_kv(event="sovereignty_finding", severity=v.severity, finding=line) + return errors + def apply(self, actions: List[Dict[str, Any]], **kwargs: Any) -> ApplyResult: """Native GCP apply is retired — GCP uses the OpenTofu engine. diff --git a/fluid_build/providers/gcp/util/sovereignty.py b/fluid_build/providers/gcp/util/sovereignty.py new file mode 100644 index 00000000..cd943ef6 --- /dev/null +++ b/fluid_build/providers/gcp/util/sovereignty.py @@ -0,0 +1,228 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The GCP provider's sovereignty hook: refuse a placement outside the policy. + +AWS refuses an out-of-jurisdiction region inside its planner +(``AwsProvider._validate_sovereignty``), and ``fluid generate iac`` / +``fluid apply`` recognise that refusal through ``native_actions``. GCP had no +such hook, so only ``fluid validate`` stood between a contract and a dataset +in the wrong jurisdiction, and a binding with no region went through every +stage and landed in BigQuery's ``US`` multi-region. + +The check is the policy engine's own (``policy.sovereignty``, +``check_placements``), not a second rule set, applied to where the data +actually goes rather than to what the binding says: + +* each gcp expose whose binding names no region (the engine's check 0: the + mode decides, strict refuses); +* each ``location`` / ``region`` a resource carries: every resource the + OpenTofu plugin emits (``GcpIacPlugin.emit``, the chokepoint of ``fluid + apply`` and ``fluid generate iac``, which runs whether or not a native + provider can be built), and every action the native planner produced + (``GcpProvider.plan``). That is where a region the provider filled in by + default (the planner's ``US`` dataset default, the provider region a + Cloud Scheduler job or a staging bucket inherits, ``--region``'s + ``europe-west3`` or the SDK's ``us-central1``) meets ``allowedRegions``. + +Error-severity findings raise :class:`~fluid_build._errors.SovereigntyViolationError` +(the typed error ``generate_iac._is_sovereignty_refusal`` recognises through +a ``ProviderError`` cause chain); warnings and info are logged and pass, +which is what ``advisory`` and ``audit`` mean. +""" + +from __future__ import annotations + +import logging +from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple + +from fluid_build.policy.sovereignty import ( + SovereigntyValidator, + SovereigntyViolation, + binding_region, +) + +_LOG = logging.getLogger(__name__) + +Placement = Tuple[str, Optional[str]] + + +def unplaced_gcp_exposes(contract: Mapping[str, Any]) -> List[Placement]: + """``(exposeId, None)`` for each gcp expose whose binding names no region.""" + out: List[Placement] = [] + for expose in contract.get("exposes") or []: + if not isinstance(expose, Mapping): + continue + binding = expose.get("binding") or {} + if not isinstance(binding, Mapping): + continue + if str(binding.get("platform") or "").lower() != "gcp": + continue + if binding_region(binding) is None: + out.append((str(expose.get("exposeId", "unknown")), None)) + return out + + +#: Services that name BigQuery's two multi-regions in their own vocabulary, by +#: resource type: ``{their location id: the BigQuery multi-region}``. A key +#: for a dataset in ``EU`` must be in the Cloud KMS location ``europe`` +#: (``us`` for ``US``; the BigQuery CMEK guide), and the taxonomy whose policy +#: tags restrict its columns in the Data Catalog location ``eu`` (``us``). +#: Read as themselves, those ids failed ``allowedRegions: [EU]`` and, under a +#: strict jurisdiction, resolved to none, so the key ring and the taxonomy of +#: an EU dataset were refused at apply after validate and plan passed. They +#: are the dataset's own place, and are checked as it. +_BIGQUERY_MULTI_REGION_ALIASES: Mapping[str, Mapping[str, str]] = { + "google_kms_key_ring": {"europe": "eu", "us": "us"}, + "google_data_catalog_taxonomy": {"eu": "eu", "us": "us"}, +} + + +def resource_placements(resources: Mapping[str, Any]) -> List[Placement]: + """``(address, location)`` for every emitted resource that names one. + + ``resources`` is the plugin's ``{type: {name: body}}``. A value that is an + OpenTofu reference (``${...}``) is not a place and is skipped. A Pub/Sub + topic has no ``location``: where its messages are stored is its + ``message_storage_policy.allowed_persistence_regions``, one placement + per region. A KMS key ring or a Data Catalog taxonomy in one of + BigQuery's multi-regions is placed at that multi-region, spelled as the + emitted dataset spells it (:data:`_BIGQUERY_MULTI_REGION_ALIASES`). + """ + spellings = _multi_region_spellings(resources) + out: List[Placement] = [] + for rtype, by_name in (resources or {}).items(): + if not isinstance(by_name, Mapping): + continue + for name, body in by_name.items(): + if not isinstance(body, Mapping): + continue + for key in ("location", "region"): + value = body.get(key) + if _is_place(value): + out.append((f"{rtype}.{name}", _as_placed(rtype, str(value), spellings))) + break + for region in _persistence_regions(body): + out.append((f"{rtype}.{name}", region)) + return out + + +def _multi_region_spellings(resources: Mapping[str, Any]) -> Dict[str, str]: + """``{"eu" | "us": location}`` as the emitted BigQuery datasets spell each multi-region.""" + out: Dict[str, str] = {} + datasets = (resources or {}).get("google_bigquery_dataset") + for body in (datasets or {}).values() if isinstance(datasets, Mapping) else (): + value = body.get("location") if isinstance(body, Mapping) else None + if _is_place(value) and str(value).lower() in ("eu", "us"): + out.setdefault(str(value).lower(), str(value)) + return out + + +def _as_placed(rtype: str, value: str, spellings: Mapping[str, str]) -> str: + """``value``, or the BigQuery multi-region it names for ``rtype``.""" + multi = _BIGQUERY_MULTI_REGION_ALIASES.get(rtype, {}).get(value.lower()) + if multi is None: + return value + return spellings.get(multi, multi.upper()) + + +def _is_place(value: Any) -> bool: + return isinstance(value, str) and bool(value) and not value.startswith("${") + + +def _persistence_regions(body: Mapping[str, Any]) -> List[str]: + """``message_storage_policy.allowed_persistence_regions`` (block as object or list).""" + policy = body.get("message_storage_policy") + blocks = policy if isinstance(policy, list) else [policy] + out: List[str] = [] + for block in blocks: + if isinstance(block, Mapping): + regions = block.get("allowed_persistence_regions") or [] + out.extend(str(r) for r in regions if _is_place(r)) + return out + + +def action_placements(actions: Iterable[Mapping[str, Any]]) -> List[Placement]: + """``(action id, location)`` for every planned action that names one.""" + out: List[Placement] = [] + for index, action in enumerate(actions or ()): + if not isinstance(action, Mapping): + continue + for key in ("location", "region"): + value = action.get(key) + if isinstance(value, str) and value: + where = str(action.get("id") or action.get("op") or f"action[{index}]") + out.append((where, value)) + break + return out + + +def gcp_sovereignty_violations( + contract: Mapping[str, Any], placements: Sequence[Placement] +) -> List[SovereigntyViolation]: + """The engine's findings for this contract's gcp exposes and ``placements``.""" + sovereignty = contract.get("sovereignty") + if not isinstance(sovereignty, Mapping) or not sovereignty: + return [] + everything = unplaced_gcp_exposes(contract) + list(placements) + if not everything: + return [] + # Every placement here is a GCP one, so a jurisdiction the table cannot + # resolve is refused under strict, as for any cloud region (check 3). + _, violations = SovereigntyValidator().check_placements( + sovereignty, everything, region_placed={where for where, _ in everything} + ) + return violations + + +def enforce_gcp_sovereignty( + contract: Mapping[str, Any], + placements: Sequence[Placement], + *, + logger: Optional[logging.Logger] = None, +) -> None: + """Raise on an error-severity finding; log the rest.""" + from fluid_build._errors import SovereigntyViolationError, doc_url + + log = logger or _LOG + violations = gcp_sovereignty_violations(contract, placements) + errors = [v for v in violations if v.severity == "error"] + for v in violations: + if v.severity != "error": + log.warning("gcp sovereignty (%s): %s: %s", v.severity, v.expose_id, v.message) + if not errors: + return + sovereignty = contract.get("sovereignty") or {} + allowed = [str(r) for r in (sovereignty.get("allowedRegions") or [])] + # ``where: message``, de-duplicated in order. Not ``[where]``: the CLI + # renders errors through rich, which reads square brackets as markup. + findings = "; ".join(dict.fromkeys(f"{v.expose_id}: {v.message}" for v in errors)) + raise SovereigntyViolationError( + what=f"GCP placement refused by the sovereignty policy: {findings}", + why=( + "contract.sovereignty is enforced " + f"({sovereignty.get('enforcementMode', 'strict')}) and " + + ( + f"allows {', '.join(allowed)}" + if allowed + else f"requires jurisdiction {sovereignty.get('jurisdiction')!r}" + ) + + "; this is where the GCP resources would be created." + ), + fix=( + "Set binding.location.region on every gcp expose to an allowed region " + "(the BigQuery dataset and bucket take it), or change the sovereignty policy." + ), + doc=doc_url("sovereignty"), + ) diff --git a/fluid_build/schedulers/airflow/fluid_apply.py b/fluid_build/schedulers/airflow/fluid_apply.py index df16ffc5..3ccb4f3d 100644 --- a/fluid_build/schedulers/airflow/fluid_apply.py +++ b/fluid_build/schedulers/airflow/fluid_apply.py @@ -497,8 +497,16 @@ def validate_env_name(raw: Any) -> str: return validate_id(raw, kind="env") -def dag_id_for(product_id: str, build_id: str) -> str: - dag_id = f"{product_id}__{build_id}" +def dag_id_for(product_id: str, build_id: str, env: Optional[str] = None) -> str: + """``__``, or ``____`` for an env's DAG. + + The env is part of the id because one contract deployed to two clouds + (``--env aws`` and ``--env gcp`` overlays) is one product id: both DAGs + had the same id, and in one Airflow the second to parse replaced the + first. Airflow keys run history on the id, so an env-bound DAG from an + earlier release starts a new history under its new id. + """ + dag_id = f"{product_id}__{env}__{build_id}" if env else f"{product_id}__{build_id}" if len(dag_id) > _MAX_DAG_ID: raise ScheduleRenderError( f"dag_id {dag_id!r} exceeds Airflow's {_MAX_DAG_ID}-character limit; " @@ -507,6 +515,18 @@ def dag_id_for(product_id: str, build_id: str) -> str: return dag_id +def schedule_scope_for(product_id: str, env: Optional[str] = None) -> str: + """The directory an env's DAGs live in, under the schedule artifacts root. + + ```` with no env, ``__`` with one. ``fluid + schedule-sync --delete-scope product`` mirrors each such directory onto + the same-named one at the scheduler, deleting what the source lacks, so + the aws and the gcp DAGs of one product need a directory each or each + sync deletes the other's. + """ + return f"{product_id}__{env}" if env else product_id + + def dag_filename_for(build_id: str) -> str: # A plain module name: dots in a file name make Airflow import the DAG # under a dotted module path. @@ -644,7 +664,7 @@ def render_dag( ")\n" "\n" "with DAG(\n" - f" dag_id={lit(dag_id_for(product_id, build.build_id))},\n" + f" dag_id={lit(dag_id_for(product_id, build.build_id, env))},\n" f" description={lit(f'fluid apply {product_id} --build-id {build.build_id}')},\n" " schedule=SCHEDULE,\n" " start_date=pendulum.datetime(2026, 1, 1, tz=TIMEZONE),\n" @@ -727,6 +747,7 @@ def render_fluid_apply_dags( "has_scheduled_builds", "render_dag", "render_fluid_apply_dags", + "schedule_scope_for", "scheduled_builds", "uses_fluid_apply_dags", "validate_contract_path", diff --git a/fluid_build/schemas/fluid-schema-0.7.6.json b/fluid_build/schemas/fluid-schema-0.7.6.json index d0173f5c..28ca63eb 100644 --- a/fluid_build/schemas/fluid-schema-0.7.6.json +++ b/fluid_build/schemas/fluid-schema-0.7.6.json @@ -1184,26 +1184,29 @@ }, "columnRestrictions": { "type": "array", - "description": "Column-level access control.", + "description": "Column-level access control. A column named in any restriction is restricted. 'deny': the principal may not read the columns. 'allow': the columns are readable only by the principals an allow names. A deny beats an allow, and a restriction never grants access: the readers are the expose's readers. Principals are logical and resolve through binding.principals. NEW in v0.7.6 (the field itself is older): `fluid apply` enforces it. GCP BigQuery: a Data Catalog taxonomy per product and dataset with fine-grained access control, a policy tag per set of restricted columns that share their readers, attached through the table schema's policyTags, and roles/datacatalog.categoryFineGrainedReader for exactly the expose's readers allowed to read them: the accessPolicy read grantees and the expose's own policy.authz.readers (a denied principal gets an access error on the columns). A restriction on an expose with no reader on the binding is refused, as it would lock the columns for everyone. A restriction's tags and labels are written into the policy tag's description; AWS has no place for them. AWS: each governance.lakeFormation grant with SELECT excludes the restricted columns its principal may not read (table_with_columns.excluded_column_names); a grant's hand-written excludedColumns must agree, and a binding with no Lake Formation grants is refused. `fluid verify` checks the policy tags and their readers on GCP, and the principals' Lake Formation permissions on AWS.", "items": { "type": "object", "additionalProperties": false, "properties": { "principal": { - "type": "string" + "type": "string", + "description": "Logical principal the restriction applies to, written as in accessPolicy (e.g. group:analysts@company.example); binding.principals maps it per cloud." }, "columns": { "type": "array", "items": { "type": "string" - } + }, + "description": "Columns of this expose's schema." }, "access": { "type": "string", "enum": [ "allow", "deny" - ] + ], + "description": "deny: the principal may not read the columns. allow: only the principals an allow names may read them." }, "tags": { "$ref": "#/$defs/tags", @@ -1626,7 +1629,7 @@ "encryption": { "type": "object", "additionalProperties": false, - "description": "NEW in v0.7.6: Encryption at rest for an AWS S3 binding. Absent: nothing is emitted and S3's default encryption (SSE-S3) applies to new objects.", + "description": "NEW in v0.7.6: Encryption at rest for an AWS S3 binding or a GCP BigQuery binding. Absent: nothing is emitted, and the cloud's default applies: SSE-S3 for new S3 objects, Google-managed keys for BigQuery.", "properties": { "kms": { "type": "string", @@ -1643,13 +1646,125 @@ }, { "pattern": "^arn:aws[a-z-]*:kms:[a-z0-9-]+:[0-9]{12}:(key|alias)/[A-Za-z0-9/_-]{1,250}$" + }, + { + "pattern": "^projects/[a-z0-9.:-]{1,100}/locations/[a-z0-9-]{1,63}/keyRings/[A-Za-z0-9_-]{1,63}/cryptoKeys/[A-Za-z0-9_-]{1,63}$" } ], - "description": "NEW in v0.7.6: The KMS key the bucket's objects are encrypted with (SSE-KMS with an S3 Bucket Key). 'product' (the default): `fluid apply` creates a customer managed key for each bucket this product owns (aws_kms_key, rotation on, alias alias/fluid//) and makes it the bucket's default encryption (aws_s3_bucket_server_side_encryption_configuration). Its key policy lets the account's IAM policies decide who may use it (the KMS default statement for the account root) and, when the binding carries governance.lakeFormation, lets the Lake Formation service-linked role AWSServiceRoleForLakeFormationDataAccess encrypt and decrypt, matched by aws:PrincipalArn so the role need not exist yet. A principal querying through Athena with credentials Lake Formation vends then needs no KMS permission of its own; a grantee in another account that the bucket policy lets read the objects directly (governance.lakeFormation.bucketPolicy) is also allowed kms:Decrypt, through S3 only. The writer (the identity the build runs as) needs kms:GenerateDataKey and kms:Decrypt from IAM. `tofu destroy` deletes the alias and schedules the key for deletion after 7 days, the minimum KMS allows; kms:CancelKeyDeletion recovers it until then. Dropping the block, or naming another key, removes the product key while the bucket keeps the objects encrypted under it, which become unreadable once the deletion window ends unless they are rewritten or the deletion is cancelled; `fluid apply` refuses that plan without --allow-data-loss. On a shared (pool) bucket the bucket's default encryption is the pool owner's, so 'product' is refused there. 'alias/' or a key or alias ARN: an existing key, looked up at plan time (data aws_kms_key, so the identity running `fluid apply` needs kms:DescribeKey on it) and made the default encryption of a bucket this product owns by its key ARN, since S3 resolves an alias in the account of whoever writes; the plan fails unless the key is Enabled and a symmetric encryption key (SYMMETRIC_DEFAULT). On a pool bucket it is only checked. Its key policy is its owner's, and must let the Lake Formation role use it when the location is registered: an AWS managed key cannot be used with that role, so alias/aws/s3 is refused with registerLocation, and the plan fails for any key that is not customer managed. 'none': nothing is emitted. `fluid verify` checks that the key is Enabled and that the objects under the binding's prefix are SSE-KMS with it." + "description": "NEW in v0.7.6: The KMS key the bucket's objects are encrypted with (SSE-KMS with an S3 Bucket Key). 'product' (the default): `fluid apply` creates a customer managed key for each bucket this product owns (aws_kms_key, rotation on, alias alias/fluid//) and makes it the bucket's default encryption (aws_s3_bucket_server_side_encryption_configuration). Its key policy lets the account's IAM policies decide who may use it (the KMS default statement for the account root) and, when the binding carries governance.lakeFormation, lets the Lake Formation service-linked role AWSServiceRoleForLakeFormationDataAccess encrypt and decrypt, matched by aws:PrincipalArn so the role need not exist yet. A principal querying through Athena with credentials Lake Formation vends then needs no KMS permission of its own; a grantee in another account that the bucket policy lets read the objects directly (governance.lakeFormation.bucketPolicy) is also allowed kms:Decrypt, through S3 only. The writer (the identity the build runs as) needs kms:GenerateDataKey and kms:Decrypt from IAM. `tofu destroy` deletes the alias and schedules the key for deletion after 7 days, the minimum KMS allows; kms:CancelKeyDeletion recovers it until then. Dropping the block, or naming another key, removes the product key while the bucket keeps the objects encrypted under it, which become unreadable once the deletion window ends unless they are rewritten or the deletion is cancelled; `fluid apply` refuses that plan without --allow-data-loss. On a shared (pool) bucket the bucket's default encryption is the pool owner's, so 'product' is refused there. 'alias/' or a key or alias ARN: an existing key, looked up at plan time (data aws_kms_key, so the identity running `fluid apply` needs kms:DescribeKey on it) and made the default encryption of a bucket this product owns by its key ARN, since S3 resolves an alias in the account of whoever writes; the plan fails unless the key is Enabled and a symmetric encryption key (SYMMETRIC_DEFAULT). On a pool bucket it is only checked. Its key policy is its owner's, and must let the Lake Formation role use it when the location is registered: an AWS managed key cannot be used with that role, so alias/aws/s3 is refused with registerLocation, and the plan fails for any key that is not customer managed. 'none': nothing is emitted. `fluid verify` checks that the key is Enabled and that the objects under the binding's prefix are SSE-KMS with it. On a gcp binding (fluid-schema 0.7.6, BigQuery tables): 'product' creates a Cloud KMS key ring and crypto key for each dataset this product writes (google_kms_key_ring fluid-- and google_kms_crypto_key 'bigquery', in the dataset's location, EU and US mapping to the europe and us key locations, rotated every 90 days), grants the project's BigQuery service agent roles/cloudkms.cryptoKeyEncrypterDecrypter on it, and makes it the dataset's default_encryption_configuration and the table's encryption_configuration. The project must have the Cloud KMS API (cloudkms.googleapis.com) enabled. A key ring and key cannot be deleted on GCP: `tofu destroy` removes them from state and schedules the key's versions for destruction (30 days by default), and a later apply adopts the same names. 'projects/

/locations//keyRings//cryptoKeys/': an existing key in the dataset's location, used as given; its owner grants the BigQuery service agent. 'none': Google-managed keys. An alias/... or arn:... value is refused on a gcp binding, and a Cloud KMS name on an aws binding. Adding a key to a table that exists replaces the table (BigQuery cannot re-key it in place), so `fluid apply` refuses that plan without --allow-data-loss; the next build lands the data again. `fluid verify` checks kmsKeyName on the dataset and the table." } } + }, + "principals": { + "type": "object", + "description": "NEW in v0.7.6: Maps the contract's LOGICAL principals (accessPolicy.grants[].principal and exposes[].policy.authz.columnRestrictions[].principal, keyed exactly as the contract writes them) to the identities they stand for on this binding's platform, so the base contract stays cloud-neutral and each environment's overlay binds it. On gcp an identity is an IAM member (user:, group:, serviceAccount: or domain:); on aws an IAM principal ARN. A value is one identity or a list; [] says the principal has no identity on this cloud, and nothing is granted to it there. When the block is present, every principal the contract names for this expose must be a key: an unmapped one is refused, never emitted as written. Without the block the principals are used as written, except that a GCP principal in a reserved top-level domain (.example, .test, .invalid, .localhost) is refused as a placeholder. Modelled on ODCS v3's roles[] bound per server (servers[].roles) and on dbt grants resolved per target.", + "propertyNames": { + "minLength": 1 + }, + "additionalProperties": { + "anyOf": [ + { + "type": "string", + "minLength": 1 + }, + { + "type": "array", + "items": { + "type": "string", + "minLength": 1 + }, + "uniqueItems": true + } + ] + } } - } + }, + "allOf": [ + { + "if": { + "properties": { + "platform": { + "const": "gcp" + } + }, + "required": [ + "platform" + ] + }, + "then": { + "properties": { + "encryption": { + "properties": { + "kms": { + "not": { + "pattern": "^(alias/|arn:)" + } + } + } + }, + "principals": { + "additionalProperties": { + "anyOf": [ + { + "type": "string", + "pattern": "^(user|group|serviceAccount|domain):.+$" + }, + { + "type": "array", + "items": { + "type": "string", + "pattern": "^(user|group|serviceAccount|domain):.+$" + } + } + ] + } + } + } + } + }, + { + "if": { + "properties": { + "platform": { + "const": "aws" + } + }, + "required": [ + "platform" + ] + }, + "then": { + "properties": { + "encryption": { + "properties": { + "kms": { + "not": { + "pattern": "^projects/" + } + } + } + }, + "principals": { + "additionalProperties": { + "anyOf": [ + { + "type": "string", + "pattern": "^arn:aws[a-z0-9-]*:iam::" + }, + { + "type": "array", + "items": { + "type": "string", + "pattern": "^arn:aws[a-z0-9-]*:iam::" + } + } + ] + } + } + } + } + } + ] }, "bindingLocation": { "type": "object", @@ -1766,7 +1881,7 @@ "items": { "type": "string" }, - "description": "NEW in v0.7.5: Iceberg partition columns — a FLUID abstraction over the Iceberg PARTITIONED BY / PartitionSpec (maps to iceberg.tables.default-partition-by)." + "description": "NEW in v0.7.5: Iceberg partition columns — a FLUID abstraction over the Iceberg PARTITIONED BY / PartitionSpec (maps to iceberg.tables.default-partition-by). On a gcp BigQuery table with exposes[].lifecycle.expire: true (fluid-schema 0.7.6), one DATE, TIMESTAMP or DATETIME column to partition by (time_partitioning.field); without it the table is partitioned by ingestion time. With a column, retention counts from the date in that column, not from landing: BigQuery deletes a partition `retention` after its date, so a backfill of older rows is deleted at once. Leave it out for the S3 rule's semantics (age since written)." } } }, @@ -2076,12 +2191,12 @@ }, "retention": { "$ref": "#/$defs/isoDuration", - "description": "How long this expose's data is kept (ISO-8601 duration, e.g. P30D, P7Y). A declaration unless `expire` is true. With `expire: true` on an AWS S3 binding, `fluid apply` expires the objects under the binding's prefix this long after they are written. Years and months are counted as 365 and 30 days, and a part of a day rounds up to a whole one." + "description": "How long this expose's data is kept (ISO-8601 duration, e.g. P30D, P7Y). A declaration unless `expire` is true. With `expire: true` on an AWS S3 binding, `fluid apply` expires the objects under the binding's prefix this long after they are written; on a GCP BigQuery table it deletes each daily partition this long after its day ends. Years and months are counted as 365 and 30 days, and a part of a day rounds up to a whole one." }, "expire": { "type": "boolean", "default": false, - "description": "NEW in v0.7.6: Delete stored data once it is older than `retention`. Default false, so a contract that declares a retention period does not start deleting data when forge-cli is upgraded. Honoured by the AWS emitter on an S3 binding (other platforms do not apply it yet): `fluid apply` writes one rule per expose into the bucket's aws_s3_bucket_lifecycle_configuration, filtered to the binding's prefix (location.path, else {database}/{table}/; a binding with neither is refused, so no rule ever expires a whole bucket). The rule expires the current version `retention` after it is written, removes a noncurrent version (on a versioning-enabled bucket) one day after it stops being current, and aborts an incomplete multipart upload after 7 days or the retention period if shorter. The lifecycle configuration is AUTHORITATIVE for the whole bucket: S3 keeps one per bucket, so rules that `fluid apply` did not write are replaced. It is therefore written only for a bucket this product owns; on a shared (pool) bucket (packaging) nothing is written, the pool's owner must hold the rule, and `fluid verify` checks that an enabled rule covering the prefix expires objects after exactly this period. Objects already older than the period when the rule is first applied are expired on S3's next lifecycle run, typically within a day. Removing `expire` later deletes the lifecycle configuration, which `fluid apply` refuses without --allow-data-loss (it refuses every plan that removes a resource); `tofu destroy` deletes it." + "description": "NEW in v0.7.6: Delete stored data once it is older than `retention`. Default false, so a contract that declares a retention period does not start deleting data when forge-cli is upgraded. Honoured by the AWS emitter on an S3 binding and by the GCP emitter on a BigQuery table; on a gcp binding that is not a BigQuery table (GCS, Pub/Sub, Iceberg storage) it is refused rather than dropped. AWS: `fluid apply` writes one rule per expose into the bucket's aws_s3_bucket_lifecycle_configuration, filtered to the binding's prefix (location.path, else {database}/{table}/; a binding with neither is refused, so no rule ever expires a whole bucket). The rule expires the current version `retention` after it is written, removes a noncurrent version (on a versioning-enabled bucket) one day after it stops being current, and aborts an incomplete multipart upload after 7 days or the retention period if shorter. The lifecycle configuration is AUTHORITATIVE for the whole bucket: S3 keeps one per bucket, so rules that `fluid apply` did not write are replaced. It is therefore written only for a bucket this product owns; on a shared (pool) bucket (packaging) nothing is written, the pool's owner must hold the rule, and `fluid verify` checks that an enabled rule covering the prefix expires objects after exactly this period. Objects already older than the period when the rule is first applied are expired on S3's next lifecycle run, typically within a day. Removing `expire` later deletes the lifecycle configuration, which `fluid apply` refuses without --allow-data-loss (it refuses every plan that removes a resource holding data or policy; removing an access grant or a policy tag is a revocation and is not gated); `tofu destroy` deletes it. GCP (fluid-schema 0.7.6): the BigQuery table is partitioned by day (time_partitioning type DAY, on binding.location.partitionBy when it names one date or timestamp column, else by ingestion time) with expiration_ms = `retention` in days, so BigQuery deletes each partition that long after its day ends. BigQuery counts from the partition's date, not from when its rows were written: by ingestion time (no partitionBy, the default and the S3 rule's semantics) no row is deleted sooner than `retention` after it landed; with partitionBy, retention is the age of the date in that column, so a backfill of rows whose date is already older than `retention` lands in expired partitions and is deleted at once. It is never a whole-table expiration, which would delete the product. BigQuery cannot partition a table that exists, so adding `expire` to a live table replaces it: `fluid apply` refuses that plan without --allow-data-loss, and the next build lands the data again. A later change of `retention` is an in-place update. Removing `expire` from a live table removes the partitioning's replacement trigger, which `fluid apply` also refuses without --allow-data-loss; BigQuery cannot un-partition a table, so to keep data longer set a longer `retention` instead. `fluid verify` checks the live table's partition expiration." }, "deprecationPolicy": { "type": "object", diff --git a/pyproject.toml b/pyproject.toml index 73c53207..fa247854 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -251,6 +251,11 @@ gcp = [ "google-auth>=2.29", "google-auth-oauthlib>=1.2", "google-cloud-bigquery>=3.20", + # An embedded-SQL build on DuckDB reads an upstream BigQuery table through + # python-bigquery's Arrow pages (``RowIterator.to_arrow_iterable``) into + # Parquet (build_runners/_bigquery_read.py). python-bigquery itself only pulls + # pyarrow in through its ``pandas`` / ``bqstorage`` extras, so it is named here. + "pyarrow>=14", "google-cloud-storage>=2.16", # V1.5 — Dataplex metadata source catalog. Optional within the # gcp extra so users wanting BigQuery alone don't pay for the @@ -403,6 +408,11 @@ dev = [ # ``cli/__init__.py`` at import time) and that the tier-0 # ``fluid_build._net`` leaf module stays free of fluid_build upstreams. "import-linter>=2.0", + # The BigQuery read/load tests for embedded-SQL builds (tests/build_runners/ + # test_embedded_sql_bigquery.py) stream Arrow batches through a fake client; + # the unit lanes install ``.[dev,local]``, which has no pyarrow, so without + # it those tests would skip in CI instead of running. + "pyarrow>=14", ] # Light (in-process) integration-test emulators — Stage 1. moto mocks the diff --git a/tests/build_runners/test_consumes_resolution.py b/tests/build_runners/test_consumes_resolution.py index c4790ad2..c6437cca 100644 --- a/tests/build_runners/test_consumes_resolution.py +++ b/tests/build_runners/test_consumes_resolution.py @@ -326,11 +326,13 @@ def test_a_json_upstream_prefix_globs_the_files_the_runner_names_ndjson(tmp_path @pytest.mark.parametrize( "binding, platform", [ + # A BigQuery table is read (tests/build_runners/test_embedded_sql_bigquery.py); + # a GCS prefix of files is still not. ( { "platform": "gcp", - "format": "bigquery_table", - "location": {"project": "p", "dataset": "d", "table": "t"}, + "format": "parquet", + "location": {"bucket": "b", "path": "gs://b/bronze/cs/"}, }, "gcp", ), diff --git a/tests/build_runners/test_embedded_sql_bigquery.py b/tests/build_runners/test_embedded_sql_bigquery.py new file mode 100644 index 00000000..4a3a1c0f --- /dev/null +++ b/tests/build_runners/test_embedded_sql_bigquery.py @@ -0,0 +1,1042 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""An embedded-SQL build on DuckDB reads BigQuery upstreams and lands in BigQuery. + +Each of these was wrong on 0.16.5, measured against the goccy emulator: + +* a ``consumes[]`` entry whose upstream is a gcp ``bigquery_table`` was an + ``UnreadableBindingError``, so a silver or gold product could not build on + GCP at all; +* with the documented ``parameters.inputs`` escape hatch, the result was + written to a LOCAL file named ``gs:/...`` (the binding's staging path) and + the build reported success with 0 rows in the table; +* a GCS or other-cloud landing did the same; +* the load wrote DuckDB's naive ``TIMESTAMP`` as a Parquet timestamp with + ``isAdjustedToUTC=false``, which BigQuery reads as ``DATETIME``; +* against the emulator, the load failed ("loaded 0 rows") because its job + reports no ``outputRows``, and a client with no ADC could not start. + +The BigQuery client is faked through ``_bigquery_load._bigquery_module`` (the +seam the acquisition load tests use), holding each table as an Arrow table, so +these run with no cloud and no emulator. The emulator lane runs the same chain +for real: ``tests/providers/test_bigquery_emulated_embedded_sql_chain.py``. +""" + +from __future__ import annotations + +import io +import logging +import re +from pathlib import Path +from types import SimpleNamespace +from typing import Any, Dict, List, Optional + +import pytest +import yaml + +duckdb = pytest.importorskip("duckdb") +pa = pytest.importorskip("pyarrow") +pq = pytest.importorskip("pyarrow.parquet") + +from fluid_build.build_runners import _bigquery_load # noqa: E402 +from fluid_build.build_runners._embedded_sql_io import ( # noqa: E402 + ConsumesResolutionError, + EmbeddedSqlLandingError, + MaskingNotAppliedError, + plan_embedded_sql_io, + resolve_consumes, +) +from fluid_build.build_runners.base import _execute_embedded_sql_build # noqa: E402 + +BRONZE = "bronze.customer_subscriptions" +SILVER = "silver.subscription_status_summary" +BRONZE_TABLE = "northwind-demo.demo_bronze.customer_subscriptions" +SILVER_TABLE = "northwind-demo.demo_silver.subscription_status_summary" +SQL = ( + "SELECT product_id, status, COUNT(*) AS subscription_count " + "FROM subscriptions GROUP BY product_id, status" +) + + +# ── A BigQuery that keeps each table as an Arrow table ────────────────── + + +class _Field(SimpleNamespace): + pass + + +def _schema(*cols: tuple) -> List[_Field]: + return [_Field(name=n, field_type=t, mode="NULLABLE") for n, t in cols] + + +class FakeBigQuery: + """The names ``_bigquery_load`` / ``_bigquery_read`` use, over in-memory tables.""" + + def __init__(self, *, output_rows: Any = "exact", load_error: Optional[Exception] = None): + self.tables: Dict[str, Dict[str, Any]] = {} + self.loads: List[Dict[str, Any]] = [] + self.queries: List[str] = [] + self.clients: List[Dict[str, Any]] = [] + fake = self + + class LoadJobConfig: + def __init__(self, **kw): + self.__dict__.update(kw) + + class QueryJobConfig(LoadJobConfig): + pass + + class _Rows: + def __init__(self, table: Any, limit: Optional[int] = None): + self._table = table if limit is None else table.slice(0, limit) + + def to_arrow_iterable(self): + # One "page" per two rows, as tabledata.list pages. + yield from self._table.to_batches(max_chunksize=2) + + def to_arrow(self): + return self._table + + class _Job: + job_id = "job_1" + + def __init__(self, rows=None, values=None): + self.output_rows = rows + self._values = values + + def result(self, timeout=None): + if self._values is None: + return self + return [SimpleNamespace(values=lambda v=self._values: tuple(v))] + + class Client: + def __init__(self, project=None, credentials=None): + self.project = project or "adc-default" + fake.clients.append({"project": project, "credentials": credentials}) + + def get_table(self, table_id): + if table_id not in fake.tables: + raise LookupError(f"404 Not found: Table {table_id}") + t = fake.tables[table_id] + return SimpleNamespace(schema=t["schema"], num_rows=None, table_id=table_id) + + def list_rows(self, table, max_results=None): + return _Rows(fake.tables[table.table_id]["data"], max_results) + + def load_table_from_file(self, fh, table_id, job_config=None, location=None): + data = pq.read_table(io.BytesIO(fh.read())) + fake.loads.append( + {"table": table_id, "location": location, "config": dict(job_config.__dict__)} + ) + if load_error is not None: + raise load_error + entry = fake.tables[table_id] + if job_config.write_disposition == "WRITE_TRUNCATE" or not entry["data"].num_rows: + entry["data"] = data + else: + entry["data"] = pa.concat_tables([entry["data"], data]) + entry["uploaded"] = data + if output_rows == "exact": + return _Job(data.num_rows) + return _Job(output_rows if not callable(output_rows) else output_rows(data)) + + def query(self, sql, job_config=None, location=None): + fake.queries.append(sql) + m = re.search(r"FROM `([^`]+)`\.`([^`]+)`\.`([^`]+)`", sql) + table_id = ".".join(m.groups()) + return _Job(values=[fake.tables[table_id]["data"].num_rows]) + + self.module = SimpleNamespace( + Client=Client, + LoadJobConfig=LoadJobConfig, + QueryJobConfig=QueryJobConfig, + ScalarQueryParameter=lambda *a: a, + ) + + def create(self, table_id: str, schema: List[_Field], data: Optional[Any] = None) -> None: + if data is None: + data = pa.table({f.name: pa.array([], type=pa.string()) for f in schema}) + self.tables[table_id] = {"schema": schema, "data": data} + + +#: What the fake hands out for ``AnonymousCredentials``: the unit lanes install +#: ``.[dev,local]``, which has no google-auth, so the real class is not imported. +ANONYMOUS = SimpleNamespace(kind="anonymous") + + +@pytest.fixture +def bq(monkeypatch): + def install(**kw) -> FakeBigQuery: + fake = FakeBigQuery(**kw) + monkeypatch.setattr(_bigquery_load, "_bigquery_module", lambda: fake.module) + monkeypatch.setattr( + _bigquery_load, "_anonymous_credentials", lambda: ANONYMOUS, raising=False + ) + return fake + + return install + + +# ── The workspace: bronze in BigQuery, silver built from it ───────────── + + +def _dump(path: Path, doc: Dict[str, Any]) -> Path: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(yaml.safe_dump(doc, sort_keys=False), encoding="utf-8") + return path + + +def _gcp_binding(dataset: str, table: str, *, project: str = "northwind-demo") -> Dict[str, Any]: + # The shape make-targets writes: a gs:// staging path beside the table, + # which the embedded-SQL path used to "land" as a local file. + return { + "platform": "gcp", + "format": "bigquery_table", + "location": { + "path": f"gs://northwind-demo-lake/staging/{table}/", + "project": project, + "dataset": dataset, + "table": table, + "region": "europe-west1", + }, + } + + +def _workspace(root: Path, *, silver_binding: Optional[Dict[str, Any]] = None) -> Path: + root.mkdir(parents=True, exist_ok=True) + (root / "fluid.workspace.yaml").write_text("workspace: {name: t}\n", encoding="utf-8") + bronze_dir = root / "contracts" / "customer_subscriptions" + _dump( + bronze_dir / "contract.fluid.yaml", + { + "fluidVersion": "0.7.6", + "kind": "DataProduct", + "id": BRONZE, + "name": "Customer Subscriptions", + "exposes": [ + { + "exposeId": "subscriptions", + "kind": "table", + "binding": { + "platform": "local", + "format": "parquet", + "location": {"path": "out/customer_subscriptions.parquet"}, + }, + } + ], + }, + ) + _dump( + bronze_dir / "overlays" / "gcp.yaml", + {"exposes": [{"binding": _gcp_binding("demo_bronze", "customer_subscriptions")}]}, + ) + silver_dir = root / "contracts" / "subscription_status_summary" + _dump(silver_dir / "contract.fluid.yaml", _silver_contract()) + _dump( + silver_dir / "overlays" / "gcp.yaml", + { + "exposes": [ + { + "binding": silver_binding + or _gcp_binding("demo_silver", "subscription_status_summary") + } + ] + }, + ) + return root + + +def _silver_contract(sql: str = SQL) -> Dict[str, Any]: + return { + "fluidVersion": "0.7.6", + "kind": "DataProduct", + "id": SILVER, + "name": "Subscription Status Summary", + "metadata": {"layer": "Silver", "productType": "ADP"}, + "consumes": [{"productId": BRONZE, "exposeId": "subscriptions"}], + "builds": [ + { + "id": "summarize_subscription_status", + "pattern": "embedded-logic", + "engine": "duckdb", + "properties": {"sql": sql}, + "outputs": ["status_summary"], + } + ], + "exposes": [ + { + "exposeId": "status_summary", + "kind": "table", + "binding": { + "platform": "local", + "format": "parquet", + "location": {"path": "out/subscription_status_summary.parquet"}, + }, + } + ], + } + + +def _silver_dir(root: Path) -> Path: + return root / "contracts" / "subscription_status_summary" + + +def _load(root: Path, env: str = "gcp") -> Dict[str, Any]: + from fluid_build._contract_loader import load_contract_with_overlay + + return load_contract_with_overlay( + str(_silver_dir(root) / "contract.fluid.yaml"), env, logging.getLogger("t") + ) + + +_BRONZE_ROWS = { + "subscription_id": ["s1", "s2", "s3", "s4", "s5"], + "product_id": ["p1", "p1", "p1", "p2", "p2"], + "status": ["active", "active", "suspended", "active", "expired"], + "created_at": pa.array( + [1767323045000000 + i * 3_600_000_000 for i in range(5)], + type=pa.timestamp("us", tz="UTC"), + ), +} + + +def _seed(fake: FakeBigQuery) -> None: + fake.create( + BRONZE_TABLE, + _schema( + ("subscription_id", "STRING"), + ("product_id", "STRING"), + ("status", "STRING"), + ("created_at", "TIMESTAMP"), + ), + pa.table(_BRONZE_ROWS), + ) + fake.create( + SILVER_TABLE, + _schema(("product_id", "STRING"), ("status", "STRING"), ("subscription_count", "INTEGER")), + ) + + +def _gs_files(root: Path) -> List[Path]: + return [p for p in root.rglob("*") if p.name.startswith("gs:")] + + +def _build(contract: Dict[str, Any], root: Path) -> int: + return _execute_embedded_sql_build( + contract["builds"][0], contract, _silver_dir(root), env="gcp" + ) + + +# ── Reads ─────────────────────────────────────────────────────────────── + + +def test_a_bigquery_upstream_resolves_to_its_table_not_its_gs_path(bq, tmp_path): + bq() + root = _workspace(tmp_path / "ws") + contract = _load(root) + [r], covered, _ = resolve_consumes( + contract, contract["builds"][0], _silver_dir(root), env="gcp" + ) + assert not covered + assert (r.table, r.platform, r.region) == (BRONZE_TABLE, "gcp", "europe-west1") + assert r.uri == f"bigquery://{BRONZE_TABLE}" + assert r.view == "subscriptions" + + +def test_the_upstream_project_is_resolved_from_the_environment(bq, tmp_path, monkeypatch): + bq() + root = _workspace(tmp_path / "ws") + overlay = root / "contracts" / "customer_subscriptions" / "overlays" / "gcp.yaml" + _dump( + overlay, + { + "exposes": [ + { + "binding": _gcp_binding( + "demo_bronze", + "customer_subscriptions", + project="{{ env.FLUID_DEMO_GCP_PROJECT }}", + ) + } + ] + }, + ) + monkeypatch.setenv("FLUID_DEMO_GCP_PROJECT", "acme-eu-demo") + contract = _load(root) + [r], _, _ = resolve_consumes(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert r.table == "acme-eu-demo.demo_bronze.customer_subscriptions" + + monkeypatch.delenv("FLUID_DEMO_GCP_PROJECT") + with pytest.raises(ConsumesResolutionError, match="FLUID_DEMO_GCP_PROJECT"): + resolve_consumes(contract, contract["builds"][0], _silver_dir(root), env="gcp") + + +def test_the_sql_reads_the_bigquery_rows_and_lands_them_in_bigquery(bq, tmp_path): + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws") + rc = _build(_load(root), root) + assert rc == 0 + + landed = fake.tables[SILVER_TABLE]["data"].to_pylist() + got = sorted((r["product_id"], r["status"], r["subscription_count"]) for r in landed) + assert got == [ + ("p1", "active", 2), + ("p1", "suspended", 1), + ("p2", "active", 1), + ("p2", "expired", 1), + ] + [load] = fake.loads + assert load["table"] == SILVER_TABLE + assert load["location"] == "europe-west1" + assert load["config"]["write_disposition"] == "WRITE_TRUNCATE" + assert load["config"]["create_disposition"] == "CREATE_NEVER" + # Nothing was written to a local file named after the binding's gs:// path. + assert _gs_files(tmp_path) == [] + # The staged copy of bronze's table is removed after the build. + assert not list((_silver_dir(root) / ".fluid" / "staging").rglob("inputs/*.parquet")) + + +def test_the_sql_sees_a_bigquery_timestamp_as_its_utc_wall_clock(bq, tmp_path, monkeypatch): + """What the same SQL sees on local and aws, where bronze lands naive timestamps.""" + monkeypatch.setenv("TZ", "Asia/Tokyo") + fake = bq() + _seed(fake) + fake.create(SILVER_TABLE, _schema(("subscription_id", "STRING"), ("created", "STRING"))) + root = _workspace(tmp_path / "ws") + silver = _silver_dir(root) / "contract.fluid.yaml" + doc = yaml.safe_load(silver.read_text()) + doc["builds"][0]["properties"]["sql"] = ( + "SELECT subscription_id, CAST(created_at AS VARCHAR) AS created FROM subscriptions " + "WHERE subscription_id = 's1'" + ) + _dump(silver, doc) + assert _build(_load(root), root) == 0 + [row] = fake.tables[SILVER_TABLE]["data"].to_pylist() + assert row == {"subscription_id": "s1", "created": "2026-01-02 03:04:05"} + + +def test_a_missing_upstream_table_fails_the_build_naming_it(bq, tmp_path, capsys): + fake = bq() + fake.create(SILVER_TABLE, _schema(("product_id", "STRING"))) + root = _workspace(tmp_path / "ws") + assert _build(_load(root), root) == 1 + assert BRONZE_TABLE in " ".join(capsys.readouterr().out.split()) + assert fake.loads == [] + + +# ── Landing ───────────────────────────────────────────────────────────── + + +def test_a_bigquery_landing_is_planned_as_a_load_not_a_local_file(bq, tmp_path): + bq() + root = _workspace(tmp_path / "ws") + contract = _load(root) + io_plan = plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert io_plan.landing is None + assert io_plan.bigquery_landing is not None + assert io_plan.bigquery_landing.table_id == SILVER_TABLE + assert io_plan.bigquery_landing.location == "europe-west1" + + +def test_a_local_upstream_can_still_land_in_bigquery(bq, tmp_path): + """The local and gcp targets mix: a local bronze file, a BigQuery silver table.""" + fake = bq() + fake.create( + SILVER_TABLE, + _schema(("product_id", "STRING"), ("status", "STRING"), ("subscription_count", "INTEGER")), + ) + root = _workspace(tmp_path / "ws") + (root / "contracts" / "customer_subscriptions" / "overlays" / "gcp.yaml").unlink() + local = root / "contracts" / "customer_subscriptions" / "out" / "customer_subscriptions.parquet" + local.parent.mkdir(parents=True) + duckdb.sql( + "COPY (SELECT * FROM (VALUES ('p1','active'), ('p1','active')) t(product_id, status)) " + f"TO '{local}' (FORMAT parquet)" + ) + assert _build(_load(root), root) == 0 + assert fake.tables[SILVER_TABLE]["data"].to_pylist() == [ + {"product_id": "p1", "status": "active", "subscription_count": 2} + ] + + +@pytest.mark.parametrize( + "binding, fragment", + [ + ( + {"platform": "gcp", "format": "parquet", "location": {"path": "gs://lake/silver/s/"}}, + "gs://", + ), + ( + {"platform": "gcp", "format": "parquet", "location": {"bucket": "lake", "path": "s/"}}, + "gcp binding", + ), + ( + {"platform": "local", "format": "parquet", "location": {"path": "gs://lake/s.parquet"}}, + "gs://", + ), + ( + {"platform": "azure", "format": "parquet", "location": {"path": "abfss://c@a/s/"}}, + "abfss://", + ), + ], +) +def test_a_landing_this_path_cannot_write_is_refused_not_written_locally( + bq, tmp_path, binding, fragment, capsys +): + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws", silver_binding=binding) + contract = _load(root) + with pytest.raises(EmbeddedSqlLandingError, match=re.escape(fragment)): + plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert _build(contract, root) == 1 + assert _gs_files(tmp_path) == [] and not list(tmp_path.rglob("abfss:*")) + assert fake.loads == [] + + +def test_a_non_inline_sql_build_bound_to_bigquery_is_refused(bq, tmp_path): + bq() + root = _workspace(tmp_path / "ws") + contract = _load(root) + contract["builds"][0] = {"id": "multi_stage", "engine": "sql", "properties": {}} + assert _build(contract, root) == 1 + assert _gs_files(tmp_path) == [] + + +def test_masking_on_a_bigquery_expose_is_refused_before_any_read(bq, tmp_path): + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws") + contract = _load(root) + contract["exposes"][0]["policy"] = { + "privacy": {"masking": [{"column": "status", "strategy": "hash"}]} + } + with pytest.raises(MaskingNotAppliedError): + plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert _build(contract, root) == 1 + assert fake.loads == [] + + +def test_a_second_bigquery_output_is_refused(bq, tmp_path): + bq() + root = _workspace(tmp_path / "ws") + contract = _load(root) + contract["exposes"].append( + {"exposeId": "extra", "kind": "table", "binding": _gcp_binding("demo_silver", "extra")} + ) + contract["builds"][0]["outputs"] = ["status_summary", "extra"] + with pytest.raises(EmbeddedSqlLandingError, match="one (result|table)"): + plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + + +def test_a_failed_load_fails_the_build(bq, tmp_path, capsys): + fake = bq(load_error=RuntimeError("quota exceeded")) + _seed(fake) + root = _workspace(tmp_path / "ws") + assert _build(_load(root), root) == 1 + # The load was attempted (the build got past reading and the SQL), it + # failed loudly, and the table was left as it was. + [load] = fake.loads + assert load["table"] == SILVER_TABLE + out = " ".join(capsys.readouterr().out.split()) + assert f"BigQuery load into {SILVER_TABLE} failed" in out and "quota exceeded" in out + assert fake.tables[SILVER_TABLE]["data"].num_rows == 0 + + +def test_a_short_load_fails_the_build(bq, tmp_path, capsys): + fake = bq(output_rows=lambda data: data.num_rows - 1) + _seed(fake) + root = _workspace(tmp_path / "ws") + assert _build(_load(root), root) == 1 + [load] = fake.loads + assert load["table"] == SILVER_TABLE + out = " ".join(capsys.readouterr().out.split()) + assert f"loaded 3 rows into {SILVER_TABLE}, but the landed file holds 4" in out + + +# ── The load itself: timestamps, and a job with no row count ──────────── + + +def _naive_parquet(path: Path) -> Path: + duckdb.sql( + "COPY (SELECT 1 AS id, TIMESTAMP '2026-01-02 03:04:05' AS created_at, " + "TIMESTAMP '2026-01-02 03:04:05' AS local_time) " + f"TO '{path}' (FORMAT parquet)" + ) + return path + + +_TS_TABLE = "p.d.events" +_TS_TARGET = {"project": "p", "dataset": "d", "table": "events", "location": "EU"} + + +def _ts_fake(bq) -> FakeBigQuery: + fake = bq() + fake.create( + _TS_TABLE, + _schema(("id", "INTEGER"), ("created_at", "TIMESTAMP"), ("local_time", "DATETIME")), + ) + return fake + + +def _timestamp_logical(table: Any, column: str) -> Any: + return table.schema.field(column).type + + +def test_a_timestamp_column_is_loaded_utc_adjusted(bq, tmp_path): + """BigQuery reads a Parquet timestamp with isAdjustedToUTC=false as DATETIME.""" + fake = _ts_fake(bq) + staged = _naive_parquet(tmp_path / "events.parquet") + facts = _bigquery_load.load_file( + str(staged), + _TS_TARGET, + mode="full_refresh", + sink_format="parquet", + expected_rows=1, + logger=logging.getLogger("t"), + ) + assert facts["rows"] == 1 + uploaded = fake.tables[_TS_TABLE]["uploaded"] + # The TIMESTAMP column is UTC-adjusted; the DATETIME one stays naive. + assert _timestamp_logical(uploaded, "created_at") == pa.timestamp("us", tz="UTC") + assert _timestamp_logical(uploaded, "local_time") == pa.timestamp("us") + assert uploaded.column("created_at")[0].value == 1767323045000000 + # The build's own landed file is left as it was, and the copy is removed. + assert pq.read_schema(staged).field("created_at").type == pa.timestamp("us") + assert sorted(p.name for p in tmp_path.iterdir()) == ["events.parquet"] + + +def test_a_job_with_no_row_count_is_checked_by_counting_the_table(bq, tmp_path): + """The goccy emulator's load jobs carry no outputRows; that is not 0 rows loaded.""" + fake = bq(output_rows=None) + fake.create(_TS_TABLE, _schema(("id", "INTEGER"))) + staged = tmp_path / "events.parquet" + duckdb.sql(f"COPY (SELECT range AS id FROM range(3)) TO '{staged}' (FORMAT parquet)") + facts = _bigquery_load.load_file( + str(staged), + _TS_TARGET, + mode="full_refresh", + sink_format="parquet", + expected_rows=3, + logger=logging.getLogger("t"), + ) + assert (facts["rows"], facts["rows_from"]) == (3, "count_after_load") + assert any("COUNT(*)" in q for q in fake.queries) + + +def test_a_count_after_load_that_disagrees_fails(bq, tmp_path): + fake = bq(output_rows=None) + fake.create(_TS_TABLE, _schema(("id", "INTEGER"))) + staged = tmp_path / "events.parquet" + duckdb.sql(f"COPY (SELECT range AS id FROM range(3)) TO '{staged}' (FORMAT parquet)") + with pytest.raises(_bigquery_load.BigQueryLoadError, match="loaded 3 rows"): + _bigquery_load.load_file( + str(staged), + _TS_TARGET, + mode="full_refresh", + sink_format="parquet", + expected_rows=4, + logger=logging.getLogger("t"), + ) + + +def test_an_append_with_no_row_count_is_refused_off_an_emulator(bq, tmp_path, monkeypatch): + monkeypatch.delenv("BIGQUERY_EMULATOR_HOST", raising=False) + fake = bq(output_rows=None) + fake.create(_TS_TABLE, _schema(("id", "INTEGER"))) + staged = tmp_path / "events.parquet" + duckdb.sql(f"COPY (SELECT range AS id FROM range(3)) TO '{staged}' (FORMAT parquet)") + with pytest.raises(_bigquery_load.BigQueryLoadError, match="no output row count"): + _bigquery_load.load_file( + str(staged), + _TS_TARGET, + mode="incremental_append", + sink_format="parquet", + expected_rows=3, + logger=logging.getLogger("t"), + ) + + +def test_an_append_on_an_emulator_is_held_to_the_rows_it_added(bq, tmp_path, monkeypatch): + monkeypatch.setenv("BIGQUERY_EMULATOR_HOST", "http://127.0.0.1:9") + fake = bq(output_rows=None) + fake.create(_TS_TABLE, _schema(("id", "INTEGER")), pa.table({"id": [100, 101]})) + staged = tmp_path / "events.parquet" + duckdb.sql(f"COPY (SELECT range AS id FROM range(3)) TO '{staged}' (FORMAT parquet)") + facts = _bigquery_load.load_file( + str(staged), + _TS_TARGET, + mode="incremental_append", + sink_format="parquet", + expected_rows=3, + logger=logging.getLogger("t"), + ) + assert (facts["rows"], facts["rows_from"]) == (3, "count_after_load") + assert fake.tables[_TS_TABLE]["data"].num_rows == 5 + + +# ── The client: anonymous against an emulator, ADC otherwise ──────────── + + +def test_an_emulator_client_is_handed_anonymous_credentials(bq, monkeypatch): + monkeypatch.setenv("BIGQUERY_EMULATOR_HOST", "http://127.0.0.1:9") + fake = bq() + _bigquery_load.bigquery_client(fake.module, "forge-emulated") + assert fake.clients == [{"project": "forge-emulated", "credentials": ANONYMOUS}] + + +def test_an_emulator_client_sends_no_credentials(monkeypatch): + bigquery = pytest.importorskip("google.cloud.bigquery") + from google.auth.credentials import AnonymousCredentials + + monkeypatch.setenv("BIGQUERY_EMULATOR_HOST", "http://127.0.0.1:19999") + client = _bigquery_load.bigquery_client(bigquery, "forge-emulated") + assert isinstance(client._credentials, AnonymousCredentials) + assert client._connection.API_BASE_URL == "http://127.0.0.1:19999" + assert client.project == "forge-emulated" + + +def test_without_an_emulator_the_client_is_the_plain_adc_one(bq, monkeypatch): + monkeypatch.delenv("BIGQUERY_EMULATOR_HOST", raising=False) + fake = bq() + _bigquery_load.bigquery_client(fake.module, "acme-eu") + assert fake.clients == [{"project": "acme-eu", "credentials": None}] + + +def test_the_plan_prints_the_bigquery_read_and_load(bq, tmp_path, capsys): + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws") + assert _build(_load(root), root) == 0 + out = " ".join(capsys.readouterr().out.split()) # the console wraps long lines + assert f"bigquery://{BRONZE_TABLE}" in out + assert f"lands BigQuery table {SILVER_TABLE}" in out + assert f"read 5 row(s) from BigQuery table {BRONZE_TABLE}" in out + assert f"loaded 4 row(s) into BigQuery table {SILVER_TABLE}" in out + + +# ── Sovereignty: the locations the reads and the load actually use ───── +# +# PR review, measured on this branch before the fix: an EU-only silver whose +# gcp overlay named no region planned its read from europe-west1 and its load +# into "US" (the IaC default), and the build loaded there, rc 0. On main the +# same build stopped earlier (UnreadableBindingError), so the copy was new. + +_EU_ONLY = { + "jurisdiction": "EU", + "allowedRegions": ["eu-north-1", "eu-west-1", "europe-west1"], + "dataResidency": True, + "crossBorderTransfer": False, +} + + +def _sovereign(root: Path, sovereignty: Dict[str, Any]) -> Dict[str, Any]: + contract = _load(root) + contract["sovereignty"] = dict(sovereignty) + return contract + + +def _no_region(dataset: str, table: str) -> Dict[str, Any]: + binding = _gcp_binding(dataset, table) + del binding["location"]["region"] + return binding + + +def test_a_sovereign_landing_with_no_region_is_refused_before_any_read(bq, tmp_path): + fake = bq() + _seed(fake) + root = _workspace( + tmp_path / "ws", silver_binding=_no_region("demo_silver", "subscription_status_summary") + ) + contract = _sovereign(root, _EU_ONLY) + with pytest.raises(EmbeddedSqlLandingError, match="names no region") as err: + plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert err.value.code == "EmbeddedSqlSovereigntyError" + assert "not in allowedRegions" in err.value.why # US is not an allowed region either + assert _build(contract, root) == 1 + assert fake.loads == [] and fake.queries == [] + + +def test_a_sovereign_upstream_with_no_region_is_refused(bq, tmp_path): + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws") + overlay = root / "contracts" / "customer_subscriptions" / "overlays" / "gcp.yaml" + _dump(overlay, {"exposes": [{"binding": _no_region("demo_bronze", "customer_subscriptions")}]}) + contract = _sovereign(root, _EU_ONLY) + with pytest.raises(EmbeddedSqlLandingError) as err: + plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert err.value.code == "EmbeddedSqlSovereigntyError" + assert f"{BRONZE_TABLE} names no region" in err.value.why + # Read from US, landed in europe-west1, crossBorderTransfer false. + assert "crossBorderTransfer is false" in err.value.why + assert _build(contract, root) == 1 and fake.loads == [] + + +@pytest.mark.parametrize( + "region, fragment", + [ + ("us-central1", "not in allowedRegions"), + ("europe-west2", "is in UK, not the required jurisdiction EU"), + ], +) +def test_a_sovereign_landing_outside_the_contract_is_refused(bq, tmp_path, region, fragment): + bq() + binding = _gcp_binding("demo_silver", "subscription_status_summary") + binding["location"]["region"] = region + root = _workspace(tmp_path / "ws", silver_binding=binding) + sovereignty = dict(_EU_ONLY) + if region == "europe-west2": + sovereignty["allowedRegions"] = [*_EU_ONLY["allowedRegions"], "europe-west2"] + contract = _sovereign(root, sovereignty) + with pytest.raises(EmbeddedSqlLandingError) as err: + plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert err.value.code == "EmbeddedSqlSovereigntyError" + assert fragment in err.value.why + + +def test_an_eu_read_and_an_eu_load_meet_the_contract(bq, tmp_path, capsys): + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws") + contract = _sovereign(root, _EU_ONLY) + io_plan = plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert io_plan.warnings == [] + assert _build(contract, root) == 0 + assert [load["location"] for load in fake.loads] == ["europe-west1"] + assert "sovereignty" not in capsys.readouterr().out + + +def test_the_eu_multi_region_counts_as_eu(bq, tmp_path): + bq() + binding = _gcp_binding("demo_silver", "subscription_status_summary") + binding["location"]["region"] = "EU" + root = _workspace(tmp_path / "ws", silver_binding=binding) + contract = _sovereign(root, {**_EU_ONLY, "allowedRegions": ["EU", "europe-west1"]}) + io_plan = plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert io_plan.bigquery_landing.location == "EU" and io_plan.warnings == [] + + +def test_an_advisory_contract_warns_and_builds(bq, tmp_path, capsys): + fake = bq() + _seed(fake) + root = _workspace( + tmp_path / "ws", silver_binding=_no_region("demo_silver", "subscription_status_summary") + ) + contract = _sovereign(root, {**_EU_ONLY, "enforcementMode": "advisory"}) + io_plan = plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert any("names no region" in w for w in io_plan.warnings) + assert _build(contract, root) == 0 + out = " ".join(capsys.readouterr().out.split()) + assert "sovereignty: expose status_summary: BigQuery table" in out + + +def test_a_contract_without_sovereignty_is_unchanged(bq, tmp_path): + """No sovereignty block, no check: the default location stays the IaC's.""" + bq() + root = _workspace( + tmp_path / "ws", silver_binding=_no_region("demo_silver", "subscription_status_summary") + ) + contract = _load(root) + io_plan = plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert io_plan.bigquery_landing.location == "US" and io_plan.warnings == [] + + +def test_an_undeclared_landing_loads_where_the_table_is(bq, tmp_path): + """The plan reasons about the IaC's default location, but the load job + names none, so it runs where the table itself is, never a guessed US.""" + fake = bq() + _seed(fake) + root = _workspace( + tmp_path / "ws", silver_binding=_no_region("demo_silver", "subscription_status_summary") + ) + contract = _load(root) + assert _build(contract, root) == 0 + assert [load["location"] for load in fake.loads] == [None] + + +# ── A landing that is one of the build's own inputs ──────────────────── +# +# PR review, measured before the fix: silver's overlay named bronze's table, +# and the SQL kept the active rows. The build returned rc 0 and bronze went +# from 5 rows to 3: a WRITE_TRUNCATE of another product's table, which no +# --allow-data-loss gate sees. + + +@pytest.mark.parametrize( + "dataset, table, project", + [ + ("demo_bronze", "customer_subscriptions", "northwind-demo"), + ("Demo_Bronze", "Customer_Subscriptions", "NORTHWIND-DEMO"), + # A project left to the client may be the upstream's: refused, not assumed. + ("demo_bronze", "customer_subscriptions", ""), + ], +) +def test_a_landing_into_an_input_table_is_refused(bq, tmp_path, dataset, table, project): + fake = bq() + _seed(fake) + binding = _gcp_binding(dataset, table, project=project) + if not project: + del binding["location"]["project"] + root = _workspace(tmp_path / "ws", silver_binding=binding) + silver = _silver_dir(root) / "contract.fluid.yaml" + doc = yaml.safe_load(silver.read_text()) + doc["builds"][0]["properties"]["sql"] = "SELECT * FROM subscriptions WHERE status = 'active'" + _dump(silver, doc) + contract = _load(root) + with pytest.raises(EmbeddedSqlLandingError, match="which consumes bronze"): + plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert _build(contract, root) == 1 + assert fake.loads == [] + assert fake.tables[BRONZE_TABLE]["data"].num_rows == 5 + + +def test_a_landing_in_another_project_with_the_same_names_is_not_an_input(bq, tmp_path): + bq() + binding = _gcp_binding("demo_bronze", "customer_subscriptions", project="acme-silver") + root = _workspace(tmp_path / "ws", silver_binding=binding) + contract = _load(root) + io_plan = plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert io_plan.bigquery_landing.table_id == "acme-silver.demo_bronze.customer_subscriptions" + + +def test_an_s3_landing_inside_an_input_prefix_is_refused(tmp_path): + """The AWS form of the same mistake: the object would become rows of bronze's table.""" + aws = { + "platform": "aws", + "format": "parquet", + "location": {"bucket": "lake", "path": "bronze/customer_subscriptions/"}, + } + root = _workspace(tmp_path / "ws", silver_binding=aws) + for name in ("customer_subscriptions", "subscription_status_summary"): + (root / "contracts" / name / "overlays" / "gcp.yaml").rename( + root / "contracts" / name / "overlays" / "aws.yaml" + ) + _dump( + root / "contracts" / "customer_subscriptions" / "overlays" / "aws.yaml", + {"exposes": [{"binding": aws}]}, + ) + contract = _load(root, env="aws") + with pytest.raises(EmbeddedSqlLandingError, match="inside s3://lake/bronze/"): + plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="aws") + + +# ── Further outputs this path never lands ─────────────────────────────── +# +# PR review, measured before the fix: a second output bound to a GCS prefix +# planned without complaint, and the build returned rc 0 with nothing written +# for it. Only a second BigQuery output was refused. + + +@pytest.mark.parametrize( + "binding", + [ + { + "platform": "gcp", + "format": "parquet", + "location": {"bucket": "lake", "path": "gs://l/x/"}, + }, + {"platform": "aws", "format": "parquet", "location": {"bucket": "lake", "path": "x/"}}, + {"platform": "local", "format": "parquet", "location": {"path": "s3://lake/x/"}}, + _gcp_binding("demo_silver", "extra"), + ], +) +def test_a_further_remote_output_is_refused(bq, tmp_path, binding): + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws") + contract = _load(root) + contract["exposes"].append({"exposeId": "extra", "kind": "table", "binding": binding}) + contract["builds"][0]["outputs"] = ["status_summary", "extra"] + # A second BigQuery output was already refused ("loads its result into one + # table"); the others were planned and silently not landed. + with pytest.raises(EmbeddedSqlLandingError, match="one (result|table)"): + plan_embedded_sql_io(contract, contract["builds"][0], _silver_dir(root), env="gcp") + assert _build(contract, root) == 1 and fake.loads == [] + + +def test_a_further_local_output_is_warned_about_not_dropped_silently(bq, tmp_path, capsys): + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws") + contract = _load(root) + contract["exposes"].append( + { + "exposeId": "extra", + "kind": "table", + "binding": {"platform": "local", "format": "parquet", "location": {"path": "x.pq"}}, + } + ) + contract["builds"][0]["outputs"] = ["status_summary", "extra"] + assert _build(contract, root) == 0 + out = " ".join(capsys.readouterr().out.split()) + assert "extra is not written by this build" in out + assert [load["table"] for load in fake.loads] == [SILVER_TABLE] + + +# ── The run record fluid verify holds the table to ────────────────────── + + +def _records(root: Path) -> List[Dict[str, Any]]: + import json + + runs = _silver_dir(root) / ".fluid" / "runs" / SILVER / "summarize_subscription_status" / "runs" + return [json.loads(p.read_text()) for p in sorted(runs.glob("*.json"))] + + +def test_a_bigquery_landing_is_recorded_as_a_run(bq, tmp_path): + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws") + assert _build(_load(root), root) == 0 + [record] = _records(root) + assert record["state"] == "succeeded" and record["records_total"] == 4 + load = record["facets"]["bigquery_load"] + assert (load["table"], load["rows"]) == (SILVER_TABLE, 4) + assert record["facets"]["landed"]["mode"] == "full_refresh" + assert record["facets"]["landed"]["rows_from"] == "write" + assert fake.tables[SILVER_TABLE]["data"].num_rows == 4 + + +def test_a_failed_load_is_recorded_as_a_failed_run_without_a_count(bq, tmp_path): + fake = bq(load_error=RuntimeError("quota exceeded")) + _seed(fake) + root = _workspace(tmp_path / "ws") + assert _build(_load(root), root) == 1 + [record] = _records(root) + assert record["state"] == "failed" and "bigquery_load" not in record["facets"] + assert "quota exceeded" in record["error"] + + +def test_the_recorded_run_is_the_count_the_bigquery_verifier_holds_the_table_to(bq, tmp_path): + """Writer and reader agree: the run the build records is the one verify compares.""" + from fluid_build.cli._verify_athena import _landed_rows + from fluid_build.cli._verify_bigquery import _loaded_into + + fake = bq() + _seed(fake) + root = _workspace(tmp_path / "ws") + contract = _load(root) + assert _build(contract, root) == 0 + landed, info = _landed_rows( + contract, + "status_summary", + _silver_dir(root), + SILVER_TABLE, + wrote_into=_loaded_into, + embedded_sql_records=True, + ) + assert (landed, info["rule"], info["build_id"]) == (4, "equal", "summarize_subscription_status") diff --git a/tests/cli/test_apply_reports_to_command_center.py b/tests/cli/test_apply_reports_to_command_center.py new file mode 100644 index 00000000..96659967 --- /dev/null +++ b/tests/cli/test_apply_reports_to_command_center.py @@ -0,0 +1,599 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""``fluid apply`` reports each run to the Command Center, against a stub server. + +A run is registered at ``POST /api/v1/executions`` and closed at ``PATCH +/api/v1/executions/{id}``, the Command Center's executions API +(``app/api/v1/executions.py``: ``ExecutionCreate`` / ``ExecutionUpdate``), +with the credential and the organization ``fluid publish`` uses. Before this, +``get_reporter`` had no callers and a Jenkins apply left no run behind. + +The stub is a real HTTP server on loopback that records every request. The +apply goes through the real parser and the real OpenTofu engine; only the +``tofu`` binary is stubbed (init, plan and apply answer with the change +summary a real run prints), and the native planner is skipped. Pinned: what +a run says (product, contract version, environment, provider, resources, +counts, timings, status), that the organization comes from the publish +configuration (id or slug), that the credential is only ever a header, that +tofu's own output never reaches the Command Center, and that an outage, a +refusal or a 500 never changes the apply's exit code. +""" + +from __future__ import annotations + +import json +import logging +import socket +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from typing import Any, Dict, Iterator, List + +import pytest +import yaml + +from fluid_build.cli import _apply_opentofu_engine as engine +from fluid_build.cli._common import CLIError +from fluid_build.iac import runner + +pytestmark = pytest.mark.unit + +_LOG = logging.getLogger("test.apply_cc_report") +_KEY = "cc-test-key-not-a-secret" # pragma: allowlist secret +_TOFU_OUTPUT_MARKER = "tofu-printed-this-attribute-value" +_ORG = "0f5e3c1a-telco" + + +class _Recorder: + def __init__(self) -> None: + self.requests: List[Dict[str, Any]] = [] + self.status = {"POST": 201, "PATCH": 200, "GET": 200} + + +@pytest.fixture +def cc() -> Iterator[Any]: + """A stub Command Center on loopback: ``(base_url, recorder)``.""" + recorder = _Recorder() + + class Handler(BaseHTTPRequestHandler): + def _answer(self, method: str) -> None: + length = int(self.headers.get("Content-Length") or 0) + raw = self.rfile.read(length) if length else b"" + recorder.requests.append( + { + "method": method, + "path": self.path, + "headers": {k: v for k, v in self.headers.items()}, + "raw": raw.decode("utf-8"), + "body": json.loads(raw) if raw else None, + } + ) + status = recorder.status[method] + if method == "GET" and self.path == "/api/v1/organizations": + payload: Any = [{"id": _ORG, "slug": "northwind-telco", "name": "Telco"}] + else: + payload = {"ok": status < 400} + data = json.dumps(payload).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(data))) + self.end_headers() + self.wfile.write(data) + + def do_POST(self): # noqa: N802 + self._answer("POST") + + def do_PATCH(self): # noqa: N802 + self._answer("PATCH") + + def do_GET(self): # noqa: N802 + self._answer("GET") + + def log_message(self, *_a): + pass + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + yield f"http://127.0.0.1:{server.server_address[1]}", recorder + finally: + server.shutdown() + server.server_close() + + +_CONTRACT = { + "fluidVersion": "0.7.5", + "kind": "DataProduct", + "id": "bronze.customer_subscriptions", + "name": "Customer Subscriptions", + "version": "1.4.0", + "description": "Run report fixture.", + "domain": "Customer", + "metadata": {"layer": "Bronze", "owner": {"team": "data-platform", "email": "dp@example.com"}}, + "exposes": [ + { + "exposeId": "subscriptions", + "kind": "table", + "binding": { + "platform": "local", + "format": "parquet", + "location": {"path": "data/customer_subscriptions.parquet"}, + }, + "contract": {"schema": [{"name": "subscription_id", "type": "STRING"}]}, + } + ], +} + +_GCP_OVERLAY = { + "exposes": [ + { + "binding": { + "platform": "gcp", + "format": "bigquery_table", + "location": { + "project": "northwind-demo", + "dataset": "demo_bronze", + "table": "customer_subscriptions", + "region": "europe-west1", + }, + } + } + ] +} + + +@pytest.fixture +def product(tmp_path: Path, monkeypatch) -> Path: + for var in ( + "FLUID_CC_ENDPOINT", + "FLUID_CATALOG_FLUID_CC_URL", + "FLUID_API_KEY", + "FLUID_BEARER_TOKEN", + "FLUID_CC_ORG_ID", + "FLUID_COMMAND_CENTER_URL", + "FLUID_COMMAND_CENTER_API_KEY", + "FLUID_COMMAND_CENTER_ENABLED", + "FLUID_STATE_BACKEND", + "FLUID_PROVIDER", + "JENKINS_URL", + "BUILD_TAG", + "GOOGLE_APPLICATION_CREDENTIALS", + ): + monkeypatch.delenv(var, raising=False) + monkeypatch.setenv("HOME", str(tmp_path / "home")) + monkeypatch.chdir(tmp_path) + (tmp_path / "overlays").mkdir() + (tmp_path / "overlays" / "gcp.yaml").write_text(yaml.safe_dump(_GCP_OVERLAY), encoding="utf-8") + contract = tmp_path / "contract.fluid.yaml" + contract.write_text(yaml.safe_dump(_CONTRACT, sort_keys=False), encoding="utf-8") + return contract + + +def _summary(add: int) -> List[Dict[str, Any]]: + return [{"type": "change_summary", "changes": {"add": add, "change": 0, "remove": 0}}] + + +@pytest.fixture +def tofu(monkeypatch) -> Dict[str, Any]: + """The ``tofu`` binary, answered: init ok, plan +2, apply +2.""" + behaviour: Dict[str, Any] = {"apply_ok": True} + ok = runner.TofuResult + monkeypatch.setattr(runner, "tofu_path", lambda: "/usr/bin/tofu") + monkeypatch.setattr(runner, "require_tofu_version", lambda *a, **k: None) + monkeypatch.setattr(runner, "tofu_init", lambda *a, **k: ok("init", 0, "", "")) + monkeypatch.setattr(runner, "tofu_state_list", lambda *a, **k: []) + monkeypatch.setattr(runner, "tofu_state_resources", lambda *a, **k: []) + monkeypatch.setattr(runner, "tofu_prior_state_resources", lambda *a, **k: []) + monkeypatch.setattr( + runner, "tofu_import", lambda *a, **k: ok("import", 1, "", "not found (stub)") + ) + monkeypatch.setattr( + runner, "tofu_plan", lambda *a, **k: ok("plan", 0, "", "", events=_summary(2)) + ) + + def _apply(*_a, **_k): + if behaviour["apply_ok"]: + return ok("apply", 0, "", "", events=_summary(2)) + return ok("apply", 1, "", f"Error: googleapi: 403 {_TOFU_OUTPUT_MARKER}") + + monkeypatch.setattr(runner, "tofu_apply", _apply) + monkeypatch.setattr(engine, "native_actions", lambda contract, logger: []) + return behaviour + + +def _apply(contract: Path, *extra: str) -> int: + from fluid_build.cli import build_parser + + args = build_parser().parse_args( + ["apply", str(contract), "--env", "gcp", "--yes", "--no-verify-federation", *extra] + ) + return args.func(args, _LOG) + + +def _by(recorder: _Recorder, method: str) -> List[Dict[str, Any]]: + return [r for r in recorder.requests if r["method"] == method] + + +def _configure(monkeypatch, url: str, *, org: bool = True) -> None: + monkeypatch.setenv("FLUID_CC_ENDPOINT", url) + monkeypatch.setenv("FLUID_API_KEY", _KEY) + if org: + monkeypatch.setenv("FLUID_CC_ORG_ID", _ORG) + + +def test_an_apply_is_registered_and_closed_with_what_it_did(product, tofu, cc, monkeypatch, capsys): + url, recorder = cc + _configure(monkeypatch, url) + + assert _apply(product) == 0 + + (post,) = _by(recorder, "POST") + (patch,) = _by(recorder, "PATCH") + assert post["path"] == "/api/v1/executions" + body = post["body"] + assert body["command"] == "apply" + assert body["status"] == "running" + assert body["provider"] == "gcp" + assert body["environment"] == "gcp" + assert body["runner"] == "cli" + meta = body["metadata"] + assert meta["product_id"] == "bronze.customer_subscriptions" + assert meta["contract_version"] == "1.4.0" + assert meta["fluid_version"] == "0.7.5" + assert meta["platform"] == "gcp" + assert meta["environment"] == "gcp" + assert len(meta["contract_hash"]) == 64 + assert meta["state"].startswith("local: ") + + assert patch["path"] == f"/api/v1/executions/{body['execution_id']}" + update = patch["body"] + assert update["status"] == "success" + result = update["result"] + assert result["planned_changes"] == {"add": 2, "change": 0, "remove": 0} + assert result["applied_changes"] == {"add": 2, "change": 0, "remove": 0} + assert result["dry_run"] is False + assert result["exit_code"] == 0 + assert result["duration_seconds"] >= 0 + assert result["started_at"] <= result["finished_at"] + assert "google_bigquery_dataset.bronze_customer_subscriptions_demo_bronze" in ( + result["resources"] + ) + assert "google_bigquery_table.bronze_customer_subscriptions_customer_subscriptions" in ( + result["resources"] + ) + + # The organization and the credential travel as headers, and only there. + for request in (post, patch): + assert request["headers"]["X-API-Key"] == _KEY + assert request["headers"]["X-Organization-Id"] == _ORG + assert _KEY not in request["raw"] + assert f"command center: run {body['execution_id']} reported" in capsys.readouterr().out + + +def test_the_organization_named_by_slug_in_fluid_config_is_resolved(product, tofu, cc, monkeypatch): + """The demo lab names each product's organization by slug in its own + fluid.config.yaml and never sets FLUID_CC_ORG_ID.""" + url, recorder = cc + _configure(monkeypatch, url, org=False) + (product.parent / "fluid.config.yaml").write_text( + yaml.safe_dump({"catalogs": {"fluid-command-center": {"organization": "northwind-telco"}}}), + encoding="utf-8", + ) + + assert _apply(product) == 0 + + assert [r["path"] for r in _by(recorder, "GET")] == ["/api/v1/organizations"] + for request in _by(recorder, "POST") + _by(recorder, "PATCH"): + assert request["headers"]["X-Organization-Id"] == _ORG + + +def test_a_failed_apply_is_closed_failed_and_tofu_output_stays_home(product, tofu, cc, monkeypatch): + url, recorder = cc + _configure(monkeypatch, url) + tofu["apply_ok"] = False + + with pytest.raises(CLIError) as exc: + _apply(product) + assert exc.value.event == "opentofu_apply_failed" + + (patch,) = _by(recorder, "PATCH") + assert patch["body"]["status"] == "failed" + assert patch["body"]["result"]["error_event"] == "opentofu_apply_failed" + assert patch["body"]["result"]["applied_changes"] is None + for request in recorder.requests: + assert _TOFU_OUTPUT_MARKER not in request["raw"] + + +@pytest.mark.parametrize("status", [500, 401]) +def test_a_command_center_that_refuses_never_changes_the_exit_code( + product, tofu, cc, monkeypatch, capsys, status +): + url, recorder = cc + _configure(monkeypatch, url) + recorder.status.update(POST=status, PATCH=status) + + assert _apply(product) == 0 + assert "not fully reported" in capsys.readouterr().out + + +def test_a_command_center_that_is_down_never_changes_the_exit_code( + product, tofu, monkeypatch, capsys +): + with socket.socket() as probe: + probe.bind(("127.0.0.1", 0)) + closed = probe.getsockname()[1] + _configure(monkeypatch, f"http://127.0.0.1:{closed}") + monkeypatch.setenv("FLUID_COMMAND_CENTER_TIMEOUT", "1") + + assert _apply(product) == 0 + assert "not fully reported" in capsys.readouterr().out + + +def test_nothing_is_sent_or_said_when_no_command_center_is_configured(product, tofu, cc, capsys): + _url, recorder = cc + assert _apply(product) == 0 + assert recorder.requests == [] + assert "command center" not in capsys.readouterr().out + + +def test_the_opt_out_sends_nothing(product, tofu, cc, monkeypatch): + url, recorder = cc + _configure(monkeypatch, url) + monkeypatch.setenv("FLUID_COMMAND_CENTER_ENABLED", "false") + + assert _apply(product) == 0 + assert recorder.requests == [] + + +def test_a_private_address_is_refused_before_the_credential_is_sent( + product, tofu, monkeypatch, capsys +): + from fluid_build.providers.catalogs.fluid_cc import FluidCommandCenterProvider + + called: List[str] = [] + monkeypatch.setattr( + FluidCommandCenterProvider, + "_list_organizations", + lambda self: called.append("organizations"), + ) + _configure(monkeypatch, "http://10.20.30.40:5200", org=False) + + assert _apply(product) == 0 + assert called == [] + assert "private or metadata address" in " ".join(capsys.readouterr().out.split()) + + +# ── A run refused before the engine registered it still names its product ── + + +def test_a_sovereignty_refusal_is_reported_with_its_product(product, tofu, cc, monkeypatch): + """The gcp overlay loses its region under a strict policy: the emitter + refuses before the run is registered, and the run still says whose it is.""" + from fluid_build._errors import SovereigntyViolationError + + url, recorder = cc + _configure(monkeypatch, url) + doc = yaml.safe_load(product.read_text(encoding="utf-8")) + doc["sovereignty"] = { + "jurisdiction": "EU", + "allowedRegions": ["europe-west1"], + "enforcementMode": "strict", + } + product.write_text(yaml.safe_dump(doc, sort_keys=False), encoding="utf-8") + overlay = product.parent / "overlays" / "gcp.yaml" + placed = yaml.safe_load(overlay.read_text(encoding="utf-8")) + placed["exposes"][0]["binding"]["location"].pop("region") + overlay.write_text(yaml.safe_dump(placed), encoding="utf-8") + + with pytest.raises(SovereigntyViolationError): + _apply(product) + + (post,) = _by(recorder, "POST") + (patch,) = _by(recorder, "PATCH") + assert post["body"]["provider"] == "gcp" + assert post["body"]["environment"] == "gcp" + metadata = post["body"]["metadata"] + assert metadata["product_id"] == "bronze.customer_subscriptions" + assert metadata["contract_version"] == "1.4.0" + assert metadata["platform"] == "gcp" + assert metadata["contract_hash"] + assert patch["body"]["status"] == "failed" + assert patch["body"]["result"]["error_event"] == "SovereigntyViolationError" + + +def test_a_run_refused_before_the_contract_loads_names_the_base_product( + product, tofu, cc, monkeypatch +): + """``--env prod`` that the workspace expects, with no prod overlay: the + loader refuses, and the run is still attached to the product.""" + url, recorder = cc + _configure(monkeypatch, url) + (product.parent / "fluid.workspace.yaml").write_text( + yaml.safe_dump( + { + "workspace": {"name": "cc"}, + "expected-environments": {"bronze.customer_subscriptions": ["gcp", "prod"]}, + } + ), + encoding="utf-8", + ) + from fluid_build import loader + from fluid_build.cli import build_parser + + loader._NOTED_MISSING_OVERLAYS.clear() + args = build_parser().parse_args( + ["apply", str(product), "--env", "prod", "--yes", "--no-verify-federation"] + ) + with pytest.raises(CLIError) as exc: + args.func(args, _LOG) + assert exc.value.event == "overlay_declared_but_missing" + + (post,) = _by(recorder, "POST") + (patch,) = _by(recorder, "PATCH") + metadata = post["body"]["metadata"] + assert metadata["product_id"] == "bronze.customer_subscriptions" + assert metadata["contract_version"] == "1.4.0" + assert post["body"]["environment"] == "prod" + # No platform (no overlay settled one) and no hash of a contract never compiled. + assert "contract_hash" not in metadata + assert post["body"]["provider"] is None + assert patch["body"]["result"]["error_event"] == "overlay_declared_but_missing" + + +def test_a_provider_that_cannot_be_resolved_is_reported_with_its_product( + product, tofu, cc, monkeypatch +): + """``--provider aws`` against the gcp overlay: refused while the apply + picks its engine, before the engine runs; the base names the product.""" + url, recorder = cc + _configure(monkeypatch, url) + + with pytest.raises(CLIError) as exc: + _apply(product, "--provider", "aws") + assert exc.value.event == "generate_iac_provider_mismatch" + + (post,) = _by(recorder, "POST") + metadata = post["body"]["metadata"] + assert metadata["product_id"] == "bronze.customer_subscriptions" + assert metadata["contract_version"] == "1.4.0" + assert "platform" not in metadata + assert "contract_hash" not in metadata + assert post["body"]["environment"] == "gcp" + + +# ── A build-augmented apply reports its builds ──────────────────────────── +# +# Measured against a stub Command Center on the integration branch: `fluid +# apply --mode amend-and-build` on a silver product whose build loaded 28 rows +# into BigQuery was closed with an infra-only result (planned and applied +# changes, resources) in phase "apply", and a bronze run whose load failed was +# closed "failed" with no error event and no error message. + +_BUILD_ID = "summarise_subscriptions" +_LOADED_TABLE = "northwind-demo.demo_bronze.customer_subscriptions" + + +def _with_build(product: Path) -> Path: + doc = yaml.safe_load(product.read_text(encoding="utf-8")) + doc["builds"] = [ + { + "id": _BUILD_ID, + "pattern": "embedded-logic", + "engine": "sql", + "properties": {"sql": "SELECT 1 AS subscription_id"}, + } + ] + product.write_text(yaml.safe_dump(doc, sort_keys=False), encoding="utf-8") + return product + + +@pytest.fixture +def build(monkeypatch) -> Dict[str, Any]: + """The embedded-SQL build, answered: it writes the run record a BigQuery load writes.""" + from fluid_build.build_runners import base + + behaviour: Dict[str, Any] = {"rc": 0, "calls": 0} + + def _execute(build, contract, contract_dir, **_kwargs): + behaviour["calls"] += 1 + runs = Path(contract_dir) / ".fluid" / "runs" / contract["id"] / build["id"] / "runs" + runs.mkdir(parents=True, exist_ok=True) + ok = behaviour["rc"] == 0 + record: Dict[str, Any] = { + "run_id": f"01RUN{behaviour['calls']:08d}", + "state": "succeeded" if ok else "failed", + "records_total": 28 if ok else 0, + "facets": {"engine": "duckdb", "pattern": "embedded-logic"}, + } + if ok: + record["facets"]["bigquery_load"] = {"table": _LOADED_TABLE, "rows": 28} + record["facets"]["landed"] = { + "mode": "full_refresh", + "rows_from": "write", + "destinations": {"subscriptions": f"bigquery://{_LOADED_TABLE}"}, + } + (runs / f"{record['run_id']}.json").write_text(json.dumps(record), encoding="utf-8") + return behaviour["rc"] + + monkeypatch.setattr(base, "_execute_embedded_sql_build", _execute) + return behaviour + + +def test_a_build_augmented_apply_reports_each_build_and_what_it_landed( + product, tofu, build, cc, monkeypatch +): + url, recorder = cc + _configure(monkeypatch, url) + _with_build(product) + + assert _apply(product, "--mode", "amend-and-build") == 0 + assert build["calls"] == 1 + + (post,) = _by(recorder, "POST") + (patch,) = _by(recorder, "PATCH") + assert post["body"]["metadata"]["mode"] == "amend-and-build" + update = patch["body"] + assert update["status"] == "success" + assert update["current_phase"] == "build" + assert update.get("error_message") is None + result = update["result"] + assert result["applied_changes"] == {"add": 2, "change": 0, "remove": 0} + assert result["builds"] == [ + { + "build_id": _BUILD_ID, + "status": "succeeded", + "run_id": "01RUN00000001", + "table": _LOADED_TABLE, + "rows": 28, + "destinations": {"subscriptions": f"bigquery://{_LOADED_TABLE}"}, + } + ] + assert "error_event" not in result + + +def test_a_failed_build_is_reported_with_its_build_id(product, tofu, build, cc, monkeypatch): + url, recorder = cc + _configure(monkeypatch, url) + _with_build(product) + build["rc"] = 1 + + assert _apply(product, "--mode", "amend-and-build") == 1 + + (patch,) = _by(recorder, "PATCH") + update = patch["body"] + assert update["status"] == "failed" + assert update["current_phase"] == "build" + assert update["error_message"] == f"fluid apply failed: build_failed:{_BUILD_ID}" + result = update["result"] + assert result["error_event"] == f"build_failed:{_BUILD_ID}" + assert result["exit_code"] == 1 + (reported,) = result["builds"] + assert reported == {"build_id": _BUILD_ID, "status": "failed", "run_id": "01RUN00000001"} + + +def test_a_build_id_that_is_not_in_the_contract_is_the_reason( + product, tofu, build, cc, monkeypatch +): + url, recorder = cc + _configure(monkeypatch, url) + _with_build(product) + + assert _apply(product, "--mode", "amend-and-build", "--build-id", "no_such_build") == 1 + + (patch,) = _by(recorder, "PATCH") + assert patch["body"]["result"]["error_event"] == "build_not_found:no_such_build" + assert patch["body"]["current_phase"] == "build" + assert build["calls"] == 0 diff --git a/tests/cli/test_diff_state_drift.py b/tests/cli/test_diff_state_drift.py index 91374471..137b8a9d 100644 --- a/tests/cli/test_diff_state_drift.py +++ b/tests/cli/test_diff_state_drift.py @@ -151,8 +151,17 @@ def __init__(self, monkeypatch, *, plan_rc: int = 0, doc: Optional[Dict] = None) from fluid_build.cli import _apply_opentofu_engine as engine monkeypatch.setattr(engine, "native_actions", lambda contract, logger: []) + # Moving a pre-provider-key state (``iac.state_migration``) is not + # under test here: nothing to move. + from fluid_build.iac.state_migration import CURRENT, StateReconciliation - def _init(self, workdir, *, backend=True, env=None): + monkeypatch.setattr( + engine, + "_reconcile_state", + lambda **kw: StateReconciliation(CURRENT, kw["legacy"], kw["current"]), + ) + + def _init(self, workdir, *, backend=True, env=None, reconfigure=False, force_copy=False): self.calls.append(f"init backend={backend}") self.module_during_run = (Path(workdir) / "main.tf.json").read_text(encoding="utf-8") return runner.TofuResult("init", 0, "", "") @@ -351,7 +360,7 @@ def test_the_state_backend_is_resolved_as_apply_resolves_it(workspace, monkeypat assert _invoke(["diff", str(contract), "--out", str(out)]) == (0, None) assert tofu.calls[0] == "init backend=True" assert '"s3"' in tofu.module_during_run - assert _state(out)["state"] == f"remote: s3://team-state/fluid/{CID}/terraform.tfstate" + assert _state(out)["state"] == f"remote: s3://team-state/fluid/{CID}/aws/terraform.tfstate" # The module the pass wrote into a fresh workdir is not left there. assert not (_workdir(workspace) / "main.tf.json").exists() diff --git a/tests/cli/test_generate_schedule_fluid_apply.py b/tests/cli/test_generate_schedule_fluid_apply.py index 3d34ece1..a6f5077d 100644 --- a/tests/cli/test_generate_schedule_fluid_apply.py +++ b/tests/cli/test_generate_schedule_fluid_apply.py @@ -419,6 +419,10 @@ def _literal_connection(contract: str = DEMO_CONTRACT) -> str: #: Variables the ``fluid apply`` code path reads that a scheduled run does not #: need, so the worker environment does not pass them. NOT_PASSED = { + "JENKINS_URL": ( + "only tags a Command Center run report as a Jenkins run; a scheduled run is not one" + ), + "ProgramData": "the Windows system config directory; the DAG task runs under bash", "PRODUCTION": "only chooses a --safe-mode tip in the CLI banner", "SOURCE_DATE_EPOCH": "tar mtimes for `fluid bundle`, which a scheduled apply never runs", "USERPROFILE": "the Windows home directory; the DAG task runs under bash", diff --git a/tests/cli/test_one_contract_two_clouds_pipeline.py b/tests/cli/test_one_contract_two_clouds_pipeline.py new file mode 100644 index 00000000..b13950f5 --- /dev/null +++ b/tests/cli/test_one_contract_two_clouds_pipeline.py @@ -0,0 +1,288 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""One contract, two clouds, one pipeline generator: the small collisions. + +* Stage 11: the aws and the gcp DAG of one product had the same ``dag_id`` + (``__``) and the same schedule directory, so in one + Airflow the second sync replaced (and ``--delete-scope product`` deleted) + the first. The env is now in both. +* Stage 6: the generated plan stage did not run ``--check-sovereignty``, for + any CI system, so a strict sovereignty violation first failed at apply. +* ``--env gcp`` with no gcp overlay applied the local base unchanged with a + warning (measured: silver validate and plan rc 0). When the workspace's + ``expected-environments`` (which fluid-demo-env keeps) declares gcp for the + product it is now an error. The contract's own ``environments`` block only + warns: forge-cli applies nothing from it. +""" + +from __future__ import annotations + +import argparse +import logging +from pathlib import Path + +import pytest +import yaml + +from fluid_build.schedulers.airflow import fluid_apply +from tests.cli._schedule_dag_fixtures import DEMO_CONTRACT, DEMO_CONTRACT_PATH, load_dag + +pytestmark = pytest.mark.unit + +_LOG = logging.getLogger("test.one_contract_two_clouds") +_PRODUCT = "bronze.customer_subscriptions" + + +# ── Stage 11: the DAG id and the schedule directory name the env ───────── + + +def _dags(env): + contract = yaml.safe_load(DEMO_CONTRACT) + return fluid_apply.render_fluid_apply_dags(contract, env=env, contract_path=DEMO_CONTRACT_PATH) + + +def test_the_aws_and_the_gcp_dag_of_one_product_have_different_ids(monkeypatch): + ids = {} + for env in ("aws", "gcp"): + (source,) = _dags(env).values() + ids[env] = load_dag(source, monkeypatch).dag["dag_id"] + assert ids == { + "aws": f"{_PRODUCT}__aws__ingest_subscriptions", + "gcp": f"{_PRODUCT}__gcp__ingest_subscriptions", + } + + +def test_a_dag_with_no_env_keeps_its_id(monkeypatch): + (source,) = _dags(None).values() + assert load_dag(source, monkeypatch).dag["dag_id"] == f"{_PRODUCT}__ingest_subscriptions" + + +def test_each_env_gets_its_own_schedule_directory(tmp_path, monkeypatch): + """schedule-sync --delete-scope product mirrors each directory with delete: + one directory per env is what keeps the aws sync off the gcp DAG.""" + from fluid_build.cli import generate_artifacts + from tests.cli._schedule_dag_fixtures import write_project + + monkeypatch.delenv("FLUID_ENV", raising=False) + write_project(tmp_path) + contract_dir = (tmp_path / DEMO_CONTRACT_PATH).parent + (contract_dir / "overlays" / "gcp.yaml").write_text( + yaml.safe_dump( + { + "exposes": [ + { + "binding": { + "platform": "gcp", + "format": "bigquery_table", + "location": { + "project": "p", + "dataset": "d", + "table": "t", + "region": "europe-west1", + }, + } + } + ] + } + ), + encoding="utf-8", + ) + monkeypatch.chdir(tmp_path) + for env in ("aws", "gcp"): + parser = argparse.ArgumentParser() + generate_artifacts.register_subcommand(parser.add_subparsers()) + argv = ["artifacts", DEMO_CONTRACT_PATH, "--out", f"dist-{env}", "--env", env] + assert generate_artifacts.run(parser.parse_args(argv), _LOG) == 0 + scopes = sorted(p.name for p in (tmp_path / f"dist-{env}" / "schedule").iterdir()) + assert scopes == [f"{_PRODUCT}__{env}"] + + +# ── Stage 6: every generated plan stage checks sovereignty ──────────────── + + +def _strings(node): + if isinstance(node, str): + yield node + elif isinstance(node, dict): + for value in node.values(): + yield from _strings(value) + elif isinstance(node, list): + for value in node: + yield from _strings(value) + + +def _plan_commands(provider, complexity): + """Every rendered command that runs ``fluid plan`` (Jenkins: stage 6's body).""" + from fluid_build.forge.core.pipeline_templates import PipelineConfig, PipelineTemplateGenerator + + files = PipelineTemplateGenerator().generate_pipeline( + PipelineConfig(provider=provider, complexity=complexity) + ) + found = [] + for text in files.values(): + if provider.value == "jenkins": + start = text.index("stage('6 - plan')") + found.append(text[start : text.index("stage('7", start)]) + continue + for doc in yaml.safe_load_all(text): + found.extend(s for s in _strings(doc) if "fluid plan" in s) + return found + + +def _cases(): + from fluid_build.forge.core.pipeline_templates import PipelineComplexity, PipelineProvider + + return [(p, c) for p in PipelineProvider for c in PipelineComplexity] + + +@pytest.mark.parametrize( + "provider, complexity", _cases(), ids=lambda v: getattr(v, "value", str(v)) +) +def test_every_generated_plan_stage_checks_sovereignty(provider, complexity): + commands = _plan_commands(provider, complexity) + if provider.value == "tekton" and complexity.value == "basic": + # The basic Tekton pipeline references a `fluid-plan` Task it does not + # render; there is no plan command in it to carry the flag. + assert commands == [] + return + assert commands, "no plan command rendered" + for command in commands: + assert "--check-sovereignty" in command, command + + +# ── --env for a declared environment with no overlay is refused ─────────── + +_SILVER = { + "fluidVersion": "0.7.5", + "kind": "DataProduct", + "id": "silver.subscription_status_summary", + "name": "Subscription Status Summary", + "description": "Declared-env fixture.", + "domain": "Customer", + "metadata": {"layer": "Silver", "owner": {"team": "data-platform", "email": "dp@example.com"}}, + "exposes": [ + { + "exposeId": "summary", + "kind": "table", + "binding": { + "platform": "local", + "format": "parquet", + "location": {"path": "o.parquet"}, + }, + "contract": {"schema": [{"name": "status", "type": "STRING"}]}, + } + ], +} + + +@pytest.fixture +def silver(tmp_path, monkeypatch): + """fluid-demo-env's layout: a workspace declaring each product's targets, + and a product with an aws overlay but (yet) no gcp one.""" + from fluid_build import loader + + loader._NOTED_MISSING_OVERLAYS.clear() + (tmp_path / "fluid.workspace.yaml").write_text( + yaml.safe_dump( + { + "workspace": {"name": "demo"}, + "expected-environments": {"subscription_status_summary": ["local", "aws", "gcp"]}, + } + ), + encoding="utf-8", + ) + product = tmp_path / "contracts" / "subscription_status_summary" + (product / "overlays").mkdir(parents=True) + (product / "overlays" / "aws.yaml").write_text( + yaml.safe_dump({"exposes": [{"binding": {"platform": "aws", "format": "parquet"}}]}), + encoding="utf-8", + ) + contract = product / "contract.fluid.yaml" + contract.write_text(yaml.safe_dump(_SILVER, sort_keys=False), encoding="utf-8") + monkeypatch.chdir(product) + return contract + + +def test_a_declared_env_with_no_overlay_is_refused(silver): + from fluid_build._contract_loader import CLIError + from fluid_build.loader import load_with_overlay + + with pytest.raises(CLIError) as exc: + load_with_overlay(silver, "gcp") + assert exc.value.event == "overlay_declared_but_missing" + assert exc.value.context["declared_by"] == ( + "fluid.workspace.yaml expected-environments (subscription_status_summary)" + ) + assert "bound to local" in str(exc.value) + + +def test_validate_plan_and_bundle_all_refuse_it(silver, tmp_path): + from fluid_build.cli import main + + assert main(["validate", str(silver), "--env", "gcp"]) == 1 + assert main(["plan", str(silver), "--env", "gcp", "--out", str(tmp_path / "p.json")]) == 1 + assert not (tmp_path / "p.json").exists() + assert main(["bundle", str(silver), "--env", "gcp", "--out", str(tmp_path / "b.tgz")]) != 0 + + +def test_the_env_the_base_is_bound_to_and_dev_are_the_base(silver): + from fluid_build.loader import load_with_overlay + + assert load_with_overlay(silver, "local")["exposes"][0]["binding"]["platform"] == "local" + assert load_with_overlay(silver, "dev")["exposes"][0]["binding"]["platform"] == "local" + assert load_with_overlay(silver, "aws")["exposes"][0]["binding"]["platform"] == "aws" + + +def test_an_undeclared_env_still_warns_and_says_what_the_base_binds_to(silver, caplog): + from fluid_build.loader import load_with_overlay + + caplog.set_level(logging.WARNING, logger="fluid.loader") + load_with_overlay(silver, "prod") + (record,) = [r for r in caplog.records if "overlay_not_found" in r.getMessage()] + assert "it binds to local, not to 'prod'" in record.getMessage() + + +def test_the_contract_s_own_environments_block_warns_and_does_not_refuse(silver, caplog): + """A schema-valid ``environments`` block keeps validating, as on 0.16.5. + + forge-cli applies nothing from that block, so refusing on it offered one + fix only: deleting a valid declaration. Only the workspace's + ``expected-environments`` refuses; the block is named in the warning. + """ + from fluid_build.loader import load_with_overlay + + doc = dict(_SILVER, environments={"staging": {}, "prod": {}}) + silver.write_text(yaml.safe_dump(doc, sort_keys=False), encoding="utf-8") + caplog.set_level(logging.WARNING, logger="fluid.loader") + base = load_with_overlay(silver, "staging") + assert base["exposes"][0]["binding"]["platform"] == "local" + (record,) = [r for r in caplog.records if "overlay_not_found" in r.getMessage()] + assert "environments block names 'staging'" in record.getMessage() + assert record.declared_in_environments_block is True + + +def test_validate_env_named_only_by_the_environments_block_passes(tmp_path, monkeypatch): + """No workspace, an ``environments`` block, no overlays: rc 0, with a warning.""" + from fluid_build import loader + from fluid_build.cli import main + + loader._NOTED_MISSING_OVERLAYS.clear() + path = tmp_path / "contract.fluid.yaml" + path.write_text( + yaml.safe_dump(dict(_SILVER, environments={"staging": {}, "prod": {}}), sort_keys=False), + encoding="utf-8", + ) + monkeypatch.chdir(tmp_path) + assert main(["validate", str(path), "--env", "prod"]) == 0 diff --git a/tests/cli/test_policy_apply_provider_message.py b/tests/cli/test_policy_apply_provider_message.py new file mode 100644 index 00000000..9d2954d6 --- /dev/null +++ b/tests/cli/test_policy_apply_provider_message.py @@ -0,0 +1,139 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Stage 8 (``fluid policy-apply``) survives a provider result that says ``message``. + +``policy_apply.run`` logs the provider's result as the event payload, +``info(logger, "policy_apply_result", **res)``. The GCP provider's result +carries a ``message`` key, which collided with ``info()``'s own ``message`` +parameter: ``TypeError: info() got multiple values for argument 'message'``, +exit 1, on every generated gcp pipeline (measured on 0.16.5 for all three +demo products). The bindings file below is the one ``fluid policy-compile`` +wrote for the demo's bronze gcp overlay. +""" + +from __future__ import annotations + +import argparse +import json +import logging +from pathlib import Path + +import pytest + +from fluid_build.cli import _logging +from fluid_build.cli.policy_apply import run + +pytestmark = pytest.mark.unit + +_GCP_BINDINGS = { + "bindings": [ + { + "dataset": "demo_bronze", + "principal": "group:data-platform@northwind.example", + "project": "northwind-demo", + "provider": "gcp", + "resource_id": "northwind-demo.demo_bronze", + "resource_type": "bigquery.dataset", + "roles": ["roles/bigquery.dataViewer"], + } + ], + "warnings": [], +} + + +def _args(path: Path, mode: str = "enforce") -> argparse.Namespace: + return argparse.Namespace(bindings=str(path), mode=mode, provider=None, project=None) + + +@pytest.fixture +def events(caplog): + caplog.set_level(logging.DEBUG, logger="test.policy_apply_message") + return caplog + + +def _payloads(caplog, name: str): + out = [] + for record in caplog.records: + try: + doc = json.loads(record.getMessage()) + except ValueError: + continue + if doc.get("message") == name: + out.append(doc) + return out + + +@pytest.mark.parametrize("mode", ["check", "enforce"]) +def test_the_gcp_provider_result_is_logged_not_a_type_error(tmp_path, events, monkeypatch, mode): + for var in ("GOOGLE_CLOUD_PROJECT", "FLUID_PROJECT", "GCLOUD_PROJECT", "FLUID_PROVIDER"): + monkeypatch.delenv(var, raising=False) + bindings = tmp_path / "bindings.json" + bindings.write_text(json.dumps(_GCP_BINDINGS), encoding="utf-8") + + rc = run(_args(bindings, mode), logging.getLogger("test.policy_apply_message")) + + assert rc == 0 + (event,) = _payloads(events, "policy_apply_result") + # The event keeps its name; the provider's own sentence is kept, renamed. + assert event["status"] == "ok" + assert "declaratively" in event["extra_message"] + assert event["bindings"] == 1 + + +class _Talkative: + """A provider whose result names every envelope key.""" + + name = "talkative" + + def apply_policy(self, data, mode="check"): + return { + "status": "ok", + "message": "provider sentence", + "time": "provider time", + "level": "provider level", + "name": "provider name", + } + + +def test_any_provider_result_that_names_an_envelope_key_is_kept(tmp_path, events, monkeypatch): + from fluid_build.cli import policy_apply + + monkeypatch.setattr(policy_apply, "build_provider", lambda *a, **k: _Talkative()) + bindings = tmp_path / "bindings.json" + bindings.write_text( + json.dumps({"bindings": [{"provider": "talkative", "roles": ["r"]}]}), encoding="utf-8" + ) + + assert run(_args(bindings), logging.getLogger("test.policy_apply_message")) == 0 + + (event,) = _payloads(events, "policy_apply_result") + assert event["level"] == "INFO" + assert event["name"] == "fluid.cli" + assert event["extra_message"] == "provider sentence" + assert event["extra_time"] == "provider time" + assert event["extra_level"] == "provider level" + assert event["extra_name"] == "provider name" + + +@pytest.mark.parametrize("helper", ["info", "warn", "error"]) +def test_every_structured_helper_takes_a_message_key(helper, caplog): + caplog.set_level(logging.DEBUG, logger="test.policy_apply_message.helpers") + log = logging.getLogger("test.policy_apply_message.helpers") + getattr(_logging, helper)(log, "the_event", message="payload", logger="also payload") + (record,) = caplog.records + doc = json.loads(record.getMessage()) + assert doc["message"] == "the_event" + assert doc["extra_message"] == "payload" + assert doc["logger"] == "also payload" diff --git a/tests/cli/test_schedule_sync_delete_scope.py b/tests/cli/test_schedule_sync_delete_scope.py index ac243113..e989c099 100644 --- a/tests/cli/test_schedule_sync_delete_scope.py +++ b/tests/cli/test_schedule_sync_delete_scope.py @@ -277,3 +277,237 @@ def test_default_sync_keeps_every_other_products_dags(self, tmp_path: Path) -> N ) assert schedule_sync.run(args) == 0 assert sorted(p.name for p in root.iterdir()) == ["new.product", "orders"] + + +# ── The DAG an env's directory replaced ─────────────────────────────────── +# +# forge-cli 0.16.7 and earlier synced a product's DAGs to ``/`` with +# dag id ``__``, whatever the env. An env's DAGs now live in +# ``__/`` as ``____``, and ``--delete-scope +# product`` mirrors only that directory. Measured on the demo's bronze product: +# after the first sync on the new release the destination held both DAGs, each +# ``FLUID_ENV_NAME = 'aws'`` and ``SCHEDULE = '0 */4 * * *'``, so Airflow ran two +# ``fluid apply --env aws`` of one product against one state at the same minute. + +_PRODUCT = "bronze.customer_subscriptions" +_BUILD = "ingest_subscriptions" + + +def _dag(env: str, *, legacy: bool = False, build: str = _BUILD) -> str: + from fluid_build.schedulers.airflow import fluid_apply + + text = fluid_apply.render_dag( + product_id=_PRODUCT, + build=fluid_apply.ScheduledBuild( + build_id=build, schedule="0 */4 * * *", timezone="UTC", retries=1 + ), + env=env or None, + contract_path="contracts/customer_subscriptions/contract.fluid.yaml", + env_names=[], + ) + if legacy and env: + # What 0.16.6 rendered: the same file, with no env in its dag id. + new_id = fluid_apply.py_str_literal(f"{_PRODUCT}__{env}__{build}") + assert text.count(new_id) == 1 + text = text.replace(new_id, fluid_apply.py_str_literal(f"{_PRODUCT}__{build}")) + return text + + +def _env_artifacts(tmp_path: Path, env: str = "aws") -> Path: + """``schedule/`` as stage 3 now writes it for ``--env ``.""" + dags = tmp_path / "dist" / "artifacts" / "schedule" + scope = dags / f"{_PRODUCT}__{env}" + scope.mkdir(parents=True) + (scope / f"{_BUILD}_dag.py").write_text(_dag(env), encoding="utf-8") + return dags + + +def _upgraded_root(tmp_path: Path) -> Path: + """The lab's DAG root after a sync by 0.16.6, plus what must survive the retirement.""" + root = tmp_path / "airflow-dags" + old = root / _PRODUCT + old.mkdir(parents=True) + (old / f"{_BUILD}_dag.py").write_text(_dag("aws", legacy=True), encoding="utf-8") + (old / "removed_build_dag.py").write_text( + _dag("aws", legacy=True, build="removed_build"), encoding="utf-8" + ) + # Not this env's, not a legacy id, not a DAG: all stay. + (old / "gcp_only_dag.py").write_text( + _dag("gcp", legacy=True, build="gcp_only"), encoding="utf-8" + ) + (old / "envless_dag.py").write_text(_dag("", build="envless"), encoding="utf-8") + (old / "notes.py").write_text("# someone's helper\n", encoding="utf-8") + (root / "billing").mkdir() + (root / "billing" / "billing_dag.py").write_text("# theirs\n", encoding="utf-8") + return root + + +def _unwrapped(text: str) -> str: + """The console wraps long lines; compare without whitespace.""" + return "".join(text.split()) + + +def _dag_ids(root: Path) -> List[str]: + return sorted( + facts["dag_id"] + for path in root.rglob("*.py") + if (facts := schedule_sync._dag_facts(path)) is not None + ) + + +class TestTheDagAnEnvDirectoryReplaced: + def test_the_scope_and_the_old_dags_are_read_from_the_dag_files(self, tmp_path: Path) -> None: + dags = _env_artifacts(tmp_path) + assert schedule_sync._replaced_scopes(dags) == [(f"{_PRODUCT}__aws", _PRODUCT, "aws")] + root = _upgraded_root(tmp_path) + assert schedule_sync._superseded_dags(root / _PRODUCT, _PRODUCT, "aws") == [ + f"{_BUILD}_dag.py", + "removed_build_dag.py", + ] + # A directory of the old layout, or of anything else, replaces nothing. + assert schedule_sync._replaced_scopes(root) == [] + + def test_the_dry_run_plans_the_retirement_after_the_sync(self, tmp_path: Path) -> None: + dags = _env_artifacts(tmp_path) + root = _upgraded_root(tmp_path) + argvs = _dispatch(dags, destination=str(root)) + dest = f"{root.resolve()}/" + assert argvs[0] == [ + "/bin/rsync", + "-av", + "--delete", + "--", + f"{dags}/{_PRODUCT}__aws/", + f"{dest}{_PRODUCT}__aws/", + ] + retire = argvs[1] + assert retire[:3] == ["/bin/rsync", "-rv", "--delete"] + assert retire[3:6] == [ + f"--include=/{_BUILD}_dag.py", + "--include=/removed_build_dag.py", + "--exclude=*", + ] + assert retire[-1] == f"{dest}{_PRODUCT}/" + assert len(argvs) == 2 + assert (root / _PRODUCT / f"{_BUILD}_dag.py").exists(), "a dry run deleted" + + @pytest.mark.skipif(shutil.which("rsync") is None, reason="rsync not installed") + def test_the_first_sync_after_the_upgrade_retires_the_old_dag(self, tmp_path: Path) -> None: + dags = _env_artifacts(tmp_path) + root = _upgraded_root(tmp_path) + report = tmp_path / "report.json" + mode = (root / _PRODUCT).stat().st_mode + + args = _args( + dags_dir=str(dags), destination=str(root), dry_run=False, env="aws", report=str(report) + ) + assert schedule_sync.run(args) == 0 + + assert _dag_ids(root) == sorted( + [ + f"{_PRODUCT}__aws__{_BUILD}", + f"{_PRODUCT}__envless", + f"{_PRODUCT}__gcp_only", + ] + ) + assert (root / _PRODUCT / "notes.py").exists() + assert (root / "billing" / "billing_dag.py").exists() + assert (root / _PRODUCT).stat().st_mode == mode + import json + + recorded = json.loads(report.read_text(encoding="utf-8"))["superseded_scopes"] + assert recorded == [ + { + "scope": f"{_PRODUCT}__aws", + "replaces": _PRODUCT, + "env": "aws", + "old_dags_retired": True, + } + ] + + # The next sync finds nothing left to retire. + assert len(_dispatch(dags, destination=str(root))) == 1 + + def test_none_scope_retires_nothing_and_says_what_is_left( + self, tmp_path: Path, capsys: pytest.CaptureFixture[str] + ) -> None: + dags = _env_artifacts(tmp_path) + root = _upgraded_root(tmp_path) + args = _args(dags_dir=str(dags), destination=str(root), delete_scope="none") + with patch.object(schedule_sync, "_which_or_raise", side_effect=_which): + assert schedule_sync.run(args) == 0 + out = _unwrapped(capsys.readouterr().out) + assert "--include=/" not in out + assert _unwrapped(f"Delete {_PRODUCT}/'s DAG files for env aws") in out + + @pytest.mark.parametrize( + "scheduler,destination", + [ + ("airflow", "s3://b/dags/"), + ("airflow", "gs://b/dags/"), + ("airflow", "ssh://u@host/opt/dags"), + ("mwaa", "s3://mwaa/dags/"), + ], + ) + def test_a_destination_that_cannot_be_read_here_gets_the_step_to_take( + self, tmp_path: Path, scheduler: str, destination: str, capsys + ) -> None: + dags = _env_artifacts(tmp_path) + report = tmp_path / "report.json" + args = _args( + dags_dir=str(dags), scheduler=scheduler, destination=destination, report=str(report) + ) + with patch.object(schedule_sync, "_which_or_raise", side_effect=_which): + assert schedule_sync.run(args) == 0 + out = _unwrapped(capsys.readouterr().out) + assert "--include=/" not in out + assert _unwrapped(f"{_PRODUCT}__ beside {_PRODUCT}__aws__") in out + assert '"old_dags_retired": false' in report.read_text(encoding="utf-8") + + def test_mirroring_the_whole_destination_needs_no_note(self, tmp_path: Path, capsys) -> None: + dags = _env_artifacts(tmp_path) + args = _args(dags_dir=str(dags), destination="s3://b/dags/", delete_scope="destination") + with patch.object(schedule_sync, "_which_or_raise", side_effect=_which): + assert schedule_sync.run(args) == 0 + assert "note:" not in capsys.readouterr().out + + def test_a_git_ssh_destination_retires_in_its_clone_before_the_commit( + self, tmp_path: Path + ) -> None: + dags = _env_artifacts(tmp_path) + upgraded = _upgraded_root(tmp_path) + args = _args( + dags_dir=str(dags), + destination="git+ssh://git@example.com/org/dags.git", + dry_run=False, + ) + cwd_calls: List[List[str]] = [] + + def _result(argv: List[str]) -> Dict[str, Any]: + return {"argv": argv, "exit_code": 0, "stdout_tail": " M x", "stderr_tail": ""} + + def _clone(argv: List[str], **_kwargs: Any) -> Dict[str, Any]: + # The clone holds what the last sync by 0.16.6 pushed. + shutil.copytree(upgraded, argv[-1]) + return _result(argv) + + def _in_clone(argv: List[str], *, cwd: str, **_kwargs: Any) -> Dict[str, Any]: + cwd_calls.append(argv) + return _result(argv) + + with ( + patch.object(schedule_sync, "_which_or_raise", side_effect=_which), + patch.object(schedule_sync, "_run_subprocess", side_effect=_clone), + patch.object(schedule_sync, "_run_subprocess_with_cwd", side_effect=_in_clone), + ): + schedule_sync._airflow_dispatch(dags, args) + + assert [argv[:2] for argv in cwd_calls[:3]] == [ + ["/bin/rsync", "-av"], + ["/bin/rsync", "-rv"], + ["/bin/git", "add"], + ] + retire = cwd_calls[1] + assert retire[-1] == f"./{_PRODUCT}/" + assert f"--include=/{_BUILD}_dag.py" in retire + assert "--include=/envless_dag.py" not in retire diff --git a/tests/cli/test_verify_bigquery_dimensions.py b/tests/cli/test_verify_bigquery_dimensions.py new file mode 100644 index 00000000..d3874efd --- /dev/null +++ b/tests/cli/test_verify_bigquery_dimensions.py @@ -0,0 +1,453 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""``fluid verify`` on a BigQuery table counts it, and checks its masked columns. + +On 0.16.5 the BigQuery verifier read only the table's metadata: a seeded +cleartext ``msisdn`` in a column the contract lands hashed passed ``--strict``, +an empty table passed, and on the goccy emulator (``numRows`` unset) the +console crashed formatting ``None`` with ``:,``. These pin the two dimensions +the Glue + Athena verifier already has (``row_count``, ``masking``), computed +from one GoogleSQL query. + +The fake client runs that query for real: the GoogleSQL is rewritten into +DuckDB's dialect (backticks, ``COUNTIF``, ``REGEXP_CONTAINS`` against the bound +parameter, both RE2) and executed over the table's rows, so a quoting or +pattern mistake fails here rather than in BigQuery. +""" + +from __future__ import annotations + +import argparse +import json +import logging +import re +from pathlib import Path +from types import SimpleNamespace +from typing import Any, Dict, List, Optional +from unittest.mock import patch + +import pytest +import yaml + +duckdb = pytest.importorskip("duckdb") + +from fluid_build.build_runners import _bigquery_load # noqa: E402 +from fluid_build.cli.verify import run, verify_bigquery_table # noqa: E402 + +_LOG = logging.getLogger("test") +PRODUCT = "bronze.customer_subscriptions" +TABLE_ID = "northwind-demo.demo_bronze.customer_subscriptions" +HASHED = "a" * 64 + + +class _Field(SimpleNamespace): + pass + + +_SCHEMA = [ + _Field(name="subscription_id", field_type="STRING", mode="REQUIRED"), + _Field(name="msisdn", field_type="STRING", mode="NULLABLE"), +] + + +class FakeBigQuery: + """A table's metadata, and a query engine for the verifier's count query.""" + + def __init__( + self, + rows: List[tuple], + *, + query_error: Optional[Exception] = None, + adc_project: Optional[str] = "adc-project", + ): + self.rows = rows + self.sql: List[str] = [] + self.params: List[Any] = [] + self.tables_read: List[str] = [] + fake = self + + class QueryJobConfig: + def __init__(self, **kw): + self.__dict__.update(kw) + + class _Job: + job_id = "q_1" + + def __init__(self, values): + self._values = values + + def result(self, timeout=None): + return [SimpleNamespace(values=lambda: tuple(self._values))] + + class Client: + def __init__(self, project=None, credentials=None): + # As google-cloud-core: only None falls back to the default + # (ADC's); an empty string is kept as the project. + self.project = adc_project if project is None else project + + def get_table(self, table_id): + fake.tables_read.append(table_id) + if table_id.startswith("."): + # python-bigquery on a table id with no project. + raise ValueError("Could not determine project ID") + return SimpleNamespace(schema=_SCHEMA, num_rows=None, created=None, modified=None) + + def get_dataset(self, ref): + return SimpleNamespace(location="europe-west1") + + def query(self, sql, job_config=None, location=None): + if query_error is not None: + raise query_error + fake.sql.append(sql) + params = list(getattr(job_config, "query_parameters", []) or []) + fake.params.append(params) + return _Job(fake._run(sql, params)) + + self.module = SimpleNamespace( + Client=Client, + QueryJobConfig=QueryJobConfig, + ScalarQueryParameter=lambda name, kind, value: (name, kind, value), + ) + + def _run(self, sql: str, params: List[tuple]) -> tuple: + """The GoogleSQL count query, run by DuckDB over ``rows``.""" + duck = re.sub(r"FROM `[^`]+`\.`[^`]+`\.`[^`]+`", "FROM t", sql) + duck = duck.replace("`", '"').replace("COUNTIF(", "count_if(") + duck = duck.replace("REGEXP_CONTAINS(", "regexp_matches(").replace( + " AS STRING)", " AS VARCHAR)" + ) + values = {name: value for name, _kind, value in params} + duck = re.sub(r"@(\w+)", lambda m: "'" + values[m.group(1)].replace("'", "''") + "'", duck) + con = duckdb.connect() + con.execute("CREATE TABLE t (subscription_id VARCHAR, msisdn VARCHAR)") + if self.rows: + con.executemany("INSERT INTO t VALUES (?, ?)", self.rows) + return con.execute(duck).fetchone() + + +@pytest.fixture +def bq(monkeypatch): + def install(rows, **kw) -> FakeBigQuery: + fake = FakeBigQuery(rows, **kw) + monkeypatch.setattr(_bigquery_load, "_bigquery_module", lambda: fake.module) + return fake + + return install + + +def _expose(project: str = "northwind-demo") -> Dict[str, Any]: + return { + "exposeId": "subscriptions", + "kind": "table", + "policy": {"privacy": {"masking": [{"column": "msisdn", "strategy": "hash"}]}}, + "binding": { + "platform": "gcp", + "format": "bigquery_table", + "location": { + "project": project, + "dataset": "demo_bronze", + "table": "customer_subscriptions", + "region": "europe-west1", + }, + }, + "contract": { + "schema": [ + {"name": "subscription_id", "type": "VARCHAR", "required": True}, + {"name": "msisdn", "type": "VARCHAR"}, + ] + }, + } + + +def _contract(expose: Optional[Dict[str, Any]] = None) -> Dict[str, Any]: + return { + "fluidVersion": "0.7.6", + "kind": "DataProduct", + "id": PRODUCT, + "name": "Customer Subscriptions", + "builds": [ + { + "id": "ingest_subscriptions", + "pattern": "acquisition", + "engine": "duckdb", + "properties": {"source": {"kind": "postgres", "mode": "full_refresh"}}, + "outputs": ["subscriptions"], + } + ], + "exposes": [expose or _expose()], + } + + +def _verify(tmp_path: Path, contract: Optional[Dict[str, Any]] = None) -> Dict[str, Any]: + contract = contract or _contract() + return verify_bigquery_table( + "northwind-demo", + "demo_bronze", + "customer_subscriptions", + contract["exposes"][0]["contract"]["schema"], + "europe-west1", + expose=contract["exposes"][0], + contract=contract, + workdir=tmp_path, + ) + + +def _run_record( + tmp_path: Path, + run_id: str, + *, + rows: int, + table: Optional[str] = TABLE_ID, + state: str = "succeeded", + build_id: str = "ingest_subscriptions", +) -> None: + facets: Dict[str, Any] = { + "landed": {"mode": "full_refresh", "rows_from": "write", "destinations": {}} + } + if table is not None: + facets["bigquery_load"] = {"table": table, "rows": rows, "job_id": "j"} + path = tmp_path / ".fluid" / "runs" / PRODUCT / build_id / "runs" + path.mkdir(parents=True, exist_ok=True) + (path / f"{run_id}.json").write_text( + json.dumps({"run_id": run_id, "state": state, "records_total": rows, "facets": facets}), + encoding="utf-8", + ) + + +# ── Masking ───────────────────────────────────────────────────────────── + + +def test_treated_values_pass_the_masking_dimension(bq, tmp_path): + bq([("s1", HASHED), ("s2", HASHED), ("s3", None)]) + result = _verify(tmp_path) + masking = result["dimensions"]["masking"] + assert masking["status"] == "pass" + assert masking["columns"][0]["non_null"] == 2 + assert result["status"] == "match" + assert result["severity"]["level"] == "SUCCESS" + + +def test_a_seeded_cleartext_value_fails_masking_as_critical(bq, tmp_path): + bq([("s1", HASHED), ("s2", "+46701234567"), ("s3", HASHED)]) + result = _verify(tmp_path) + masking = result["dimensions"]["masking"] + assert masking["status"] == "fail" + assert masking["columns"][0]["offending"] == 1 + assert "+46701234567" not in json.dumps(result) # counts only, never the value + assert result["status"] == "mismatch" + assert result["severity"]["level"] == "CRITICAL" + + +def test_a_hash_with_a_trailing_newline_is_not_a_hash(bq, tmp_path): + """The pattern is anchored with \\A and \\z: REGEXP_CONTAINS finds, it does not match.""" + bq([("s1", HASHED + "\n"), ("s2", "x" + HASHED)]) + assert _verify(tmp_path)["dimensions"]["masking"]["columns"][0]["offending"] == 2 + + +def test_the_query_quotes_with_backticks_and_binds_the_pattern(bq, tmp_path): + fake = bq([("s1", HASHED)]) + _verify(tmp_path) + [sql] = fake.sql + # A double-quoted name is a string literal in GoogleSQL: count("msisdn") counts rows. + assert '"msisdn"' not in sql and "count(`msisdn`)" in sql + assert "FROM `northwind-demo`.`demo_bronze`.`customer_subscriptions`" in sql + [[(name, kind, value)]] = fake.params + assert (kind, value) == ("STRING", r"\A(?:[0-9a-f]{64})\z") + assert f"@{name}" in sql + + +# ── Row count ─────────────────────────────────────────────────────────── + + +def test_the_count_equal_to_the_last_load_passes(bq, tmp_path): + bq([("s1", HASHED), ("s2", HASHED)]) + _run_record(tmp_path, "20260928T000001Z", rows=2) + rc = _verify(tmp_path)["dimensions"]["row_count"] + assert (rc["status"], rc["actual"], rc["expected"]) == ("pass", 2, 2) + assert rc["compared_with"]["rule"] == "equal" + + +def test_a_count_below_the_last_load_fails_as_critical(bq, tmp_path): + bq([("s1", HASHED)]) + _run_record(tmp_path, "20260928T000001Z", rows=2) + result = _verify(tmp_path) + assert result["dimensions"]["row_count"]["status"] == "fail" + assert "BigQuery counted 1 rows" in result["dimensions"]["row_count"]["message"] + assert result["severity"]["level"] == "CRITICAL" + + +def test_a_newer_local_run_of_the_same_build_is_passed_over(bq, tmp_path): + bq([("s1", HASHED), ("s2", HASHED)]) + _run_record(tmp_path, "20260928T000001Z", rows=2) + _run_record(tmp_path, "20260928T000002Z", rows=7, table=None) # a local run + rc = _verify(tmp_path)["dimensions"]["row_count"] + assert (rc["status"], rc["expected"]) == ("pass", 2) + assert rc["compared_with"]["other_target_runs_skipped"] == 1 + + +def test_a_failed_last_run_is_not_a_count_to_hold_the_table_to(bq, tmp_path): + bq([("s1", HASHED)]) + _run_record(tmp_path, "20260928T000001Z", rows=5) + _run_record(tmp_path, "20260928T000002Z", rows=0, table=None, state="failed") + rc = _verify(tmp_path)["dimensions"]["row_count"] + assert rc["status"] == "pass" and rc["expected"] is None + + +def test_an_empty_table_fails(bq, tmp_path): + bq([]) + result = _verify(tmp_path) + assert result["dimensions"]["row_count"]["status"] == "fail" + assert result["severity"]["level"] == "CRITICAL" + + +def test_a_count_that_cannot_run_is_an_error_not_a_pass(bq, tmp_path): + bq([("s1", HASHED)], query_error=PermissionError("403 bigquery.jobs.create")) + result = _verify(tmp_path) + assert result["status"] == "error" + assert "bigquery.jobs.create" in result["error"] + + +def test_the_direct_call_without_an_expose_keeps_its_four_dimensions(bq, tmp_path): + fake = bq([("s1", HASHED)]) + result = verify_bigquery_table( + "northwind-demo", "demo_bronze", "customer_subscriptions", [], "europe-west1" + ) + assert "row_count" not in result["dimensions"] and fake.sql == [] + + +# ── Through the CLI ───────────────────────────────────────────────────── + + +def _args(contract_path: Path, *, strict: bool = True) -> argparse.Namespace: + return argparse.Namespace( + contract=str(contract_path), + expose_id=None, + strict=strict, + out=None, + show_diffs=False, + env=None, + ) + + +def _write(tmp_path: Path, contract: Dict[str, Any]) -> Path: + path = tmp_path / "contract.fluid.yaml" + path.write_text(yaml.safe_dump(contract, sort_keys=False), encoding="utf-8") + return path + + +def test_strict_verify_fails_on_seeded_cleartext_and_passes_when_treated(bq, tmp_path): + path = _write(tmp_path, _contract()) + _run_record(tmp_path, "20260928T000001Z", rows=2) + bq([("s1", HASHED), ("s2", HASHED)]) + assert run(_args(path), _LOG) == 0 + bq([("s1", HASHED), ("s2", "+46701234567")]) + assert run(_args(path), _LOG) == 1 + + +def test_a_table_that_reports_no_row_count_is_rendered_not_a_crash(tmp_path, capsys): + """0.16.5 formatted ``numRows`` with ``:,``, and the emulator leaves it None.""" + path = _write(tmp_path, _contract()) + result = { + "status": "match", + "exists": True, + "table_id": TABLE_ID, + "severity": {"level": "SUCCESS", "symbol": "🟢", "impact": "NONE", "remediation": "NONE"}, + "dimensions": {}, + "metadata": {"num_rows": None}, + } + with patch("fluid_build.cli.verify.verify_bigquery_table", return_value=result): + assert run(_args(path, strict=False), _LOG) == 0 + assert "Table Rows: unknown" in capsys.readouterr().out + + +def test_verify_resolves_the_project_the_way_the_load_does(tmp_path, monkeypatch): + monkeypatch.setenv("FLUID_DEMO_GCP_PROJECT", "acme-eu-demo") + path = _write(tmp_path, _contract(_expose(project="{{ env.FLUID_DEMO_GCP_PROJECT }}"))) + seen: Dict[str, Any] = {} + + def fake_verify(**kw): + seen.update(kw) + return {"status": "match", "dimensions": {}} + + with patch("fluid_build.cli.verify.verify_bigquery_table", side_effect=fake_verify): + run(_args(path, strict=False), _LOG) + assert seen["project"] == "acme-eu-demo" + assert seen["expose"]["exposeId"] == "subscriptions" + + +# ── The project, when the binding names none ──────────────────────────── +# +# PR review: the load and the read name the table from ``client.project`` +# (ADC's when nothing else names one), but verify passed ``project=""``, which +# the client keeps, and looked up ``.demo_bronze.customer_subscriptions``. + + +def test_a_binding_with_no_project_is_verified_in_the_clients_project(bq, tmp_path): + fake = bq([("s1", HASHED)]) + result = verify_bigquery_table("", "demo_bronze", "customer_subscriptions", [], "europe-west1") + assert fake.tables_read == ["adc-project.demo_bronze.customer_subscriptions"] + assert result["exists"] is True + assert result["table_id"] == "adc-project.demo_bronze.customer_subscriptions" + + +def test_no_project_anywhere_is_an_error_naming_the_table(bq, tmp_path): + fake = bq([("s1", HASHED)], adc_project=None) + result = verify_bigquery_table("", "demo_bronze", "customer_subscriptions", [], "europe-west1") + assert result["status"] == "error" + assert "No project for demo_bronze.customer_subscriptions" in result["error"] + assert fake.tables_read == [] + + +# ── An embedded-SQL build's load is held to, as an acquisition load is ── + + +def _embedded_sql_contract() -> Dict[str, Any]: + contract = _contract() + contract["builds"] = [ + { + "id": "summarize", + "pattern": "embedded-logic", + "engine": "duckdb", + "properties": {"sql": "SELECT * FROM upstream"}, + "outputs": ["subscriptions"], + } + ] + return contract + + +def _record_embedded_sql_load(tmp_path: Path, rows: int) -> None: + """The record an embedded-SQL build writes after its load: the acquisition + load's shape (``_embedded_sql_io.write_bigquery_run_record``; the build + side is pinned in ``tests/build_runners/test_embedded_sql_bigquery.py``).""" + _run_record(tmp_path, "20260928T000001Z", rows=rows, build_id="summarize") + + +def test_an_embedded_sql_table_is_held_to_the_rows_its_load_landed(bq, tmp_path): + contract = _embedded_sql_contract() + _record_embedded_sql_load(tmp_path, rows=2) + bq([("s1", HASHED), ("s2", HASHED)]) + rc = _verify(tmp_path, contract)["dimensions"]["row_count"] + assert (rc["status"], rc["actual"], rc["expected"]) == ("pass", 2, 2) + assert rc["compared_with"]["rule"] == "equal" + assert rc["compared_with"]["build_id"] == "summarize" + + +def test_an_embedded_sql_table_short_of_its_load_fails_as_critical(bq, tmp_path): + contract = _embedded_sql_contract() + _record_embedded_sql_load(tmp_path, rows=3) + bq([("s1", HASHED), ("s2", HASHED)]) + result = _verify(tmp_path, contract) + assert result["dimensions"]["row_count"]["status"] == "fail" + assert result["severity"]["level"] == "CRITICAL" diff --git a/tests/cli/test_verify_bigquery_governance.py b/tests/cli/test_verify_bigquery_governance.py new file mode 100644 index 00000000..d8363809 --- /dev/null +++ b/tests/cli/test_verify_bigquery_governance.py @@ -0,0 +1,487 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""``fluid verify`` on BigQuery: partition expiration, kmsKeyName and policy tags. + +Before this, ``verify_bigquery_table`` checked schema and location only, so a +table with no retention, no key and no column restriction passed ``--strict`` +whatever the contract declared. The dataset and table here are real +``google.cloud.bigquery`` ``Dataset`` / ``Table`` objects built from the REST +representation BigQuery returns (``from_api_repr``), so the attribute paths are the +library's; the Data Catalog calls go to a recorded fake session. No emulator: +goccy's does not implement policy tags or CMEK, and nothing here needs a network. +""" + +from __future__ import annotations + +import copy +from types import SimpleNamespace +from typing import Any, Dict, List, Optional + +import pytest + +bigquery = pytest.importorskip("google.cloud.bigquery") + +from fluid_build.cli import _verify_bigquery_governance as bqgov # noqa: E402 +from fluid_build.cli.verify import verify_bigquery_table # noqa: E402 + +PROJECT = "northwind-demo" +KEY = ( + "projects/northwind-demo/locations/europe-west1/keyRings/" + "fluid-gold_retention_candidates-demo_gold/cryptoKeys/bigquery" +) +TAG = "projects/northwind-demo/locations/europe-west1/taxonomies/111/policyTags/222" +PLATFORM = "group:data-platform@northwind.com" +PIPELINE = "serviceAccount:fluid-pipeline@northwind-demo.iam.gserviceaccount.com" +ANALYSTS = "group:analysts@northwind.com" + + +def _contract(**expose: Any) -> Dict[str, Any]: + binding = { + "platform": "gcp", + "format": "bigquery_table", + "location": { + "project": PROJECT, + "dataset": "demo_gold", + "table": "retention_candidates", + "region": "europe-west1", + }, + "principals": { + "group:data-platform@northwind.example": PLATFORM, + "group:analysts@northwind.example": ANALYSTS, + "serviceAccount:fluid-pipeline@northwind.example": PIPELINE, + }, + **expose.pop("binding", {}), + } + exposure = { + "exposeId": "candidates", + "binding": binding, + "contract": { + "schema": [ + {"name": "customer_id", "type": "string", "required": True}, + {"name": "msisdn", "type": "string"}, + ] + }, + **expose, + } + return { + "id": "gold.retention_candidates", + "accessPolicy": { + "grants": [ + {"principal": "group:data-platform@northwind.example", "permissions": ["read"]}, + {"principal": "group:analysts@northwind.example", "permissions": ["read"]}, + { + "principal": "serviceAccount:fluid-pipeline@northwind.example", + "permissions": ["read", "write"], + }, + ] + }, + "exposes": [exposure], + } + + +GOVERNED = dict( + lifecycle={"retention": "P90D", "expire": True}, + binding={"encryption": {"kms": "product"}}, + policy={ + "authz": { + "columnRestrictions": [ + { + "principal": "group:analysts@northwind.example", + "columns": ["msisdn"], + "access": "deny", + } + ] + } + }, +) + + +def _table( + *, + partitioning: Optional[Dict[str, Any]] = None, + kms: Optional[str] = KEY, + tag: Optional[str] = TAG, + expires: Optional[str] = None, +) -> Any: + msisdn: Dict[str, Any] = {"name": "msisdn", "type": "STRING", "mode": "NULLABLE"} + if tag: + msisdn["policyTags"] = {"names": [tag]} + resource: Dict[str, Any] = { + "tableReference": { + "projectId": PROJECT, + "datasetId": "demo_gold", + "tableId": "retention_candidates", + }, + "schema": { + "fields": [{"name": "customer_id", "type": "STRING", "mode": "REQUIRED"}, msisdn] + }, + "numRows": "3", + } + if partitioning is not None: + resource["timePartitioning"] = partitioning + if kms: + resource["encryptionConfiguration"] = {"kmsKeyName": kms} + if expires: + resource["expirationTime"] = expires + return bigquery.Table.from_api_repr(resource) + + +def _dataset(kms: Optional[str] = KEY) -> Any: + resource: Dict[str, Any] = { + "datasetReference": {"projectId": PROJECT, "datasetId": "demo_gold"}, + "location": "europe-west1", + } + if kms: + resource["defaultEncryptionConfiguration"] = {"kmsKeyName": kms} + return bigquery.Dataset.from_api_repr(resource) + + +PARTITIONED = {"type": "DAY", "expirationMs": str(90 * 86_400_000)} + + +class _Response: + def __init__(self, status: int, body: Dict[str, Any]) -> None: + self.status_code = status + self._body = body + + def json(self) -> Dict[str, Any]: + return self._body + + +class _Catalog: + """Records Data Catalog calls; answers the tag's IAM policy and its taxonomy.""" + + def __init__(self, readers: List[str], *, enforced: bool = True, status: int = 200) -> None: + self.readers = readers + self.enforced = enforced + self.status = status + self.calls: List[str] = [] + + def post(self, url: str, json: Any = None) -> _Response: + self.calls.append(f"POST {url}") + return _Response( + self.status, + { + "bindings": [ + {"role": "roles/datacatalog.categoryFineGrainedReader", "members": self.readers} + ] + }, + ) + + def get(self, url: str) -> _Response: + self.calls.append(f"GET {url}") + types = ["FINE_GRAINED_ACCESS_CONTROL"] if self.enforced else [] + return _Response(self.status, {"activatedPolicyTypes": types}) + + +def _dims(contract: Dict[str, Any], table: Any, dataset: Any, catalog: Any) -> Dict[str, Any]: + return bqgov.governance_dimensions( + contract["exposes"][0], + contract=contract, + bq_dataset=dataset, + bq_table=table, + project=PROJECT, + session_factory=lambda: catalog, + ) + + +def test_a_governed_table_passes_all_three(): + catalog = _Catalog([PLATFORM, PIPELINE]) + dims = _dims( + _contract(**copy.deepcopy(GOVERNED)), _table(partitioning=PARTITIONED), _dataset(), catalog + ) + assert {name: d["status"] for name, d in dims.items()} == { + "retention": "pass", + "encryption": "pass", + "columnRestrictions": "pass", + } + assert catalog.calls == [ + f"POST https://datacatalog.googleapis.com/v1/{TAG}:getIamPolicy", + "GET https://datacatalog.googleapis.com/v1/projects/northwind-demo/locations/europe-west1/taxonomies/111", + ] + + +def test_nothing_declared_nothing_checked(): + assert _dims(_contract(), _table(), _dataset(), _Catalog([])) == {} + + +@pytest.mark.parametrize( + "partitioning, expires, fragment", + [ + (None, None, "not partitioned"), + ({"type": "DAY", "expirationMs": str(30 * 86_400_000)}, None, "after 30 days"), + ({"type": "DAY"}, None, "expire never"), + ({"type": "MONTH", "expirationMs": str(90 * 86_400_000)}, None, "not by DAY"), + (PARTITIONED, "1790000000000", "deletes the whole product"), + ], +) +def test_retention_fails_when_partitions_do_not_expire_as_declared(partitioning, expires, fragment): + contract = _contract(lifecycle={"retention": "P90D", "expire": True}) + dims = _dims( + contract, _table(partitioning=partitioning, expires=expires), _dataset(), _Catalog([]) + ) + assert dims["retention"]["status"] == "fail" + assert fragment in dims["retention"]["message"] + + +def test_retention_checks_the_declared_partition_column(): + contract = _contract( + lifecycle={"retention": "P90D", "expire": True}, + binding={ + "location": {**_contract()["exposes"][0]["binding"]["location"], "partitionBy": ["ts"]} + }, + ) + contract["exposes"][0]["contract"]["schema"].append({"name": "ts", "type": "timestamp"}) + dims = _dims(contract, _table(partitioning=PARTITIONED), _dataset(), _Catalog([])) + assert "partitioned on ingestion time, not on ts" in dims["retention"]["message"] + + +@pytest.mark.parametrize( + "table_kms, dataset_kms, fragment", + [ + (None, KEY, "the table is encrypted with a Google-managed key"), + (KEY, None, "the dataset's default key is a Google-managed key"), + (KEY.replace("bigquery", "other"), KEY, "the table is encrypted with"), + ], +) +def test_encryption_fails_on_the_wrong_key(table_kms, dataset_kms, fragment): + contract = _contract(binding={"encryption": {"kms": "product"}}) + dims = _dims(contract, _table(kms=table_kms), _dataset(dataset_kms), _Catalog([])) + assert dims["encryption"]["status"] == "fail" + assert fragment in dims["encryption"]["message"] + + +def test_a_key_version_in_kms_key_name_still_matches(): + contract = _contract(binding={"encryption": {"kms": "product"}}) + dims = _dims(contract, _table(kms=f"{KEY}/cryptoKeyVersions/3"), _dataset(), _Catalog([])) + assert dims["encryption"]["status"] == "pass" + + +@pytest.mark.parametrize( + "readers, tag, enforced, fragment", + [ + ([PLATFORM, PIPELINE, ANALYSTS], TAG, True, f"{ANALYSTS} can read msisdn"), + ([PLATFORM], TAG, True, f"{PIPELINE} cannot read msisdn"), + ([PLATFORM, PIPELINE], None, True, "carries no policy tag"), + ([PLATFORM, PIPELINE], TAG, False, "does not enforce fine-grained access control"), + ], +) +def test_column_restrictions_fail_when_the_tag_does_not_hold(readers, tag, enforced, fragment): + contract = _contract(policy=copy.deepcopy(GOVERNED["policy"])) + dims = _dims(contract, _table(tag=tag), _dataset(), _Catalog(readers, enforced=enforced)) + assert dims["columnRestrictions"]["status"] == "fail" + assert fragment in dims["columnRestrictions"]["message"] + + +def test_a_catalog_that_cannot_be_read_is_an_error_not_a_pass(): + contract = _contract(policy=copy.deepcopy(GOVERNED["policy"])) + dims = _dims(contract, _table(), _dataset(), _Catalog([PLATFORM, PIPELINE], status=403)) + assert dims["columnRestrictions"]["status"] == "error" + assert "HTTP 403" in dims["columnRestrictions"]["message"] + + +# ── where Data Catalog cannot be reached ───────────────────────────────── +# +# Measured on the BigQuery emulator (goccy, BIGQUERY_EMULATOR_HOST, no ADC): a +# table with no policy tag on its restricted column was reported "error: +# DefaultCredentialsError", because the Data Catalog session was opened before +# the tags were read. The missing tag is the finding. + + +def _no_credentials() -> Any: + raise RuntimeError("DefaultCredentialsError: Your default credentials were not found") + + +def _dims_without_a_session(contract: Dict[str, Any], table: Any) -> Dict[str, Any]: + return bqgov.governance_dimensions( + contract["exposes"][0], + contract=contract, + bq_dataset=_dataset(), + bq_table=table, + project=PROJECT, + session_factory=_no_credentials, + ) + + +def test_a_missing_tag_fails_without_asking_data_catalog(monkeypatch): + monkeypatch.delenv("BIGQUERY_EMULATOR_HOST", raising=False) + contract = _contract(policy=copy.deepcopy(GOVERNED["policy"])) + opened: List[int] = [] + + def factory() -> Any: + opened.append(1) + return _Catalog([PLATFORM, PIPELINE]) + + dims = bqgov.governance_dimensions( + contract["exposes"][0], + contract=contract, + bq_dataset=_dataset(), + bq_table=_table(tag=None), + project=PROJECT, + session_factory=factory, + ) + assert dims["columnRestrictions"]["status"] == "fail" + assert "column msisdn carries no policy tag" in dims["columnRestrictions"]["message"] + assert opened == [] + + +@pytest.mark.parametrize("emulator", ["http://127.0.0.1:9050", ""]) +def test_a_missing_tag_is_the_finding_where_there_are_no_credentials(monkeypatch, emulator): + monkeypatch.setenv("BIGQUERY_EMULATOR_HOST", emulator) + contract = _contract(policy=copy.deepcopy(GOVERNED["policy"])) + dims = _dims_without_a_session(contract, _table(tag=None)) + assert dims["columnRestrictions"]["status"] == "fail" + assert "carries no policy tag" in dims["columnRestrictions"]["message"] + assert "DefaultCredentialsError" not in dims["columnRestrictions"]["message"] + + +def test_on_the_emulator_a_tagged_column_s_readers_are_not_verifiable(monkeypatch): + monkeypatch.setenv("BIGQUERY_EMULATOR_HOST", "http://127.0.0.1:9050") + contract = _contract(policy=copy.deepcopy(GOVERNED["policy"])) + catalog = _Catalog([PLATFORM, PIPELINE]) + dims = _dims(contract, _table(), _dataset(), catalog) + dimension = dims["columnRestrictions"] + assert dimension["status"] == "unsupported" + assert "every restricted column carries a policy tag" in dimension["message"] + assert "no emulator serves Data Catalog" in dimension["message"] + assert catalog.calls == [] + assert bqgov.governance_errors(dims) == [] + assert bqgov.governance_problems(dims) == [] + + +def test_off_the_emulator_a_tagged_column_without_credentials_is_still_an_error(monkeypatch): + monkeypatch.delenv("BIGQUERY_EMULATOR_HOST", raising=False) + contract = _contract(policy=copy.deepcopy(GOVERNED["policy"])) + dims = _dims_without_a_session(contract, _table()) + assert dims["columnRestrictions"]["status"] == "error" + assert "DefaultCredentialsError" in dims["columnRestrictions"]["message"] + + +# ── through verify_bigquery_table ──────────────────────────────────────── + + +class _Client: + def __init__(self, table: Any, dataset: Any) -> None: + self._table, self._dataset = table, dataset + + def get_table(self, _table_id: str) -> Any: + return self._table + + def get_dataset(self, _dataset_id: str) -> Any: + return self._dataset + + def query(self, sql: str, job_config: Any = None, location: Any = None) -> Any: + # verify_bigquery_table with an expose also counts the table (the + # row_count and masking dimensions): one row, a non-empty count, then + # zero for each masking check the query selects. + row = SimpleNamespace(values=lambda: (3, *([0] * sql.count("COUNTIF(")))) + return SimpleNamespace(job_id="job_1", result=lambda timeout=None: [row]) + + +def _verify(monkeypatch, contract, table, dataset, catalog) -> Dict[str, Any]: + monkeypatch.setattr(bigquery, "Client", lambda project=None: _Client(table, dataset)) + exposure = contract["exposes"][0] + return verify_bigquery_table( + PROJECT, + "demo_gold", + "retention_candidates", + exposure["contract"]["schema"], + "europe-west1", + expose=exposure, + contract=contract, + catalog_session_factory=lambda: catalog, + ) + + +def test_verify_bigquery_table_is_critical_on_a_governance_mismatch(monkeypatch): + contract = _contract(**copy.deepcopy(GOVERNED)) + result = _verify( + monkeypatch, contract, _table(partitioning=None), _dataset(), _Catalog([PLATFORM, PIPELINE]) + ) + assert result["status"] == "mismatch" + assert result["severity"]["level"] == "CRITICAL" + assert "not partitioned" in result["severity"]["reason"] + assert result["dimensions"]["structure"]["status"] == "pass" + assert result["dimensions"]["retention"]["status"] == "fail" + + +def test_verify_bigquery_table_matches_a_governed_table(monkeypatch): + contract = _contract(**copy.deepcopy(GOVERNED)) + result = _verify( + monkeypatch, + contract, + _table(partitioning=PARTITIONED), + _dataset(), + _Catalog([PLATFORM, PIPELINE]), + ) + assert result["status"] == "match", result + + +def test_verify_bigquery_table_errors_when_a_policy_cannot_be_checked(monkeypatch): + contract = _contract(**copy.deepcopy(GOVERNED)) + result = _verify( + monkeypatch, + contract, + _table(partitioning=PARTITIONED), + _dataset(), + _Catalog([PLATFORM, PIPELINE], status=500), + ) + assert result["status"] == "error" + assert "HTTP 500" in result["error"] + + +# ── through `fluid verify` itself ──────────────────────────────────────── + + +def test_fluid_verify_checks_the_governance_of_a_gcp_expose(tmp_path, monkeypatch): + """``verify.run`` must hand the expose and the contract to the BigQuery check. + + Without them every governance dimension is skipped and a table with no + retention, key or policy tag passes: the call site is the only wiring. + """ + import argparse + import json + import logging + + import yaml + + from fluid_build.cli import verify + + contract = _contract(**copy.deepcopy(GOVERNED)) + contract.update(fluidVersion="0.7.6", kind="DataProduct", name="Retention candidates") + contract["metadata"] = {"owner": {"team": "data-platform"}} + path = tmp_path / "contract.fluid.yaml" + path.write_text(yaml.safe_dump(contract), encoding="utf-8") + monkeypatch.chdir(tmp_path) + table, dataset = _table(partitioning=None), _dataset() + monkeypatch.setattr(bigquery, "Client", lambda project=None: _Client(table, dataset)) + monkeypatch.setattr(bqgov, "default_session", lambda: _Catalog([PLATFORM, PIPELINE])) + out = tmp_path / "report.json" + args = argparse.Namespace( + contract=str(path), + expose_id=None, + strict=True, + fail_on_warning=False, + out=str(out), + show_diffs=False, + env=None, + ) + code = verify.run(args, logging.getLogger("test")) + result = json.loads(out.read_text(encoding="utf-8"))["results"]["candidates"] + assert {"retention", "encryption", "columnRestrictions"} <= set(result["dimensions"]) + assert result["dimensions"]["retention"]["status"] == "fail" + assert result["dimensions"]["encryption"]["status"] == "pass" + assert result["dimensions"]["columnRestrictions"]["status"] == "pass" + assert code != 0 diff --git a/tests/forge/test_artifact_fanout_schedule.py b/tests/forge/test_artifact_fanout_schedule.py index cc395aca..f65c77b8 100644 --- a/tests/forge/test_artifact_fanout_schedule.py +++ b/tests/forge/test_artifact_fanout_schedule.py @@ -46,6 +46,9 @@ LOG = logging.getLogger("test.artifact_fanout_schedule") DAG_REL = "schedule/bronze.customer_subscriptions/ingest_subscriptions_dag.py" +#: An env's DAGs get a directory of their own (``__``), so an aws +#: and a gcp pipeline syncing to one Airflow never delete each other's DAG. +DAG_REL_AWS = "schedule/bronze.customer_subscriptions__aws/ingest_subscriptions_dag.py" #: An aws overlay that also moves the schedule, so the DAG shows which #: contract it was rendered from. @@ -140,7 +143,7 @@ def test_a_bundle_takes_its_env_and_contract_path_from_the_flags( ) == 0 ) - dag = tmp_path / "dist" / "artifacts" / DAG_REL + dag = tmp_path / "dist" / "artifacts" / DAG_REL_AWS loaded = load_dag(dag.read_text(), monkeypatch) assert loaded.namespace["FLUID_ENV_NAME"] == "aws" assert loaded.namespace["CONTRACT_PATH"] == DEMO_CONTRACT_PATH @@ -170,7 +173,7 @@ def test_a_raw_contract_is_rendered_with_its_env_overlay( ) argv = (DEMO_CONTRACT_PATH, "--out", "dist/artifacts", "--env", "aws") assert _stage3(tmp_path, monkeypatch, *argv) == 0 - loaded = load_dag((tmp_path / "dist" / "artifacts" / DAG_REL).read_text(), monkeypatch) + loaded = load_dag((tmp_path / "dist" / "artifacts" / DAG_REL_AWS).read_text(), monkeypatch) assert loaded.dag["schedule"] == "30 1 * * *" assert loaded.namespace["FLUID_ENV_NAME"] == "aws" # Fanned out through a temporary --env bundle, the DAG still runs the @@ -326,7 +329,7 @@ def test_without_env_stage_3_uses_fluid_env( ) monkeypatch.setenv("FLUID_ENV", "aws") assert _stage3(tmp_path, monkeypatch, DEMO_CONTRACT_PATH, "--out", "dist/artifacts") == 0 - loaded = load_dag((tmp_path / "dist" / "artifacts" / DAG_REL).read_text(), monkeypatch) + loaded = load_dag((tmp_path / "dist" / "artifacts" / DAG_REL_AWS).read_text(), monkeypatch) assert loaded.namespace["FLUID_ENV_NAME"] == "aws" assert loaded.dag["schedule"] == "30 1 * * *" @@ -337,7 +340,7 @@ def test_the_flag_beats_fluid_env( monkeypatch.setenv("FLUID_ENV", "gcp") argv = (DEMO_CONTRACT_PATH, "--out", "dist/artifacts", "--env", "aws") assert _stage3(tmp_path, monkeypatch, *argv) == 0 - loaded = load_dag((tmp_path / "dist" / "artifacts" / DAG_REL).read_text(), monkeypatch) + loaded = load_dag((tmp_path / "dist" / "artifacts" / DAG_REL_AWS).read_text(), monkeypatch) assert loaded.namespace["FLUID_ENV_NAME"] == "aws" @pytest.mark.parametrize("fluid_env", [None, ""]) diff --git a/tests/iac/_fake_bigquery.py b/tests/iac/_fake_bigquery.py new file mode 100644 index 00000000..e6f03f93 --- /dev/null +++ b/tests/iac/_fake_bigquery.py @@ -0,0 +1,258 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""A minimal in-process stand-in for the BigQuery v2 REST API, for real ``tofu`` runs. + +``tofu apply`` against the goccy bigquery-emulator crashes inside +terraform-provider-google (the emulator returns a dataset without ``selfLink``, +and ``resource_bigquery_dataset.go`` asserts it is a string), so the emulator +cannot hold the state a governance test needs. This server stores exactly what +the provider sends for datasets and tables and returns it with the output-only +fields the provider reads (ids, ``selfLink``, ``etag``, timestamps), and a new +dataset gets the default access entries BigQuery gives one (the project's owners, +writers and readers, and the creator). Nothing else is emulated: no query, no +load, no IAM evaluation, no Cloud KMS and no Data Catalog. What it proves is what +the real provider and ``tofu`` plan for a live table, which is where the +data-loss gate decides; what BigQuery itself accepts is not proven here. + +The provider is pointed at it with ``bigquery_custom_endpoint`` and a dummy +``access_token``, so no credential and no Google endpoint is involved. +""" + +from __future__ import annotations + +import copy +import json +import re +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from typing import Any, Dict, List, Optional, Tuple +from urllib.parse import urlparse + +_DATASET_RE = re.compile(r"^/bigquery/v2/projects/([^/]+)/datasets/?([^/]*)$") +_TABLE_RE = re.compile(r"^/bigquery/v2/projects/([^/]+)/datasets/([^/]+)/tables/?([^/]*)$") + +#: What BigQuery puts on a dataset created without ``access`` (BigQuery +#: documentation, "Dataset access": the project's basic roles and the creator). +DEFAULT_ACCESS = [ + {"role": "OWNER", "specialGroup": "projectOwners"}, + {"role": "OWNER", "userByEmail": "creator@fake.iam.gserviceaccount.com"}, + {"role": "WRITER", "specialGroup": "projectWriters"}, + {"role": "READER", "specialGroup": "projectReaders"}, +] + + +#: BigQuery returns the basic dataset roles by their legacy names, whatever it was +#: sent (terraform-provider-google's ``iam_bigquery_dataset.go``: "API changes +#: certain IAM roles to legacy roles"). +_LEGACY_ROLES = { + "roles/bigquery.dataOwner": "OWNER", + "roles/bigquery.dataEditor": "WRITER", + "roles/bigquery.dataViewer": "READER", +} + + +def _legacy_access(dataset: Dict[str, Any]) -> None: + for entry in dataset.get("access") or []: + if isinstance(entry, dict) and entry.get("role") in _LEGACY_ROLES: + entry["role"] = _LEGACY_ROLES[entry["role"]] + + +class FakeBigQuery: + """Datasets and tables by ``(project, dataset[, table])``; every request is logged.""" + + def __init__(self) -> None: + self.datasets: Dict[Tuple[str, str], Dict[str, Any]] = {} + self.tables: Dict[Tuple[str, str, str], Dict[str, Any]] = {} + self.requests: List[Tuple[str, str]] = [] + self._lock = threading.Lock() + self._server: Optional[ThreadingHTTPServer] = None + self._thread: Optional[threading.Thread] = None + + # ── lifecycle ──────────────────────────────────────────────────────── + def start(self) -> "FakeBigQuery": + fake = self + + class Handler(BaseHTTPRequestHandler): + def log_message(self, *args: Any) -> None: # quiet + return + + def _body(self) -> Dict[str, Any]: + length = int(self.headers.get("Content-Length") or 0) + raw = self.rfile.read(length) if length else b"" + return json.loads(raw) if raw else {} + + def _send(self, status: int, body: Optional[Dict[str, Any]]) -> None: + data = json.dumps(body).encode() if body is not None else b"" + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(data))) + self.end_headers() + if data: + self.wfile.write(data) + + def _handle(self, method: str) -> None: + path = urlparse(self.path).path + with fake._lock: + fake.requests.append((method, path)) + status, body = fake._route( + method, path, self._body() if method in ("POST", "PUT", "PATCH") else {} + ) + self._send(status, body) + + def do_GET(self) -> None: # noqa: N802 — http.server's naming + self._handle("GET") + + def do_POST(self) -> None: # noqa: N802 + self._handle("POST") + + def do_PUT(self) -> None: # noqa: N802 + self._handle("PUT") + + def do_PATCH(self) -> None: # noqa: N802 + self._handle("PATCH") + + def do_DELETE(self) -> None: # noqa: N802 + self._handle("DELETE") + + self._server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + self._thread = threading.Thread(target=self._server.serve_forever, daemon=True) + self._thread.start() + return self + + def stop(self) -> None: + if self._server is not None: + self._server.shutdown() + self._server.server_close() + + @property + def endpoint(self) -> str: + assert self._server is not None + host, port = self._server.server_address[:2] + return f"http://{host}:{port}/bigquery/v2/" + + # ── routing ────────────────────────────────────────────────────────── + @staticmethod + def _not_found(what: str) -> Tuple[int, Dict[str, Any]]: + return 404, { + "error": { + "code": 404, + "message": f"Not found: {what}", + "errors": [{"reason": "notFound", "message": f"Not found: {what}"}], + "status": "NOT_FOUND", + } + } + + def _stamp(self, body: Dict[str, Any]) -> None: + now = str(int(time.time() * 1000)) + body.setdefault("creationTime", now) + body["lastModifiedTime"] = now + body["etag"] = f"etag-{now}" + + def _route(self, method: str, path: str, body: Dict[str, Any]) -> Tuple[int, Any]: + match = _TABLE_RE.match(path) + if match: + return self._table(method, *match.groups(), body) + match = _DATASET_RE.match(path) + if match: + return self._dataset(method, *match.groups(), body) + return self._not_found(path) + + def _dataset(self, method: str, project: str, dataset: str, body: Dict[str, Any]) -> Any: + if method == "POST" and not dataset: + ref = body.get("datasetReference") or {} + key = (project, ref.get("datasetId") or "") + stored = copy.deepcopy(body) + stored.setdefault("access", copy.deepcopy(DEFAULT_ACCESS)) + _legacy_access(stored) + stored["id"] = f"{project}:{key[1]}" + stored["selfLink"] = f"{self.endpoint}projects/{project}/datasets/{key[1]}" + stored["kind"] = "bigquery#dataset" + stored.setdefault("location", "US") + self._stamp(stored) + self.datasets[key] = stored + return 200, stored + key = (project, dataset) + if key not in self.datasets: + return self._not_found(f"Dataset {project}:{dataset}") + if method == "GET": + return 200, self.datasets[key] + if method in ("PATCH", "PUT"): + stored = ( + self.datasets[key] + if method == "PATCH" + else { + k: v for k, v in self.datasets[key].items() if k in ("id", "selfLink", "kind") + } + ) + stored.update(copy.deepcopy(body)) + _legacy_access(stored) + self._stamp(stored) + self.datasets[key] = stored + return 200, stored + if method == "DELETE": + del self.datasets[key] + return 204, None + return self._not_found(path_of(method, project, dataset)) + + def _table( + self, method: str, project: str, dataset: str, table: str, body: Dict[str, Any] + ) -> Any: + if (project, dataset) not in self.datasets: + return self._not_found(f"Dataset {project}:{dataset}") + if method == "POST" and not table: + ref = body.get("tableReference") or {} + key = (project, dataset, ref.get("tableId") or "") + if key in self.tables: + return 409, {"error": {"code": 409, "message": "Already Exists"}} + stored = copy.deepcopy(body) + stored["id"] = f"{project}:{dataset}.{key[2]}" + stored["selfLink"] = ( + f"{self.endpoint}projects/{project}/datasets/{dataset}/tables/{key[2]}" + ) + stored["kind"] = "bigquery#table" + stored.setdefault("type", "TABLE") + stored["location"] = self.datasets[(project, dataset)].get("location", "US") + stored.setdefault("numRows", "0") + stored.setdefault("numBytes", "0") + self._stamp(stored) + self.tables[key] = stored + return 200, stored + key = (project, dataset, table) + if key not in self.tables: + return self._not_found(f"Table {project}:{dataset}.{table}") + if method == "GET": + return 200, self.tables[key] + if method in ("PUT", "PATCH"): + stored = self.tables[key] + keep = {k: stored[k] for k in ("id", "selfLink", "kind", "creationTime", "location")} + merged = dict(stored) if method == "PATCH" else {} + merged.update(copy.deepcopy(body)) + merged.update(keep) + merged.setdefault("type", "TABLE") + self._stamp(merged) + self.tables[key] = merged + return 200, merged + if method == "DELETE": + del self.tables[key] + return 204, None + return self._not_found(path_of(method, project, dataset, table)) + + +def path_of(method: str, *parts: str) -> str: + return f"{method} {'/'.join(parts)}" + + +__all__ = ["DEFAULT_ACCESS", "FakeBigQuery"] diff --git a/tests/iac/test_iac_apply_engine.py b/tests/iac/test_iac_apply_engine.py index d2d4dd2b..f2c9d72a 100644 --- a/tests/iac/test_iac_apply_engine.py +++ b/tests/iac/test_iac_apply_engine.py @@ -46,6 +46,74 @@ def test_destructive_plan_allowed_with_flag(self): def test_additive_plan_never_blocked(self): assert engine._data_loss_blocked({"add": 5, "change": 1, "remove": 0}, False) is False + def test_a_revoked_grant_or_policy_tag_is_not_counted(self): + removals = [ + ("google_bigquery_dataset_iam_member.a", "google_bigquery_dataset_iam_member"), + ("google_data_catalog_policy_tag.t", "google_data_catalog_policy_tag"), + ("aws_lakeformation_permissions.p", "aws_lakeformation_permissions"), + ] + counted, revoked = engine._data_bearing_changes( + {"add": 0, "change": 0, "remove": 3}, removals + ) + assert counted["remove"] == 0 and len(revoked) == 3 + assert engine._data_loss_blocked(counted, False) is False + + def test_a_table_or_a_key_grant_is_still_counted(self): + removals = [ + ("google_bigquery_table.t", "google_bigquery_table"), + ("google_kms_crypto_key_iam_member.k", "google_kms_crypto_key_iam_member"), + ] + counted, revoked = engine._data_bearing_changes( + {"add": 1, "change": 0, "remove": 2}, removals + ) + assert counted["remove"] == 2 and revoked == [] + + +class TestReconcileWithState: + """The engine's side of the GCP plugin's one-time access-list reconciliation.""" + + class _Plugin: + def __init__(self, reports): + self.reports = reports + + def reconcile_state(self, module, state): + module["patched"] = True + return self.reports + + def test_a_patched_module_is_written_back(self, monkeypatch, tmp_path): + monkeypatch.setattr(engine.runner, "tofu_state_resources", lambda *a, **k: [{"x": 1}]) + path = tmp_path / "main.tf.json" + path.write_text("{}", encoding="utf-8") + reports = [{"dataset": "sales", "revoked": ["READER group:gone@x.com"], "blocked": False}] + got = engine._reconcile_with_state( + self._Plugin(reports), path, str(tmp_path), {}, logging.getLogger("test") + ) + assert got == reports + assert json.loads(path.read_text(encoding="utf-8")) == {"patched": True} + + def test_a_list_that_cannot_be_narrowed_refuses_the_apply(self, monkeypatch, tmp_path): + monkeypatch.setattr(engine.runner, "tofu_state_resources", lambda *a, **k: [{"x": 1}]) + path = tmp_path / "main.tf.json" + path.write_text("{}", encoding="utf-8") + reports = [{"dataset": "sales", "revoked": ["READER group:gone@x.com"], "blocked": True}] + with pytest.raises(CLIError) as refused: + engine._reconcile_with_state( + self._Plugin(reports), path, str(tmp_path), {}, logging.getLogger("test") + ) + assert refused.value.event == "opentofu_dataset_access_unreconciled" + assert "group:gone@x.com" in refused.value.context["error"] + assert path.read_text(encoding="utf-8") == "{}" + + def test_no_state_is_a_no_op(self, monkeypatch, tmp_path): + monkeypatch.setattr(engine.runner, "tofu_state_resources", lambda *a, **k: []) + path = tmp_path / "main.tf.json" + path.write_text("{}", encoding="utf-8") + plugin = self._Plugin([{"dataset": "d", "revoked": [], "blocked": True}]) + assert ( + engine._reconcile_with_state(plugin, path, str(tmp_path), {}, logging.getLogger("t")) + == [] + ) + class TestLoadContract: def test_loads_yaml_contract(self): @@ -192,6 +260,8 @@ def test_workdir_includes_contract_id_segment(self, monkeypatch, tmp_path): # subprocess shells so this test keeps focusing on the workdir # layout invariant it owns. monkeypatch.setattr(engine.runner, "tofu_state_list", lambda *a, **k: []) + # The GCP plugin's one-time access-list reconciliation reads the state too. + monkeypatch.setattr(engine.runner, "tofu_state_resources", lambda *a, **k: []) monkeypatch.setattr( engine.runner, "tofu_import", diff --git a/tests/iac/test_iac_aws_column_restrictions.py b/tests/iac/test_iac_aws_column_restrictions.py new file mode 100644 index 00000000..bc775c1c --- /dev/null +++ b/tests/iac/test_iac_aws_column_restrictions.py @@ -0,0 +1,532 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""``policy.authz.columnRestrictions`` on AWS: Lake Formation's excluded columns. + +The base contract's column restriction is the same field the GCP emitter turns +into policy tags; on AWS each Lake Formation grant with ``SELECT`` excludes the +restricted columns its principal may not read. The overlay's hand-written +``governance.lakeFormation.grants[].excludedColumns`` keeps working: with no +restriction it is emitted as before, and with one it must agree. + +The rendering tests need nothing. The moto tests run a real ``tofu plan`` / +``apply`` (moto's ``ThreadedMotoServer`` for S3, Glue, STS and Lake Formation, no +account, no credentials) and ``fluid verify``'s Lake Formation check against what +the apply granted, then against a grant made outside the contract. moto stores +Lake Formation permissions but enforces none of them, so that a denied principal's +Athena query is refused rests on the Lake Formation documentation. +""" + +from __future__ import annotations + +import json +import subprocess +import time +from pathlib import Path +from typing import Any, Dict, Iterator, List, Optional + +import pytest + +from fluid_build.cli._verify_lf_columns import column_restrictions_dimension +from fluid_build.iac import build_module, get_iac_plugin, runner +from fluid_build.iac.base import UnsupportedBindingError +from fluid_build.iac.column_access import lf_expected_exclusions +from fluid_build.iac.credentials import build_tofu_env +from fluid_build.iac.governance_validation import validate_governance + +ACCOUNT = "123456789012" # moto's +REGION = "us-east-1" +STEWARD = f"arn:aws:iam::{ACCOUNT}:role/fluid-demo-lab-steward" +ANALYST = f"arn:aws:iam::{ACCOUNT}:role/fluid-demo-lab-analyst" +LOGICAL_ANALYSTS = "group:analysts@northwind.example" +DENY = [{"principal": LOGICAL_ANALYSTS, "columns": ["customer_id", "msisdn"], "access": "deny"}] + + +def _contract( + *, + restrictions: Optional[List[Dict[str, Any]]] = None, + principals: Optional[Dict[str, Any]] = None, + analyst_excluded: Optional[List[str]] = None, + analyst_columns: Optional[List[str]] = None, + lake_formation: bool = True, + fmt: str = "parquet", +) -> Dict[str, Any]: + analyst: Dict[str, Any] = {"principal": ANALYST, "permissions": ["SELECT", "DESCRIBE"]} + if analyst_excluded is not None: + analyst["excludedColumns"] = analyst_excluded + if analyst_columns is not None: + analyst["columns"] = analyst_columns + binding: Dict[str, Any] = { + "platform": "aws", + "format": fmt, + "location": { + "database": "demo_gold", + "table": "retention_candidates", + "bucket": "cr-moto-lake", + "path": "gold/retention_candidates/", + }, + } + if lake_formation: + binding["governance"] = { + "lakeFormation": { + "registerLocation": True, + "grants": [ + {"principal": STEWARD, "permissions": ["SELECT", "DESCRIBE"]}, + analyst, + ], + } + } + if principals is not None: + binding["principals"] = principals + exposure: Dict[str, Any] = { + "exposeId": "candidates", + "binding": binding, + "contract": { + "schema": [ + {"name": "customer_id", "type": "string"}, + {"name": "status", "type": "string"}, + {"name": "msisdn", "type": "string"}, + ] + }, + } + if restrictions is not None: + exposure["policy"] = {"authz": {"columnRestrictions": restrictions}} + return {"fluidVersion": "0.7.6", "id": "cr.moto", "exposes": [exposure]} + + +def _grants(contract: Dict[str, Any]) -> Dict[str, Dict[str, Any]]: + res = get_iac_plugin("aws").emit(contract) + return {g["principal"]: g for g in res["aws_lakeformation_permissions"].values()} + + +def _excluded(grant: Dict[str, Any]) -> Optional[List[str]]: + twc = grant.get("table_with_columns") + return twc[0].get("excluded_column_names") if twc else None + + +def _refusal(contract: Dict[str, Any]) -> UnsupportedBindingError: + with pytest.raises(UnsupportedBindingError) as raised: + get_iac_plugin("aws").emit(contract) + return raised.value + + +MAPPED = {LOGICAL_ANALYSTS: ANALYST} + + +class TestRendering: + def test_a_restriction_becomes_the_grants_excluded_columns(self): + grants = _grants(_contract(restrictions=DENY, principals=MAPPED)) + assert _excluded(grants[ANALYST]) == ["customer_id", "msisdn"] + # A derived exclusion makes the grant column-limited, so it carries SELECT + # alone like a hand-written one (#679): Lake Formation refuses DESCRIBE to + # a principal holding a partial SELECT. + assert grants[ANALYST]["permissions"] == ["SELECT"] + # The steward may read everything: a plain table grant, as before. + assert "table" in grants[STEWARD] and "table_with_columns" not in grants[STEWARD] + assert grants[STEWARD]["permissions"] == ["SELECT", "DESCRIBE"] + + def test_the_hand_written_exclusion_keeps_working_without_a_restriction(self): + grants = _grants(_contract(analyst_excluded=["customer_id", "msisdn"])) + assert _excluded(grants[ANALYST]) == ["customer_id", "msisdn"] + + def test_both_present_and_agreeing_emit_once(self): + grants = _grants( + _contract( + restrictions=DENY, principals=MAPPED, analyst_excluded=["msisdn", "customer_id"] + ) + ) + assert sorted(_excluded(grants[ANALYST]) or []) == ["customer_id", "msisdn"] + + @pytest.mark.parametrize("excluded", [["msisdn"], ["customer_id", "msisdn", "status"], []]) + def test_both_present_and_disagreeing_is_refused(self, excluded): + error = _refusal(_contract(restrictions=DENY, principals=MAPPED, analyst_excluded=excluded)) + assert error.kind == "column-restriction-conflict" + assert "must agree" in str(error) + + def test_a_projection_that_reaches_a_restricted_column_is_refused(self): + error = _refusal( + _contract(restrictions=DENY, principals=MAPPED, analyst_columns=["status", "msisdn"]) + ) + assert error.kind == "column-restriction-conflict" + + def test_a_projection_that_avoids_them_is_kept(self): + grants = _grants( + _contract(restrictions=DENY, principals=MAPPED, analyst_columns=["status"]) + ) + assert grants[ANALYST]["table_with_columns"][0]["column_names"] == ["status"] + + def test_an_allow_list_excludes_the_column_for_everyone_else(self): + restrictions = [{"principal": "steward", "columns": ["msisdn"], "access": "allow"}] + grants = _grants(_contract(restrictions=restrictions, principals={"steward": STEWARD})) + assert _excluded(grants[ANALYST]) == ["msisdn"] + assert "table" in grants[STEWARD] + + def test_a_restriction_may_name_an_arn_without_a_mapping(self): + restrictions = [{"principal": ANALYST, "columns": ["msisdn"], "access": "deny"}] + grants = _grants(_contract(restrictions=restrictions)) + assert _excluded(grants[ANALYST]) == ["msisdn"] + + def test_an_unmapped_logical_principal_is_refused(self): + assert _refusal(_contract(restrictions=DENY)).kind == "principal-invalid" + error = _refusal(_contract(restrictions=DENY, principals={"someone-else": ANALYST})) + assert error.kind == "principal-unmapped" + + def test_a_restriction_nothing_enforces_is_refused(self): + error = _refusal(_contract(restrictions=DENY, principals=MAPPED, lake_formation=False)) + assert error.kind == "column-restriction-unenforceable" + + def test_a_restriction_on_a_non_glue_binding_is_refused(self): + contract = _contract(restrictions=DENY, principals=MAPPED, fmt="kinesis_stream") + assert _refusal(contract).kind == "column-restriction-unenforceable" + + def test_a_restriction_on_a_binding_with_no_glue_table_is_refused(self): + contract = _contract(restrictions=DENY, principals=MAPPED) + del contract["exposes"][0]["binding"]["location"]["table"] + assert _refusal(contract).kind == "column-restriction-unenforceable" + errors, _ = validate_governance(contract) + assert any("names no Glue table" in e for e in errors), errors + + # A grant the restrictions leave with no column. Hand-written as + # excludedColumns it was refused ("excludes every column of the table"), but + # derived from columnRestrictions it was emitted: a column wildcard that + # excludes every column (measured on the integration with 0.16.6's #675). + ALL_COLUMNS = ["customer_id", "status", "msisdn"] + + @pytest.mark.parametrize( + "restrictions, principals", + [ + ([{"principal": LOGICAL_ANALYSTS, "columns": ALL_COLUMNS, "access": "deny"}], MAPPED), + ([{"principal": STEWARD, "columns": ALL_COLUMNS, "access": "allow"}], None), + ], + ids=["denied-every-column", "allowed-to-someone-else-only"], + ) + def test_restrictions_that_leave_a_reader_no_column_are_refused(self, restrictions, principals): + contract = _contract(restrictions=restrictions, principals=principals) + error = _refusal(contract) + assert error.kind == "lakeformation-grant-columns" + assert f"grants[1] gives {ANALYST} read access" in str(error) + assert "read no column of the table" in str(error) + errors, _ = validate_governance(contract) + assert any("read no column of the table" in e for e in errors), errors + + def test_restrictions_that_leave_a_reader_one_column_still_emit(self): + deny = [{"principal": LOGICAL_ANALYSTS, "columns": self.ALL_COLUMNS[:2], "access": "deny"}] + grants = _grants(_contract(restrictions=deny, principals=MAPPED)) + twc = grants[ANALYST]["table_with_columns"][0] + assert twc["wildcard"] is True + assert twc["excluded_column_names"] == ["customer_id", "status"] + + def test_the_grant_check_reads_the_derived_exclusions(self): + from fluid_build.iac.providers.aws import _check_lf_grant_columns + + contract = _contract() + exposure = contract["exposes"][0] + binding = exposure["binding"] + schema = exposure["contract"]["schema"] + with pytest.raises(UnsupportedBindingError) as raised: + _check_lf_grant_columns( + binding, binding["location"], "parquet", schema, {1: tuple(self.ALL_COLUMNS)} + ) + assert "excludes every column of the table" in str(raised.value) + with pytest.raises(UnsupportedBindingError) as raised: + _check_lf_grant_columns(binding, binding["location"], "parquet", schema, {1: ("x",)}) + assert "['x']" in str(raised.value) + + def test_fluid_validate_reports_the_refusal(self): + errors, _ = validate_governance( + _contract(restrictions=DENY, principals=MAPPED, analyst_excluded=["msisdn"]) + ) + assert any("must agree" in e for e in errors) + + def test_verify_holds_every_denied_principal_and_iam_allowed_principals(self): + exposure = _contract(restrictions=DENY, principals=MAPPED)["exposes"][0] + expected = lf_expected_exclusions(exposure, exposure["binding"]) + assert expected == { + ANALYST: ("customer_id", "msisdn"), + "IAM_ALLOWED_PRINCIPALS": ("customer_id", "msisdn"), + } + + +# ── against moto: a real tofu plan / apply, and verify's LF check ──────── + + +def _have_moto_server() -> bool: + try: + from moto.server import ThreadedMotoServer # noqa: F401 + + return True + except Exception: # noqa: BLE001 — Flask (the `server` extra) may be absent + return False + + +_SKIP = runner.tofu_path() is None or not _have_moto_server() +_SKIP_REASON = "needs `tofu` on PATH + moto server extra (pip install 'moto[glue,server]')" + + +@pytest.fixture +def moto_endpoint() -> Iterator[str]: + import urllib.request + + from moto.server import ThreadedMotoServer + + server = ThreadedMotoServer(port=0, verbose=False) + server.start() + try: + _, port = server.get_host_and_port() + endpoint = f"http://127.0.0.1:{port}" + reset = urllib.request.Request(f"{endpoint}/moto-api/reset", method="POST") + with urllib.request.urlopen(reset, timeout=30) as answered: # noqa: S310 — local moto + assert answered.status == 200 + yield endpoint + finally: + server.stop() + + +@pytest.fixture(scope="module") +def tofu_env(tmp_path_factory: pytest.TempPathFactory) -> Dict[str, str]: + env = {k: v for k, v in build_tofu_env().items() if not k.startswith("AWS_")} + env.setdefault("TF_PLUGIN_CACHE_DIR", str(tmp_path_factory.mktemp("tofu-plugin-cache"))) + env["AWS_CONFIG_FILE"] = "/dev/null" + env["AWS_SHARED_CREDENTIALS_FILE"] = "/dev/null" + env["AWS_EC2_METADATA_DISABLED"] = "true" + return env + + +def _tofu(workdir: Path, env: Dict[str, str], *args: str) -> "subprocess.CompletedProcess[str]": + return subprocess.run( + [str(runner.tofu_path()), *args], + cwd=workdir, + env=env, + capture_output=True, + text=True, + timeout=600, + ) + + +def _write(contract: Dict[str, Any], workdir: Path, endpoint: str) -> None: + workdir.mkdir(parents=True, exist_ok=True) + (workdir / "main.tf.json").write_text(build_module(get_iac_plugin("aws"), contract)) + services = ("s3", "sts", "iam", "glue", "lakeformation", "kms") + provider = { + "provider": { + "aws": { + "region": REGION, + "access_key": "testing", + "secret_key": "testing", # pragma: allowlist secret — moto dummy + "skip_credentials_validation": True, + "skip_metadata_api_check": True, + "skip_requesting_account_id": True, + "s3_use_path_style": True, + "endpoints": {svc: endpoint for svc in services}, + } + } + } + (workdir / "provider.tf.json").write_text(json.dumps(provider)) + + +def _init(workdir: Path, env: Dict[str, str]) -> None: + for attempt in range(4): + done = _tofu(workdir, env, "init", "-backend=false", "-input=false", "-no-color") + if done.returncode == 0 or "unable to acquire file lock" not in done.stdout + done.stderr: + break + time.sleep(2 * (attempt + 1)) + assert done.returncode == 0, f"tofu init failed:\n{done.stdout}\n{done.stderr}" + + +def _plan(workdir: Path, env: Dict[str, str]) -> Dict[str, Dict[str, Any]]: + done = _tofu(workdir, env, "plan", "-input=false", "-no-color", "-out=plan.bin") + assert done.returncode == 0, f"tofu plan failed:\n{done.stdout}\n{done.stderr}" + shown = _tofu(workdir, env, "show", "-json", "plan.bin") + return {c["address"]: c for c in json.loads(shown.stdout)["resource_changes"]} + + +def _planned_grant(planned: Dict[str, Dict[str, Any]], principal: str) -> Dict[str, Any]: + return next( + c["change"]["after"] + for c in planned.values() + if c["type"] == "aws_lakeformation_permissions" + and c["change"]["after"]["principal"] == principal + ) + + +def _boto(service: str, endpoint: str) -> Any: + import boto3 + + return boto3.client( + service, + endpoint_url=endpoint, + aws_access_key_id="testing", + aws_secret_access_key="testing", # noqa: S106 # pragma: allowlist secret — moto dummy + region_name=REGION, + ) + + +def _check(contract: Dict[str, Any], endpoint: str) -> Dict[str, Any]: + exposure = contract["exposes"][0] + dimension = column_restrictions_dimension( + "candidates", + exposure, + exposure["binding"], + region=REGION, + factory=lambda service, _region: _boto(service, endpoint), + ) + assert dimension is not None + return dimension + + +@pytest.mark.skipif(_SKIP, reason=_SKIP_REASON) +@pytest.mark.parametrize("source", ["restriction", "hand-written"]) +def test_the_excluded_columns_plan_through_the_provider(tmp_path, moto_endpoint, tofu_env, source): + """A real ``tofu plan`` accepts the grant, from the restriction or the overlay. + + The overlay's own ``excludedColumns`` failed this plan before ("one of + column_names, wildcard must be specified"): the provider needs ``wildcard`` + beside ``excluded_column_names``, and the emitter never wrote it. + """ + contract = ( + _contract(restrictions=DENY, principals=MAPPED) + if source == "restriction" + else _contract(analyst_excluded=["customer_id", "msisdn"]) + ) + _write(contract, tmp_path, moto_endpoint) + _init(tmp_path, tofu_env) + planned = _plan(tmp_path, tofu_env) + twc = _planned_grant(planned, ANALYST)["table_with_columns"][0] + assert sorted(twc["excluded_column_names"]) == ["customer_id", "msisdn"] + assert twc["wildcard"] is True + steward = _planned_grant(planned, STEWARD) + assert steward["table"] and not steward.get("table_with_columns") + + +def _lf_table(**columns: Any) -> Dict[str, Any]: + ref: Dict[str, Any] = {"DatabaseName": "demo_gold", "Name": "retention_candidates"} + if not columns: + return {"Table": ref} + return {"TableWithColumns": {**ref, **columns}} + + +def _grant_as_apply_would(lf: Any) -> None: + """The two grants the emitted module makes, made through the API.""" + lf.grant_permissions( + Principal={"DataLakePrincipalIdentifier": STEWARD}, + Resource=_lf_table(), + Permissions=["SELECT", "DESCRIBE"], + ) + lf.grant_permissions( + Principal={"DataLakePrincipalIdentifier": ANALYST}, + Resource=_lf_table(ColumnWildcard={"ExcludedColumnNames": ["customer_id", "msisdn"]}), + Permissions=["SELECT"], + ) + + +@pytest.mark.skipif(not _have_moto_server(), reason=_SKIP_REASON) +def test_verify_reads_lake_formation_and_fails_on_a_grant_outside_the_contract(moto_endpoint): + """moto's Lake Formation stores grants; verify reads them as the real API returns them. + + (moto cannot read back a column-level grant for the provider, so the grants + here are made through the API directly, as the apply would.) + """ + contract = _contract(restrictions=DENY, principals=MAPPED) + lf = _boto("lakeformation", moto_endpoint) + _grant_as_apply_would(lf) + passed = _check(contract, moto_endpoint) + assert passed["status"] == "pass", passed + assert set(passed["checked"]) == {ANALYST, "IAM_ALLOWED_PRINCIPALS"} + + # A column list reaching a denied column, granted outside the contract. + lf.grant_permissions( + Principal={"DataLakePrincipalIdentifier": ANALYST}, + Resource=_lf_table(ColumnNames=["status", "msisdn"]), + Permissions=["SELECT"], + ) + failed = _check(contract, moto_endpoint) + assert failed["status"] == "fail" + assert f"{ANALYST} can read msisdn" in failed["message"] + + +@pytest.mark.skipif(not _have_moto_server(), reason=_SKIP_REASON) +def test_verify_fails_when_the_table_still_grants_iam_allowed_principals(moto_endpoint): + """Lake Formation's default for a new table lets any IAM principal read every column.""" + contract = _contract(restrictions=DENY, principals=MAPPED) + lf = _boto("lakeformation", moto_endpoint) + _grant_as_apply_would(lf) + lf.grant_permissions( + Principal={"DataLakePrincipalIdentifier": "IAM_ALLOWED_PRINCIPALS"}, + Resource=_lf_table(), + Permissions=["ALL"], + ) + failed = _check(contract, moto_endpoint) + assert failed["status"] == "fail" + assert "IAM_ALLOWED_PRINCIPALS can read customer_id, msisdn" in failed["message"] + + +@pytest.mark.skipif(not _have_moto_server(), reason=_SKIP_REASON) +def test_verify_errors_when_it_cannot_see_the_contracts_own_grants(moto_endpoint): + """A caller that sees only part of the permissions must not read as a pass.""" + contract = _contract(restrictions=DENY, principals=MAPPED) + lf = _boto("lakeformation", moto_endpoint) + lf.grant_permissions( + Principal={"DataLakePrincipalIdentifier": ANALYST}, + Resource=_lf_table(ColumnWildcard={"ExcludedColumnNames": ["customer_id", "msisdn"]}), + Permissions=["SELECT"], + ) + unseen = _check(contract, moto_endpoint) + assert unseen["status"] == "error" + assert STEWARD in unseen["message"] and "Lake Formation administrator" in unseen["message"] + + +class _ListedPermissions: + """A Lake Formation client whose ListPermissions returns what it is given.""" + + def __init__(self, entries: List[Dict[str, Any]]) -> None: + self.entries = entries + + def list_permissions(self, **_kwargs: Any) -> Dict[str, Any]: + return {"PrincipalResourcePermissions": self.entries} + + +def _entry(principal: str, resource: Dict[str, Any], *permissions: str) -> Dict[str, Any]: + return { + "Principal": {"DataLakePrincipalIdentifier": principal}, + "Resource": resource, + "Permissions": list(permissions or ("SELECT",)), + } + + +def test_a_database_wide_select_to_a_denied_principal_fails_verify(): + """``Table: {TableWildcard: {}}`` reaches every column of every table in the database.""" + contract = _contract(restrictions=DENY, principals=MAPPED) + exposure = contract["exposes"][0] + own = [ + _entry(STEWARD, _lf_table(), "SELECT", "DESCRIBE"), + _entry( + ANALYST, _lf_table(ColumnWildcard={"ExcludedColumnNames": ["customer_id", "msisdn"]}) + ), + ] + + def check(extra: List[Dict[str, Any]]) -> Dict[str, Any]: + client = _ListedPermissions(own + extra) + dimension = column_restrictions_dimension( + "candidates", exposure, exposure["binding"], region=REGION, factory=lambda *_: client + ) + assert dimension is not None + return dimension + + assert check([])["status"] == "pass" + wildcard = {"Table": {"DatabaseName": "demo_gold", "TableWildcard": {}}} + failed = check([_entry(ANALYST, wildcard)]) + assert failed["status"] == "fail" + assert f"{ANALYST} can read customer_id, msisdn" in failed["message"] + elsewhere = {"Table": {"DatabaseName": "other_db", "TableWildcard": {}}} + assert check([_entry(ANALYST, elsewhere)])["status"] == "pass" diff --git a/tests/iac/test_iac_cross_account_emit.py b/tests/iac/test_iac_cross_account_emit.py index 4638cb6a..49f3c065 100644 --- a/tests/iac/test_iac_cross_account_emit.py +++ b/tests/iac/test_iac_cross_account_emit.py @@ -238,10 +238,14 @@ def test_schema_validates_lf_grant_without_extra_fields(self): class TestGcpCrossProjectEmit: """Cross-project SA access uses the existing ``metadata.policies`` - surface — no new schema fields. The plugin's ``_bq_access_entries`` - helper maps each policy entry to a ``user_by_email`` row on the - dataset's ``access[]`` block, and BQ accepts cross-project SA - emails via the ``user_by_email`` field.""" + surface — no new schema fields. Each policy entry becomes a + non-authoritative ``google_bigquery_dataset_iam_member`` (it was a row of + the dataset's authoritative ``access[]`` block), and BigQuery accepts a + service account of another project as a dataset member.""" + + @staticmethod + def _members(res): + return {m["member"] for m in (res.get("google_bigquery_dataset_iam_member") or {}).values()} def test_cross_project_sa_lands_in_dataset_access(self): contract = _gcp_contract( @@ -255,12 +259,10 @@ def test_cross_project_sa_lands_in_dataset_access(self): } ) res = get_iac_plugin("gcp").emit(contract, []) - ds = next(iter(res["google_bigquery_dataset"].values())) - access = ds.get("access") or [] - emails = {e.get("user_by_email") for e in access if "user_by_email" in e} + members = self._members(res) assert ( - "consumer@other-project.iam.gserviceaccount.com" in emails - ), f"cross-project SA not in dataset.access[] — got {access}" + "serviceAccount:consumer@other-project.iam.gserviceaccount.com" in members + ), f"cross-project SA is not a dataset member — got {members}" def test_multiple_principals_emit_multiple_access_entries(self): contract = _gcp_contract( @@ -275,19 +277,18 @@ def test_multiple_principals_emit_multiple_access_entries(self): } ) res = get_iac_plugin("gcp").emit(contract, []) - ds = next(iter(res["google_bigquery_dataset"].values())) - emails = {e.get("user_by_email") for e in ds.get("access", []) if "user_by_email" in e} - assert emails == { - "consumer-a@p1.iam.gserviceaccount.com", - "consumer-b@p2.iam.gserviceaccount.com", + assert self._members(res) == { + "serviceAccount:consumer-a@p1.iam.gserviceaccount.com", + "serviceAccount:consumer-b@p2.iam.gserviceaccount.com", } def test_no_policies_no_access_block(self): contract = _gcp_contract({}) res = get_iac_plugin("gcp").emit(contract, []) ds = next(iter(res["google_bigquery_dataset"].values())) - # No access[] when no policies — existing behaviour preserved. + # No access[] and no members when no policies — existing behaviour preserved. assert "access" not in ds + assert not self._members(res) # Note: ``metadata.policies`` is read by the existing GCP plugin # (``_bq_access_entries`` → dataset.access[] block) but is NOT in diff --git a/tests/iac/test_iac_cross_project_gcp_emulator.py b/tests/iac/test_iac_cross_project_gcp_emulator.py index 6ed31326..1edbcaa9 100644 --- a/tests/iac/test_iac_cross_project_gcp_emulator.py +++ b/tests/iac/test_iac_cross_project_gcp_emulator.py @@ -138,9 +138,9 @@ def _xproj_contract(dataset: str, *consumer_sas: str) -> Dict[str, Any]: """A producer-project contract granting BQ read to SAs in other projects. Cross-project access rides the existing ``metadata.policies`` surface — - the GCP plugin's ``_bq_access_entries`` maps each principal to a - ``user_by_email`` row on the dataset's ``access[]`` block. Zero new - schema fields. + the GCP plugin makes each principal a ``google_bigquery_dataset_iam_member``, + which the provider adds to the dataset's ``access[]`` as a ``userByEmail`` + row. Zero new schema fields. """ return { "fluidVersion": "0.7.6", @@ -178,12 +178,35 @@ def _xproj_contract(dataset: str, *consumer_sas: str) -> Dict[str, Any]: } +#: The basic dataset roles, as BigQuery reports them in ``access[]`` +#: (terraform-provider-google ``iam_bigquery_dataset.go``: the API changes these +#: IAM roles to the legacy names). +_LEGACY_ROLES = { + "roles/bigquery.dataOwner": "OWNER", + "roles/bigquery.dataEditor": "WRITER", + "roles/bigquery.dataViewer": "READER", +} + + def _emitted_access_block(contract: Mapping[str, Any]) -> List[Dict[str, str]]: - """Run the real GCP emitter and return the dataset's ``access[]`` block.""" + """Run the real GCP emitter; the ``access[]`` rows its dataset grants become. + + The emitter writes each grant as a non-authoritative + ``google_bigquery_dataset_iam_member`` (it used to be a row of the + dataset's authoritative ``access[]`` block). The provider adds each member + to the dataset's access list, as ``userByEmail`` for ``user:`` and + ``serviceAccount:`` and ``groupByEmail`` for ``group:`` + (``iamMemberToAccess``); this replays that mapping in the provider's + snake_case, so the round-trip below still drives the emitter's own output. + """ module = json.loads(build_module(get_iac_plugin("gcp"), contract)) - datasets = module["resource"]["google_bigquery_dataset"] - body = next(iter(datasets.values())) - return body.get("access", []) + members = module["resource"].get("google_bigquery_dataset_iam_member") or {} + rows = [] + for body in members.values(): + kind, _, email = body["member"].partition(":") + field = "group_by_email" if kind == "group" else "user_by_email" + rows.append({"role": _LEGACY_ROLES.get(body["role"], body["role"]), field: email}) + return rows def _tf_access_to_bq_api(entries: List[Dict[str, str]]) -> List[Dict[str, str]]: diff --git a/tests/iac/test_iac_gcp.py b/tests/iac/test_iac_gcp.py index 0cd3cbdd..3a0585f6 100644 --- a/tests/iac/test_iac_gcp.py +++ b/tests/iac/test_iac_gcp.py @@ -216,7 +216,7 @@ class TestGcpIam: _POLICIES = { "policies": { "analysts": { - "principals": ["alice@example.com", "data-team"], + "principals": ["alice@example.com", "data-team@corp-a.com"], "permissions": ["read"], }, "writers": { @@ -248,11 +248,46 @@ def test_bigquery_dataset_gets_access_entries(self): ] ) ) - access = next(iter(res["google_bigquery_dataset"].values()))["access"] - assert {e["role"] for e in access} == {"READER", "WRITER"} - # '@' -> user_by_email; bare name -> group_by_email - assert {"role": "READER", "user_by_email": "alice@example.com"} in access - assert {"role": "READER", "group_by_email": "data-team"} in access + # Non-authoritative members, not the dataset's authoritative ``access`` + # list, which replaced every entry the dataset had. + assert "access" not in next(iter(res["google_bigquery_dataset"].values())) + members = res["google_bigquery_dataset_iam_member"].values() + assert {m["role"] for m in members} == { + "roles/bigquery.dataViewer", + "roles/bigquery.dataEditor", + } + pairs = {(m["role"], m["member"]) for m in members} + # '@' -> user, as the legacy surface always inferred (it cannot tell a + # group address from a user's; accessPolicy declares the type). + assert ("roles/bigquery.dataViewer", "user:alice@example.com") in pairs + assert ("roles/bigquery.dataViewer", "user:data-team@corp-a.com") in pairs + + @pytest.mark.parametrize( + "binding", + [ + {"format": "bigquery_table", "location": {"dataset": "d", "table": "t"}}, + {"format": "gcs_bucket", "location": {"bucket": "my-bucket"}}, + ], + ids=["bigquery", "gcs"], + ) + def test_a_bare_legacy_name_is_refused_not_emitted(self, binding): + """A bare name was inferred ``group:data-team``, which no cloud accepts. + + BigQuery refuses a ``groupByEmail`` that is not an address and IAM a + ``group:`` member without one, so the grant never applied; it is refused + at plan now, with the fix, instead of failing at apply. + """ + from fluid_build.iac.base import UnsupportedBindingError + + contract = { + "id": "analytics.demo", + "metadata": {"policies": {"a": {"principals": ["data-team"], "permissions": ["read"]}}}, + "exposes": [{"exposeId": "t", "binding": binding}], + } + with pytest.raises(UnsupportedBindingError) as caught: + _gcp().emit(contract) + assert caught.value.kind == "principal-placeholder" + assert "group:data-team" in str(caught.value) def test_gcs_bucket_gets_iam_members(self): res = _gcp().emit( @@ -272,7 +307,7 @@ def test_gcs_bucket_gets_iam_members(self): } member_strs = {m["member"] for m in members} assert "user:alice@example.com" in member_strs - assert "group:data-team" in member_strs + assert "user:data-team@corp-a.com" in member_strs assert "serviceAccount:svc@proj.iam.gserviceaccount.com" in member_strs def test_no_policies_means_no_iam(self): @@ -290,6 +325,7 @@ def test_no_policies_means_no_iam(self): ) ) assert "access" not in next(iter(res["google_bigquery_dataset"].values())) + assert "google_bigquery_dataset_iam_member" not in res assert "google_storage_bucket_iam_member" not in res diff --git a/tests/iac/test_iac_gcp_governance.py b/tests/iac/test_iac_gcp_governance.py new file mode 100644 index 00000000..90954ed7 --- /dev/null +++ b/tests/iac/test_iac_gcp_governance.py @@ -0,0 +1,1267 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""GCP governance: retention, CMEK, column restrictions and mapped principals, rendered. + +Before this, a gcp binding's ``lifecycle.expire``, ``binding.encryption`` and +``policy.authz.columnRestrictions`` produced no resource and no warning (the +module was byte-identical with and without them), and the contract's principals +were written as given into the dataset's AUTHORITATIVE ``access`` list. These +tests pin what the emitter writes now, what it refuses, and that ``tofu validate`` +accepts each shape (``test_iac_gcp_governance_plan.py`` runs the real provider's +plan on a live table). +""" + +from __future__ import annotations + +import copy +import json +import shutil +import subprocess +from typing import Any, Dict, List, Optional + +import pytest + +from fluid_build.iac import build_module, get_iac_plugin +from fluid_build.iac.base import UnsupportedBindingError +from fluid_build.iac.governance_validation import validate_governance +from fluid_build.iac.providers import gcp_governance as gov + +CID = "gold_retention_candidates" +TABLE = f"google_bigquery_table.{CID}_retention_candidates" +PLATFORM = "group:data-platform@northwind.example" +ANALYSTS = "group:analysts@northwind.example" +PIPELINE = "serviceAccount:fluid-pipeline@northwind.example" +MAPPING = { + PLATFORM: "group:data-platform@northwind.com", + ANALYSTS: "group:analysts@northwind.com", + PIPELINE: "serviceAccount:fluid-pipeline@northwind-demo.iam.gserviceaccount.com", +} + + +def _contract( + *, + principals: Optional[Dict[str, Any]] = None, + lifecycle: Optional[Dict[str, Any]] = None, + encryption: Optional[Dict[str, Any]] = None, + restrictions: Optional[List[Dict[str, Any]]] = None, + partition_by: Optional[List[str]] = None, + fmt: str = "bigquery_table", + region: str = "europe-west1", +) -> Dict[str, Any]: + location: Dict[str, Any] = { + "project": "northwind-demo", + "dataset": "demo_gold", + "table": "retention_candidates", + "region": region, + } + if partition_by is not None: + location["partitionBy"] = partition_by + binding: Dict[str, Any] = {"platform": "gcp", "format": fmt, "location": location} + if principals is not None: + binding["principals"] = principals + if encryption is not None: + binding["encryption"] = encryption + exposure: Dict[str, Any] = { + "exposeId": "candidates", + "binding": binding, + "contract": { + "schema": [ + {"name": "customer_id", "type": "VARCHAR", "required": True}, + {"name": "status", "type": "VARCHAR"}, + {"name": "msisdn", "type": "VARCHAR", "description": "salted ${hash}"}, + {"name": "created_at", "type": "timestamp"}, + {"name": "cohort_size", "type": "BIGINT"}, + ] + }, + } + if lifecycle is not None: + exposure["lifecycle"] = lifecycle + if restrictions is not None: + exposure["policy"] = {"authz": {"columnRestrictions": restrictions}} + return { + "fluidVersion": "0.7.6", + "id": "gold.retention_candidates", + "accessPolicy": { + "grants": [ + {"principal": PLATFORM, "permissions": ["read", "select", "query"]}, + {"principal": ANALYSTS, "permissions": ["read"]}, + {"principal": PIPELINE, "permissions": ["read", "write", "insert"]}, + ] + }, + "exposes": [exposure], + } + + +def _module(contract: Dict[str, Any]) -> Dict[str, Any]: + return json.loads(build_module(get_iac_plugin("gcp"), contract)) + + +def _resources(contract: Dict[str, Any]) -> Dict[str, Any]: + return _module(contract).get("resource") or {} + + +def _refusal(contract: Dict[str, Any]) -> UnsupportedBindingError: + with pytest.raises(UnsupportedBindingError) as raised: + build_module(get_iac_plugin("gcp"), contract) + return raised.value + + +# ── D9: logical principals, mapped; non-authoritative dataset grants ───── + + +class TestPrincipals: + def test_the_demo_placeholders_are_refused_not_written_into_iam(self): + """The verification's finding: ``@northwind.example`` became an authoritative ACL.""" + error = _refusal(_contract()) + assert error.kind == "principal-placeholder" + assert "northwind.example" in str(error) + assert "binding.principals" in " ".join(error.remediation) + + def test_mapped_principals_become_non_authoritative_dataset_members(self): + res = _resources(_contract(principals=MAPPING)) + dataset = res["google_bigquery_dataset"][f"{CID}_demo_gold"] + assert "access" not in dataset + pairs = { + (m["role"], m["member"]) for m in res["google_bigquery_dataset_iam_member"].values() + } + assert pairs == { + ("roles/bigquery.dataViewer", "group:data-platform@northwind.com"), + ("roles/bigquery.dataViewer", "group:analysts@northwind.com"), + ( + "roles/bigquery.dataViewer", + "serviceAccount:fluid-pipeline@northwind-demo.iam.gserviceaccount.com", + ), + ( + "roles/bigquery.dataEditor", + "serviceAccount:fluid-pipeline@northwind-demo.iam.gserviceaccount.com", + ), + } + assert "northwind.example" not in json.dumps(res) + for member in res["google_bigquery_dataset_iam_member"].values(): + assert member["project"] == "northwind-demo" + assert ( + member["dataset_id"] == f"${{google_bigquery_dataset.{CID}_demo_gold.dataset_id}}" + ) + + def test_an_unmapped_principal_is_refused_once_a_binding_maps_principals(self): + mapping = {k: v for k, v in MAPPING.items() if k != ANALYSTS} + error = _refusal(_contract(principals=mapping)) + assert error.kind == "principal-unmapped" + assert ANALYSTS in str(error) + + def test_a_principal_mapped_to_nothing_is_granted_nothing(self): + res = _resources(_contract(principals={**MAPPING, ANALYSTS: []})) + members = {m["member"] for m in res["google_bigquery_dataset_iam_member"].values()} + assert "group:analysts@northwind.com" not in members + assert len(members) == 2 + + def test_a_principal_may_map_to_several_identities(self): + res = _resources( + _contract( + principals={ + **MAPPING, + PLATFORM: ["group:dp-eu@northwind.com", "group:dp-us@northwind.com"], + } + ) + ) + members = {m["member"] for m in res["google_bigquery_dataset_iam_member"].values()} + assert {"group:dp-eu@northwind.com", "group:dp-us@northwind.com"} <= members + + @pytest.mark.parametrize( + "identity", + [ + "arn:aws:iam::111111111111:role/analyst", + "data-platform@northwind.com", + "group:", + "group:not-an-email", + ], + ) + def test_a_mapped_identity_that_is_not_an_iam_member_is_refused(self, identity): + error = _refusal(_contract(principals={**MAPPING, PLATFORM: identity})) + assert error.kind == "principal-invalid" + + def test_an_env_template_in_an_identity_validates_as_it_will_emit(self): + """``fluid validate`` does not resolve ``{{ env.* }}``; apply resolves it first.""" + templated = ( + "serviceAccount:fluid-pipeline@{{ env.FLUID_DEMO_GCP_PROJECT }}.iam.gserviceaccount.com" + ) + contract = _contract(principals={**MAPPING, PIPELINE: templated}) + assert validate_governance(contract) == ([], []) + members = { + m["member"] for m in _resources(contract)["google_bigquery_dataset_iam_member"].values() + } + assert templated in members + + def test_a_placeholder_mapped_to_a_placeholder_is_still_refused(self): + error = _refusal(_contract(principals={**MAPPING, PLATFORM: "group:x@corp.test"})) + assert error.kind == "principal-placeholder" + + def test_a_contract_without_a_mapping_keeps_its_real_principals(self): + contract = _contract() + contract["accessPolicy"]["grants"] = [ + {"principal": "group:analysts@company.com", "permissions": ["read"]} + ] + res = _resources(contract) + (member,) = res["google_bigquery_dataset_iam_member"].values() + assert member["member"] == "group:analysts@company.com" + + def test_a_logical_key_matches_whatever_case_its_prefix_is_written_in(self): + contract = _contract( + principals={ + "ServiceAccount:fluid-pipeline@northwind.example": MAPPING[PIPELINE], + PLATFORM: MAPPING[PLATFORM], + ANALYSTS: MAPPING[ANALYSTS], + } + ) + assert _resources(contract)["google_bigquery_dataset_iam_member"] + + def test_gcs_grants_are_mapped_too(self): + contract = _contract(principals=MAPPING, fmt="gcs_bucket") + contract["exposes"][0]["binding"]["location"] = {"bucket": "northwind-demo-lake"} + res = _resources(contract) + members = {m["member"] for m in res["google_storage_bucket_iam_member"].values()} + assert "group:analysts@northwind.com" in members + assert "northwind.example" not in json.dumps(res) + + def test_a_pubsub_expose_carries_no_grant_and_is_not_refused(self): + contract = _contract(fmt="pubsub_topic") + contract["exposes"][0]["binding"]["location"] = {"topic": "events"} + assert "google_pubsub_topic" in _resources(contract) + + +# ── F7: retention as partition expiration ──────────────────────────────── + + +class TestRetention: + def test_expire_becomes_daily_ingestion_time_partitions_that_expire(self): + res = _resources( + _contract(principals=MAPPING, lifecycle={"retention": "P90D", "expire": True}) + ) + table = res["google_bigquery_table"][f"{CID}_retention_candidates"] + assert table["time_partitioning"] == {"type": "DAY", "expiration_ms": 90 * 86_400_000} + # Never a whole-table TTL, which would delete the product. + assert "expiration_time" not in table + trigger = f"{CID}_retention_candidates_partitioning" + assert table["lifecycle"] == {"replace_triggered_by": [f"terraform_data.{trigger}"]} + assert res["terraform_data"][trigger] == {"input": {"type": "DAY", "field": ""}} + + def test_a_declared_partition_column_is_the_partition_field(self): + res = _resources( + _contract( + principals=MAPPING, + lifecycle={"retention": "P7Y", "expire": True}, + partition_by=["created_at"], + ) + ) + table = res["google_bigquery_table"][f"{CID}_retention_candidates"] + assert table["time_partitioning"] == { + "type": "DAY", + "field": "created_at", + "expiration_ms": 7 * 365 * 86_400_000, + } + (trigger,) = res["terraform_data"].values() + assert trigger["input"] == {"type": "DAY", "field": "created_at"} + + def test_the_trigger_holds_the_shape_not_the_period(self): + """A new retention must be an in-place change, not a replacement.""" + a = _resources( + _contract(principals=MAPPING, lifecycle={"retention": "P30D", "expire": True}) + ) + b = _resources( + _contract(principals=MAPPING, lifecycle={"retention": "P90D", "expire": True}) + ) + assert a["terraform_data"] == b["terraform_data"] + + def test_a_part_day_rounds_up_so_nothing_expires_early(self): + res = _resources( + _contract(principals=MAPPING, lifecycle={"retention": "PT36H", "expire": True}) + ) + table = res["google_bigquery_table"][f"{CID}_retention_candidates"] + assert table["time_partitioning"]["expiration_ms"] == 2 * 86_400_000 + + def test_retention_without_expire_stays_a_declaration(self): + res = _resources(_contract(principals=MAPPING, lifecycle={"retention": "P90D"})) + table = res["google_bigquery_table"][f"{CID}_retention_candidates"] + assert "time_partitioning" not in table and "lifecycle" not in table + assert "terraform_data" not in res + + @pytest.mark.parametrize( + "partition_by, why", + [ + (["status"], "not a DATE, TIMESTAMP or DATETIME"), + (["created_at", "status"], "at most one column"), + (["missing"], "not a DATE, TIMESTAMP or DATETIME"), + ], + ) + def test_an_unusable_partition_column_is_refused(self, partition_by, why): + error = _refusal( + _contract( + principals=MAPPING, + lifecycle={"retention": "P30D", "expire": True}, + partition_by=partition_by, + ) + ) + assert error.kind == "retention-partition-column" + assert why in str(error) + + @pytest.mark.parametrize("lifecycle", [{"expire": True}, {"expire": True, "retention": "30"}]) + def test_expire_without_a_usable_period_is_refused(self, lifecycle): + assert ( + _refusal(_contract(principals=MAPPING, lifecycle=lifecycle)).kind == "retention-period" + ) + + def test_expire_on_a_view_is_refused(self): + contract = _contract(principals=MAPPING, lifecycle={"retention": "P30D", "expire": True}) + contract["exposes"][0]["binding"]["format"] = "bigquery_view" + assert _refusal(contract).kind == "retention-view" + + def test_expire_on_a_gcs_expose_is_refused_not_dropped(self): + contract = _contract( + principals=MAPPING, lifecycle={"retention": "P30D", "expire": True}, fmt="gcs_bucket" + ) + contract["exposes"][0]["binding"]["location"] = {"bucket": "lake"} + error = _refusal(contract) + assert error.kind == "gcp-governance-unsupported-target" + assert "lifecycle.expire" in str(error) + + +# ── F6: customer-managed encryption ────────────────────────────────────── + + +class TestEncryption: + def test_a_product_key_is_created_granted_and_used_by_dataset_and_table(self): + contract = _contract(principals=MAPPING, encryption={"kms": "product"}) + module = _module(contract) + res = module["resource"] + ident = f"{CID}_demo_gold_kms" + assert res["google_kms_key_ring"][ident] == { + "name": "fluid-gold_retention_candidates-demo_gold", + "location": "europe-west1", + "project": "northwind-demo", + } + key = res["google_kms_crypto_key"][ident] + assert key["rotation_period"] == "7776000s" + assert key["purpose"] == "ENCRYPT_DECRYPT" + assert key["key_ring"] == f"${{google_kms_key_ring.{ident}.id}}" + grant = res["google_kms_crypto_key_iam_member"][ident] + assert grant["role"] == "roles/cloudkms.cryptoKeyEncrypterDecrypter" + assert grant["member"] == ( + f"serviceAccount:${{data.google_bigquery_default_service_account.{ident}.email}}" + ) + assert module["data"]["google_bigquery_default_service_account"][ident] == { + "project": "northwind-demo" + } + ref = f"${{google_kms_crypto_key.{ident}.id}}" + dataset = res["google_bigquery_dataset"][f"{CID}_demo_gold"] + table = res["google_bigquery_table"][f"{CID}_retention_candidates"] + assert dataset["default_encryption_configuration"] == {"kms_key_name": ref} + assert table["encryption_configuration"] == {"kms_key_name": ref} + assert table["depends_on"] == [f"google_kms_crypto_key_iam_member.{ident}"] + assert dataset["depends_on"] == [f"google_kms_crypto_key_iam_member.{ident}"] + + @pytest.mark.parametrize("region, location", [("EU", "europe"), ("US", "us")]) + def test_a_multi_region_dataset_uses_the_matching_key_location(self, region, location): + res = _resources( + _contract(principals=MAPPING, encryption={"kms": "product"}, region=region) + ) + (ring,) = res["google_kms_key_ring"].values() + assert ring["location"] == location + + def test_an_existing_key_is_named_as_given(self): + key = "projects/sec/locations/europe-west1/keyRings/shared/cryptoKeys/gold" + module = _module(_contract(principals=MAPPING, encryption={"kms": key})) + res = module["resource"] + assert "google_kms_key_ring" not in res and "data" not in module + table = res["google_bigquery_table"][f"{CID}_retention_candidates"] + assert table["encryption_configuration"] == {"kms_key_name": key} + + def test_a_region_that_is_not_a_location_never_reaches_the_key_import_id(self): + contract = _contract( + principals=MAPPING, + encryption={"kms": "product"}, + region="europe-west1/keyRings/other/../../../projects/victim", + ) + assert _refusal(contract).kind == "encryption-kms-location" + with pytest.raises(UnsupportedBindingError): + get_iac_plugin("gcp").discover_imports(contract) + + def test_an_existing_key_in_another_location_is_refused(self): + key = "projects/sec/locations/us-central1/keyRings/shared/cryptoKeys/gold" + error = _refusal(_contract(principals=MAPPING, encryption={"kms": key})) + assert error.kind == "encryption-kms-location" + + @pytest.mark.parametrize( + "kms", ["alias/fluid/gold", "arn:aws:kms:eu-north-1:111111111111:key/abc", "gold-key"] + ) + def test_an_aws_or_malformed_key_is_refused_on_gcp(self, kms): + error = _refusal(_contract(principals=MAPPING, encryption={"kms": kms})) + assert error.kind == "encryption-kms" + if kms.startswith(("alias/", "arn:")): + assert "an AWS KMS key, on a gcp binding" in str(error) + + def test_none_means_google_managed_keys(self): + res = _resources(_contract(principals=MAPPING, encryption={"kms": "none"})) + assert "google_kms_crypto_key" not in res + assert ( + "encryption_configuration" + not in res["google_bigquery_table"][f"{CID}_retention_candidates"] + ) + + def test_a_product_key_ring_and_key_are_adopted_on_re_apply(self): + """Neither can be deleted on GCP, so a re-apply after a destroy must import them.""" + blocks = get_iac_plugin("gcp").discover_imports( + _contract(principals=MAPPING, encryption={"kms": "product"}) + ) + ids = {b.to: b.id for b in blocks} + ring = "projects/northwind-demo/locations/europe-west1/keyRings/fluid-gold_retention_candidates-demo_gold" + assert ids[f"google_kms_key_ring.{CID}_demo_gold_kms"] == ring + assert ids[f"google_kms_crypto_key.{CID}_demo_gold_kms"] == f"{ring}/cryptoKeys/bigquery" + + def test_encryption_on_a_gcs_expose_is_refused_not_dropped(self): + contract = _contract(principals=MAPPING, encryption={"kms": "product"}, fmt="gcs_bucket") + contract["exposes"][0]["binding"]["location"] = {"bucket": "lake"} + assert _refusal(contract).kind == "gcp-governance-unsupported-target" + + +# ── F8: column restrictions as policy tags ─────────────────────────────── + + +class TestColumnRestrictions: + DENY = [{"principal": ANALYSTS, "columns": ["customer_id", "msisdn"], "access": "deny"}] + + def test_denied_columns_get_a_policy_tag_read_by_every_other_reader(self): + res = _resources(_contract(principals=MAPPING, restrictions=self.DENY)) + taxonomy = res["google_data_catalog_taxonomy"][f"{CID}_demo_gold_taxonomy"] + assert taxonomy["activated_policy_types"] == ["FINE_GRAINED_ACCESS_CONTROL"] + assert taxonomy["region"] == "europe-west1" + assert taxonomy["project"] == "northwind-demo" + (tag_key,) = res["google_data_catalog_policy_tag"] + tag = res["google_data_catalog_policy_tag"][tag_key] + assert tag["taxonomy"] == f"${{google_data_catalog_taxonomy.{CID}_demo_gold_taxonomy.id}}" + readers = { + (m["role"], m["member"], m["policy_tag"]) + for m in res["google_data_catalog_policy_tag_iam_member"].values() + } + ref = f"${{google_data_catalog_policy_tag.{tag_key}.name}}" + assert readers == { + ( + "roles/datacatalog.categoryFineGrainedReader", + "group:data-platform@northwind.com", + ref, + ), + ( + "roles/datacatalog.categoryFineGrainedReader", + "serviceAccount:fluid-pipeline@northwind-demo.iam.gserviceaccount.com", + ref, + ), + } + schema = json.loads(res["google_bigquery_table"][f"{CID}_retention_candidates"]["schema"]) + tagged = {f["name"]: f.get("policyTags") for f in schema} + assert tagged["customer_id"] == {"names": [ref]} + assert tagged["msisdn"] == {"names": [ref]} + assert tagged["status"] is None and tagged["cohort_size"] is None + + def test_the_schema_still_escapes_contract_text_while_the_tag_is_interpolated(self): + text = build_module( + get_iac_plugin("gcp"), _contract(principals=MAPPING, restrictions=self.DENY) + ) + schema = json.loads(text)["resource"]["google_bigquery_table"][ + f"{CID}_retention_candidates" + ]["schema"] + # A description's ${...} stays literal ($${) as everywhere else in the module. + assert "salted $${hash}" in schema + assert "${google_data_catalog_policy_tag." in schema + + def test_columns_with_different_readers_get_different_tags(self): + restrictions = [ + {"principal": ANALYSTS, "columns": ["msisdn"], "access": "deny"}, + {"principal": ANALYSTS, "columns": ["customer_id"], "access": "deny"}, + {"principal": PIPELINE, "columns": ["customer_id"], "access": "deny"}, + ] + res = _resources(_contract(principals=MAPPING, restrictions=restrictions)) + assert len(res["google_data_catalog_policy_tag"]) == 2 + + def test_an_allow_makes_its_principals_the_only_readers(self): + restrictions = [{"principal": PLATFORM, "columns": ["msisdn"], "access": "allow"}] + res = _resources(_contract(principals=MAPPING, restrictions=restrictions)) + members = {m["member"] for m in res["google_data_catalog_policy_tag_iam_member"].values()} + assert members == {"group:data-platform@northwind.com"} + + def test_a_deny_beats_an_allow(self): + restrictions = [ + {"principal": PLATFORM, "columns": ["msisdn"], "access": "allow"}, + {"principal": PLATFORM, "columns": ["msisdn"], "access": "deny"}, + ] + res = _resources(_contract(principals=MAPPING, restrictions=restrictions)) + assert "google_data_catalog_policy_tag_iam_member" not in res + assert len(res["google_data_catalog_policy_tag"]) == 1 + + def test_an_allow_never_grants_access_to_a_non_reader(self, caplog): + contract = _contract( + principals={**MAPPING, "group:outsiders@northwind.example": "group:o@northwind.com"}, + restrictions=[ + { + "principal": "group:outsiders@northwind.example", + "columns": ["msisdn"], + "access": "allow", + } + ], + ) + with caplog.at_level("WARNING"): + res = _resources(contract) + assert "google_data_catalog_policy_tag_iam_member" not in res + assert "column_restriction_allow_not_a_reader" in caplog.text + + @pytest.mark.parametrize( + "restriction, fragment", + [ + ({"principal": ANALYSTS, "columns": ["nope"], "access": "deny"}, "does not declare"), + ({"principal": ANALYSTS, "columns": ["msisdn"]}, "must be 'allow' or 'deny'"), + ({"columns": ["msisdn"], "access": "deny"}, "names no principal"), + ({"principal": ANALYSTS, "columns": [], "access": "deny"}, "must list the columns"), + ], + ) + def test_a_malformed_restriction_is_refused(self, restriction, fragment): + error = _refusal(_contract(principals=MAPPING, restrictions=[restriction])) + assert error.kind == "column-restriction" + assert fragment in str(error) + + def test_a_restriction_on_an_unmapped_principal_is_refused(self): + restrictions = [ + {"principal": "group:ghosts@northwind.example", "columns": ["msisdn"], "access": "deny"} + ] + assert _refusal(_contract(principals=MAPPING, restrictions=restrictions)).kind == ( + "principal-unmapped" + ) + + def test_a_restriction_on_a_view_is_refused(self): + contract = _contract(principals=MAPPING, restrictions=self.DENY) + contract["exposes"][0]["binding"]["format"] = "bigquery_view" + assert _refusal(contract).kind == "column-restriction-view" + + +# ── what did not change ────────────────────────────────────────────────── + + +def test_a_contract_declaring_none_of_it_emits_what_it_did_minus_the_acl(): + """No governance field: no key, no tag, no partition; only the grants moved.""" + contract = _contract(principals=MAPPING) + res = _resources(contract) + assert set(res) == { + "google_bigquery_dataset", + "google_bigquery_dataset_iam_member", + "google_bigquery_table", + } + table = res["google_bigquery_table"][f"{CID}_retention_candidates"] + assert set(table) == { + "dataset_id", + "table_id", + "labels", + "deletion_protection", + "project", + "schema", + } + + +def test_the_derivations_are_shared_with_verify(): + exposure = _contract(principals=MAPPING, restrictions=TestColumnRestrictions.DENY)["exposes"][0] + groups = gov.tag_groups(_contract(principals=MAPPING), exposure, CID, "demo_gold", "t") + assert [g.columns for g in groups] == [("customer_id", "msisdn")] + assert gov.kms_location("EU") == "europe" + assert gov.product_key_name("p", "europe-west1", "r").endswith( + "/keyRings/r/cryptoKeys/bigquery" + ) + + +def test_fluid_validate_reports_each_refusal_at_stage_2(): + errors, _ = validate_governance(_contract()) + assert any("placeholder" in e and "binding.principals" in e for e in errors) + errors, _ = validate_governance(_contract(principals=MAPPING, encryption={"kms": "alias/x"})) + assert any("an AWS KMS key, on a gcp binding" in e for e in errors) + assert validate_governance( + _contract( + principals=MAPPING, + lifecycle={"retention": "P30D", "expire": True}, + encryption={"kms": "product"}, + restrictions=TestColumnRestrictions.DENY, + ) + ) == ([], []) + + +# ── tofu validate ───────────────────────────────────────────────────────── + +_TOFU = shutil.which("tofu") + + +@pytest.mark.integration +@pytest.mark.provider +@pytest.mark.skipif(_TOFU is None, reason="tofu not on PATH") +@pytest.mark.parametrize( + "shape", + ["all", "existing-key-partition-column", "grants-only", "multi-region-product-key"], +) +def test_governed_modules_pass_tofu_validate(shape, tmp_path): + from tests.iac.test_iac_tofu_validate import _tofu_init_or_skip + + contract = { + "all": lambda: _contract( + principals=MAPPING, + lifecycle={"retention": "P90D", "expire": True}, + encryption={"kms": "product"}, + restrictions=TestColumnRestrictions.DENY, + ), + "existing-key-partition-column": lambda: _contract( + principals=MAPPING, + lifecycle={"retention": "P7Y", "expire": True}, + partition_by=["created_at"], + encryption={"kms": "projects/sec/locations/europe-west1/keyRings/r/cryptoKeys/k"}, + restrictions=[{"principal": PLATFORM, "columns": ["msisdn"], "access": "allow"}], + ), + "grants-only": lambda: _contract(principals=MAPPING), + "multi-region-product-key": lambda: _contract( + principals=MAPPING, encryption={"kms": "product"}, region="EU" + ), + }[shape]() + (tmp_path / "main.tf.json").write_text( + build_module(get_iac_plugin("gcp"), copy.deepcopy(contract)) + ) + _tofu_init_or_skip(tmp_path) + done = subprocess.run( + [_TOFU, "validate", "-no-color"], cwd=tmp_path, capture_output=True, text=True + ) + assert done.returncode == 0, done.stderr or done.stdout + + +# ── contract text never becomes an interpolation ───────────────────────── + + +def test_every_interpolation_is_one_the_emitter_built(): + import re + + contract = _contract( + principals=MAPPING, + lifecycle={"retention": "P90D", "expire": True}, + encryption={"kms": "product"}, + restrictions=TestColumnRestrictions.DENY, + ) + exposure = contract["exposes"][0] + exposure["contract"]["schema"][0]["description"] = 'x ${file("/etc/passwd")} %{ if 1 }' + exposure["binding"]["location"]["dataset"] = 'ds${file("/etc/hosts")}' + rendered = build_module(get_iac_plugin("gcp"), contract) + live = re.findall(r"(? Dict[str, Any]: + contract = _contract(**kwargs) + contract.update( + kind="DataProduct", + name="Retention candidates", + metadata={"owner": {"team": "data-platform"}}, + ) + contract["exposes"][0]["kind"] = "table" + return contract + + @staticmethod + def _errors(contract: Dict[str, Any]) -> List[str]: + from fluid_build.schema_manager import FluidSchemaManager + + return list(FluidSchemaManager().validate_contract(contract).errors) + + def test_the_preview_schema_accepts_the_gcp_fields(self): + contract = self._schema_contract( + principals={**MAPPING, ANALYSTS: ["group:a@northwind.com", "group:b@northwind.com"]}, + lifecycle={"retention": "P90D", "expire": True}, + encryption={"kms": "projects/sec/locations/europe-west1/keyRings/r/cryptoKeys/k"}, + restrictions=TestColumnRestrictions.DENY, + partition_by=["created_at"], + ) + assert self._errors(contract) == [] + + def test_the_aws_shapes_still_validate(self): + contract = self._schema_contract( + principals={ANALYSTS: "arn:aws:iam::111111111111:role/analyst"}, + encryption={"kms": "alias/fluid/gold"}, + ) + contract["exposes"][0]["binding"]["platform"] = "aws" + contract["exposes"][0]["binding"]["format"] = "parquet" + assert self._errors(contract) == [] + + @pytest.mark.parametrize( + "platform, kms", + [ + ("gcp", "alias/fluid/gold"), + ("gcp", "arn:aws:kms:eu-north-1:111111111111:key/abc"), + ("aws", "projects/sec/locations/europe-west1/keyRings/r/cryptoKeys/k"), + ], + ) + def test_a_key_reference_validates_only_on_its_own_cloud(self, platform, kms): + contract = self._schema_contract(principals=MAPPING, encryption={"kms": kms}) + contract["exposes"][0]["binding"]["platform"] = platform + if platform == "aws": + contract["exposes"][0]["binding"]["principals"] = {} + errors = self._errors(contract) + assert errors and all("kms" in e or "encryption" in e for e in errors), errors + + @pytest.mark.parametrize( + "platform, identity", + [ + ("gcp", "arn:aws:iam::111111111111:role/analyst"), + ("aws", "group:analysts@northwind.com"), + ], + ) + def test_a_mapped_identity_validates_only_on_its_own_cloud(self, platform, identity): + contract = self._schema_contract(principals={ANALYSTS: identity}) + contract["exposes"][0]["binding"]["platform"] = platform + errors = self._errors(contract) + assert errors and all("principals" in e for e in errors), errors + + def test_a_templated_identity_passes_the_schema(self): + contract = self._schema_contract( + principals={ + PIPELINE: "serviceAccount:p@{{ env.FLUID_DEMO_GCP_PROJECT }}.iam.gserviceaccount.com" + } + ) + assert self._errors(contract) == [] + + def test_principals_are_0_7_6_only(self): + contract = self._schema_contract(principals=MAPPING) + contract["fluidVersion"] = "0.7.5" + errors = self._errors(contract) + assert errors and any("principals" in e for e in errors), errors + del contract["exposes"][0]["binding"]["principals"] + assert self._errors(contract) == [] + + +class TestTheGateIsWiredIntoFluidValidate: + """The real ``fluid validate``, so the wiring in ``cli/validate.py`` is exercised. + + Its ``try/except`` around each binding check would swallow an import error + into a verbose-only note; only a real run shows the refusal reaches stage 2. + """ + + _CONTRACT = """\ +fluidVersion: "0.7.6" +kind: DataProduct +id: gold.retention_candidates +name: demo +metadata: + owner: + team: data-platform +accessPolicy: + grants: + - principal: group:data-platform@northwind.example + permissions: [read] +exposes: + - exposeId: candidates + kind: table + binding: + platform: gcp + format: bigquery_table + location: + project: northwind-demo + dataset: demo_gold + table: retention_candidates + region: europe-west1 +{extra} + contract: + schema: + - name: id + type: string +""" + + def _validate(self, tmp_path, extra=""): + import sys + + path = tmp_path / "contract.fluid.yaml" + path.write_text(self._CONTRACT.format(extra=extra), encoding="utf-8") + return subprocess.run( + [sys.executable, "-m", "fluid_build.cli", "validate", str(path)], + capture_output=True, + text=True, + cwd=tmp_path, + ) + + def test_a_placeholder_principal_fails_validate(self, tmp_path): + result = self._validate(tmp_path) + assert result.returncode != 0 + assert "reserved top-level domain" in result.stdout + + def test_a_mapped_principal_passes_validate(self, tmp_path): + extra = ( + " principals:\n" + " group:data-platform@northwind.example: group:data-platform@northwind.com" + ) + result = self._validate(tmp_path, extra) + assert result.returncode == 0, result.stdout + result.stderr + + +# ── One base contract, two overlays: fluid validate --strict on each cloud ── + + +class TestOneContractValidatesStrictOnBothClouds: + """The demo's shape: accessPolicy in the base for GCP, Lake Formation on AWS. + + ``fluid validate --strict`` is stage 2 of every generated Jenkinsfile + (``VALIDATE_STRICT:-true``) and fails on any warning. A warning that + ``accessPolicy`` is not emitted on aws, given for every aws binding, failed the + demo's three AWS pipelines while the aws overlay's Lake Formation grants were + the AWS form of the same access. + """ + + _BASE = """\ +fluidVersion: "0.7.6" +kind: DataProduct +id: bronze.customer_subscriptions +name: Customer Subscriptions +metadata: + owner: + team: data-platform +accessPolicy: + grants: + - principal: group:data-platform@northwind.example + permissions: [read, select, query] + - principal: serviceAccount:fluid-pipeline@northwind.example + permissions: [read, write, insert] +exposes: + - exposeId: subscriptions + kind: table + binding: + platform: local + format: parquet + location: + path: data/customer_subscriptions.parquet + contract: + schema: + - name: subscription_id + type: VARCHAR + required: true + - name: msisdn + type: VARCHAR +""" + + _AWS = """\ +exposes: + - binding: + platform: aws + format: parquet + location: + database: demo_bronze + table: customer_subscriptions + bucket: northwind-demo-lake + path: bronze/customer_subscriptions/ + region: eu-north-1 +{governance} +""" + + _LF = """\ + governance: + lakeFormation: + registerLocation: true + grants: + - principal: arn:aws:iam::111111111111:role/fluid-demo-lab-steward + permissions: [SELECT, DESCRIBE] + - principal: arn:aws:iam::111111111111:role/fluid-demo-lab-analyst + permissions: [SELECT, DESCRIBE] + excludedColumns: [msisdn]""" + + _GCP = """\ +exposes: + - binding: + platform: gcp + format: bigquery_table + location: + project: northwind-demo + dataset: demo_bronze + table: customer_subscriptions + region: europe-west1 + principals: + group:data-platform@northwind.example: group:data-platform@northwind.com + serviceAccount:fluid-pipeline@northwind.example: serviceAccount:fluid-pipeline@northwind-demo.iam.gserviceaccount.com +""" + + def _validate(self, tmp_path, env: str, governance: str = _LF): + import sys + + (tmp_path / "overlays").mkdir() + (tmp_path / "contract.fluid.yaml").write_text(self._BASE, encoding="utf-8") + (tmp_path / "overlays" / "aws.yaml").write_text( + self._AWS.format(governance=governance), encoding="utf-8" + ) + (tmp_path / "overlays" / "gcp.yaml").write_text(self._GCP, encoding="utf-8") + return subprocess.run( + [ + sys.executable, + "-m", + "fluid_build.cli", + "validate", + str(tmp_path / "contract.fluid.yaml"), + "--env", + env, + "--strict", + ], + capture_output=True, + text=True, + cwd=tmp_path, + ) + + @pytest.mark.parametrize("env", ["aws", "gcp"]) + def test_the_base_contract_passes_strict_on_each_cloud(self, tmp_path, env): + result = self._validate(tmp_path, env) + assert result.returncode == 0, result.stdout + result.stderr + assert "not enforced" not in result.stdout + + def test_an_aws_binding_with_no_lake_formation_grant_is_warned(self, tmp_path): + """The one case where the contract's access intent is unenforced on AWS.""" + result = self._validate(tmp_path, "aws", governance="") + out = " ".join(result.stdout.split()) + assert result.returncode != 0 + assert "not enforced on aws binding(s) subscriptions" in out + + +# ── Column restrictions: the expose's own readers ──────────────────────── + + +def _restricted( + *, readers: Optional[List[str]] = None, grants: Optional[List[Dict[str, Any]]] = None, **kw +) -> Dict[str, Any]: + contract = _contract( + restrictions=[ + {"principal": "group:interns@company.com", "columns": ["msisdn"], "access": "deny"} + ], + **kw, + ) + contract["accessPolicy"]["grants"] = grants or [] + if readers is not None: + contract["exposes"][0]["policy"]["authz"]["readers"] = readers + return contract + + +def _tag_readers(res: Dict[str, Any]) -> set: + return { + m["member"] for m in (res.get("google_data_catalog_policy_tag_iam_member") or {}).values() + } + + +class TestRestrictionReaders: + def test_a_deny_for_one_group_leaves_the_exposes_other_readers_reading(self): + """``policy.authz.readers`` are readers too; a deny for interns was a tag with none.""" + res = _resources( + _restricted(readers=["group:analytics@company.com", "group:finance@company.com"]) + ) + assert _tag_readers(res) == {"group:analytics@company.com", "group:finance@company.com"} + + def test_authz_readers_and_access_policy_readers_are_both_readers(self): + res = _resources( + _restricted( + readers=["group:analytics@company.com"], + grants=[{"principal": "group:bi@company.com", "permissions": ["read"]}], + ) + ) + assert _tag_readers(res) == {"group:analytics@company.com", "group:bi@company.com"} + + def test_authz_readers_are_mapped_through_binding_principals(self): + res = _resources( + _restricted( + readers=[ANALYSTS], + principals={ + ANALYSTS: "group:analysts@northwind.com", + "group:interns@company.com": [], + }, + ) + ) + assert _tag_readers(res) == {"group:analysts@northwind.com"} + + def test_a_restriction_with_no_reader_at_all_is_refused_not_a_locked_column(self): + error = _refusal(_restricted()) + assert error.kind == "column-restriction-no-readers" + errors, _ = validate_governance(_restricted()) + assert any("unreadable by everyone" in e for e in errors) + + @pytest.mark.parametrize( + "example, readers", + [ + ( + "examples/policy-examples/customer-profiles-contract.yaml", + {"group:data-analysts@company.com", "group:data-scientists@company.com"}, + ), + ( + "examples/bitcoin-price-api-declarative-part-c/contract.fluid.yaml", + {"group:data-analysts@company.com", "group:data-engineers@company.com"}, + ), + ( + # Mapped through binding.principals; the denied interns are no reader. + "examples/bitcoin-price-api-declarative-part-b/contract.fluid.yaml", + { + "group:data-analytics@example.com", + "group:finance-team@example.com", + "group:trading-desk@example.com", + "serviceAccount:looker@example.com", + }, + ), + ], + ) + def test_the_shipped_examples_keep_their_readers(self, example, readers): + import pathlib + + import yaml + + path = pathlib.Path(__file__).resolve().parents[2] / example + res = _resources(yaml.safe_load(path.read_text(encoding="utf-8"))) + assert res["google_data_catalog_policy_tag"] + assert _tag_readers(res) == readers + + def test_an_example_reader_left_as_a_placeholder_is_refused(self): + """part-b's looker reader is ``serviceAccount:looker@<>...``. + + The example maps it in ``binding.principals``; without that block it is + emitted as written, which is the placeholder the emitter refuses. + """ + import pathlib + + import yaml + + path = ( + pathlib.Path(__file__).resolve().parents[2] + / "examples/bitcoin-price-api-declarative-part-b/contract.fluid.yaml" + ) + contract = yaml.safe_load(path.read_text(encoding="utf-8")) + binding = contract["exposes"][0]["binding"] + assert "serviceAccount:looker@<>.iam.gserviceaccount.com" in ( + binding["principals"] + ) + del binding["principals"] + error = _refusal(contract) + assert error.kind == "principal-placeholder" + assert "<>" in str(error) + + +# ── IAM resource names: one per member, whatever it folds to ───────────── + + +def test_principals_that_fold_to_one_name_keep_one_grant_each(): + """``safe_ident`` gave data.eng@, data-eng@ and data_eng@ one resource name.""" + groups = [f"group:data{sep}eng@corp.com" for sep in (".", "-", "_")] + contract = _contract( + restrictions=[ + {"principal": "group:blocked@corp.com", "columns": ["msisdn"], "access": "deny"} + ] + ) + contract["accessPolicy"]["grants"] = [ + {"principal": g, "permissions": ["read"]} for g in groups + ["group:blocked@corp.com"] + ] + res = _resources(contract) + dataset_members = [m["member"] for m in res["google_bigquery_dataset_iam_member"].values()] + assert sorted(dataset_members) == sorted(groups + ["group:blocked@corp.com"]) + assert _tag_readers(res) == set(groups) + assert len(res["google_data_catalog_policy_tag_iam_member"]) == 3 + + +# ── Platform first: an AWS Iceberg binding is the AWS emitter's ────────── + + +def _aws_iceberg(**expose: Any) -> Dict[str, Any]: + return { + "fluidVersion": "0.7.6", + "id": "sales.lakehouse", + "accessPolicy": {"grants": [{"principal": PLATFORM, "permissions": ["read"]}]}, + "exposes": [ + { + "exposeId": "orders", + "binding": { + "platform": "aws", + "format": "iceberg", + "location": { + "bucket": "acme-sales-lakehouse", + "database": "sales", + "table": "orders", + "path": "iceberg/sales/orders/", + "region": "eu-west-1", + }, + "encryption": {"kms": "product"}, + **expose.pop("binding", {}), + }, + "lifecycle": {"retention": "P30D", "expire": True}, + "contract": {"schema": [{"name": "id", "type": "string"}]}, + **expose, + } + ], + } + + +class TestAwsIcebergIsValidatedAsAws: + def test_retention_encryption_and_a_logical_principal_are_not_refused_as_gcp(self): + errors, warnings = validate_governance(_aws_iceberg()) + assert errors == [] + assert any("not enforced on aws binding(s) orders" in w for w in warnings) + + def test_a_column_restriction_goes_through_the_lake_formation_check(self): + contract = _aws_iceberg( + policy={ + "authz": { + "columnRestrictions": [ + {"principal": "arn:aws:iam::1:role/x", "columns": ["id"], "access": "deny"} + ] + } + } + ) + errors, _ = validate_governance(contract) + assert any("declares no governance.lakeFormation.grants" in e for e in errors) + assert not any("GCP" in e for e in errors) + + +# ── Unmapped principals that are not IAM members ───────────────────────── + + +@pytest.mark.parametrize( + "principal", + ["group:data-platform", "analysts", "role:analyst"], + ids=["no-domain", "bare", "role"], +) +def test_an_unmapped_principal_that_is_not_an_iam_member_is_refused(principal): + contract = _contract() + contract["accessPolicy"]["grants"] = [{"principal": principal, "permissions": ["read"]}] + assert _refusal(contract).kind == "principal-placeholder" + errors, _ = validate_governance(contract) + assert any("is not a GCP IAM member" in e for e in errors) + + +# ── One dataset, one key ───────────────────────────────────────────────── + + +def _two_in_one_dataset( + first: Optional[Dict[str, Any]], second: Optional[Dict[str, Any]], *, view: bool = False +): + contract = _contract(principals=MAPPING) + one = contract["exposes"][0] + other = copy.deepcopy(one) + other["exposeId"] = "summary" + other["binding"]["location"]["table"] = "summary" + if view: + other["binding"]["format"] = "bigquery_view" + other["binding"]["location"]["query"] = "SELECT 1" + for exposure, block in ((one, first), (other, second)): + exposure["binding"].pop("encryption", None) + if block is not None: + exposure["binding"]["encryption"] = block + contract["exposes"] = [other, one] if view else [one, other] + return contract + + +class TestOneDatasetOneKey: + @pytest.mark.parametrize("order", ["keyed-first", "unkeyed-first"]) + def test_a_keyed_and_an_unkeyed_table_in_one_dataset_are_refused(self, order): + pair = ({"kms": "product"}, None) if order == "keyed-first" else (None, {"kms": "product"}) + contract = _two_in_one_dataset(*pair) + assert _refusal(contract).kind == "encryption-kms-mixed-dataset" + errors, _ = validate_governance(contract) + assert any("different encryption" in e for e in errors) + + def test_tables_with_the_same_key_share_the_dataset_default(self): + res = _resources(_two_in_one_dataset({"kms": "product"}, {"kms": "product"})) + dataset = res["google_bigquery_dataset"][f"{CID}_demo_gold"] + assert "default_encryption_configuration" in dataset + keys = { + json.dumps(t.get("encryption_configuration")) + for t in res["google_bigquery_table"].values() + } + assert len(keys) == 1 and "null" not in keys + + def test_an_unkeyed_view_emitted_first_does_not_drop_the_datasets_key(self): + """The dataset body used to be the first expose's, so its default key depended on order.""" + res = _resources(_two_in_one_dataset({"kms": "product"}, None, view=True)) + dataset = res["google_bigquery_dataset"][f"{CID}_demo_gold"] + assert dataset["default_encryption_configuration"] == { + "kms_key_name": f"${{google_kms_crypto_key.{CID}_demo_gold_kms.id}}" + } + assert dataset["depends_on"] == [f"google_kms_crypto_key_iam_member.{CID}_demo_gold_kms"] + + +# ── Retention on a partition column is event-time age ──────────────────── + + +def test_retention_on_a_partition_column_is_said_to_count_from_its_date(caplog): + """BigQuery expires a partition relative to its date, not to when rows landed.""" + import logging + + with caplog.at_level(logging.INFO, logger="fluid_build.iac.providers.gcp_governance"): + _resources( + _contract( + principals=MAPPING, + lifecycle={"retention": "P30D", "expire": True}, + partition_by=["created_at"], + ) + ) + assert any("bigquery_retention_event_time" in r.getMessage() for r in caplog.records) + import pathlib + + schema = json.loads( + ( + pathlib.Path(__file__).resolve().parents[2] + / "fluid_build/schemas/fluid-schema-0.7.6.json" + ).read_text(encoding="utf-8") + ) + text = " ".join( + [ + schema["$defs"]["bindingLocation"]["properties"]["partitionBy"]["description"], + schema["$defs"]["exposeLifecycle"]["properties"]["expire"]["description"], + ] + ) + assert "backfill" in text and "date in that column" in text + + +# ── A restriction's tags and labels reach the policy tag ───────────────── + + +def test_restriction_tags_and_labels_are_written_on_the_policy_tag(): + contract = _contract( + principals=MAPPING, + restrictions=[ + { + "principal": ANALYSTS, + "columns": ["msisdn"], + "access": "deny", + "tags": ["sensitive-financial-data"], + "labels": {"reason": "pii"}, + } + ], + ) + (tag,) = _resources(contract)["google_data_catalog_policy_tag"].values() + assert "Tags: sensitive-financial-data." in tag["description"] + assert "Labels: reason=pii." in tag["description"] + + +def test_a_restriction_label_cannot_become_an_interpolation(): + contract = _contract( + principals=MAPPING, + restrictions=[ + { + "principal": ANALYSTS, + "columns": ["msisdn"], + "access": "deny", + "labels": {"reason": '${file("/etc/passwd")}'}, + } + ], + ) + rendered = build_module(get_iac_plugin("gcp"), contract) + assert '$${file(\\"/etc/passwd\\")}' in rendered + assert '"${file' not in rendered and " ${file" not in rendered diff --git a/tests/iac/test_iac_gcp_governance_plan.py b/tests/iac/test_iac_gcp_governance_plan.py new file mode 100644 index 00000000..bece2a84 --- /dev/null +++ b/tests/iac/test_iac_gcp_governance_plan.py @@ -0,0 +1,410 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""GCP governance through the real ``tofu`` and terraform-provider-google, on a live table. + +What the data-loss gate decides depends on what the provider plans for a table +that already exists: BigQuery cannot partition an existing table, or re-key one in +place, so both must plan a replacement (``remove`` in the change summary, which +``fluid apply`` refuses without ``--allow-data-loss``), while a new retention period +must stay an in-place change. Only a real plan against real state shows that. + +The state is made by a real ``tofu apply`` against ``_fake_bigquery.FakeBigQuery``, +an in-process stand-in for the BigQuery REST API (the goccy emulator crashes the +provider on apply). It also shows the dataset grants are added to the dataset's +access list, not written over it. No Google credential or endpoint is involved. + +Not proven here: what real BigQuery accepts (a load into a policy-tagged column, +the service agent's use of the key), Cloud KMS and Data Catalog resources (the +fake serves BigQuery only; they are proven by ``tofu validate`` and the rendering +tests), and IAM evaluation. + +Skipped unless ``tofu`` is on PATH; ``tofu init`` needs the registry (or a +provider cache) for hashicorp/google. +""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any, Dict, Iterator, List, Optional + +import pytest + +from fluid_build.cli._apply_opentofu_engine import _data_loss_blocked +from fluid_build.iac import build_module, get_iac_plugin, runner +from fluid_build.iac.credentials import build_tofu_env + +from ._fake_bigquery import DEFAULT_ACCESS, FakeBigQuery + +pytestmark = [pytest.mark.integration, pytest.mark.gcp, pytest.mark.provider] + +_SKIP = runner.tofu_path() is None +PROJECT = "fluid-fake" +TABLE = "google_bigquery_table.gov_plan_orders" +DATASET = "google_bigquery_dataset.gov_plan_sales" +KEY = f"projects/{PROJECT}/locations/europe-west1/keyRings/ring/cryptoKeys/orders" + + +@pytest.fixture +def fake() -> Iterator[FakeBigQuery]: + server = FakeBigQuery().start() + try: + yield server + finally: + server.stop() + + +@pytest.fixture(scope="module") +def tofu_env(tmp_path_factory: pytest.TempPathFactory) -> Dict[str, str]: + """No Google credential reachable: every ``GOOGLE_*`` / gcloud variable is dropped.""" + env = { + k: v + for k, v in build_tofu_env().items() + if not k.startswith(("GOOGLE_", "CLOUDSDK_", "GCLOUD_")) + } + env["CLOUDSDK_CONFIG"] = str(tmp_path_factory.mktemp("no-gcloud")) + env.setdefault("TF_PLUGIN_CACHE_DIR", str(tmp_path_factory.mktemp("tofu-plugin-cache"))) + return env + + +def _contract( + *, retention: Optional[str] = None, kms: Optional[str] = None, field: bool = False +) -> Dict[str, Any]: + binding: Dict[str, Any] = { + "platform": "gcp", + "format": "bigquery_table", + "location": { + "project": PROJECT, + "dataset": "sales", + "table": "orders", + "region": "europe-west1", + }, + "principals": {"group:readers@company.example": "group:readers@fluid-fake.test-corp.com"}, + } + if field: + binding["location"]["partitionBy"] = ["created_at"] + if kms: + binding["encryption"] = {"kms": kms} + exposure: Dict[str, Any] = { + "exposeId": "orders", + "binding": binding, + "contract": { + "schema": [ + {"name": "id", "type": "string", "required": True}, + {"name": "created_at", "type": "timestamp"}, + ] + }, + } + if retention: + exposure["lifecycle"] = {"retention": retention, "expire": True} + return { + "fluidVersion": "0.7.6", + "id": "gov.plan", + "accessPolicy": { + "grants": [{"principal": "group:readers@company.example", "permissions": ["read"]}] + }, + "exposes": [exposure], + } + + +def _write(contract: Dict[str, Any], workdir: Path, endpoint: str) -> None: + workdir.mkdir(parents=True, exist_ok=True) + (workdir / "main.tf.json").write_text(build_module(get_iac_plugin("gcp"), contract)) + provider = { + "provider": { + "google": { + "project": PROJECT, + "access_token": "fake-token", # pragma: allowlist secret — not a credential + "big_query_custom_endpoint": endpoint, + "add_terraform_attribution_label": False, + } + } + } + (workdir / "provider.tf.json").write_text(json.dumps(provider)) + + +def _init(workdir: Path, env: Dict[str, str]) -> None: + done = runner.tofu_init(str(workdir), backend=False, env=env) + if not done.ok and "registry" in (done.stderr + done.stdout).lower(): + pytest.skip(f"tofu init could not reach the provider registry: {done.stderr[:200]}") + assert done.ok, done.stderr or done.stdout + + +def _plan(workdir: Path, env: Dict[str, str]) -> Dict[str, Any]: + plan = runner.tofu_plan(str(workdir), out_file="tfplan", env=env) + assert plan.ok, plan.stderr or plan.stdout + shown = runner.tofu_show_plan(str(workdir), plan_file="tfplan", env=env) + assert shown is not None + return { + "summary": runner.change_summary(plan), + "changes": {c["address"]: c for c in shown.get("resource_changes") or []}, + } + + +def _apply(workdir: Path, env: Dict[str, str]) -> None: + done = runner.tofu_apply(str(workdir), plan_file="tfplan", env=env) + assert done.ok, done.stderr or done.stdout + + +def _actions(plan: Dict[str, Any], address: str) -> List[str]: + return list(plan["changes"][address]["change"]["actions"]) + + +def _live(contract: Dict[str, Any], workdir: Path, fake: FakeBigQuery, env: Dict[str, str]): + _write(contract, workdir, fake.endpoint) + _init(workdir, env) + _plan(workdir, env) + _apply(workdir, env) + + +@pytest.mark.skipif(_SKIP, reason="needs `tofu` on PATH") +def test_dataset_grants_are_added_to_its_access_list_not_written_over_it(tmp_path, fake, tofu_env): + """The dataset keeps the entries BigQuery gave it, and gains the mapped member. + + Before, the emitter wrote an authoritative ``access`` list of the contract's + principals as written: the project's owners and the creator were dropped, and + the logical principal itself was the grantee. + """ + _live(_contract(), tmp_path, fake, tofu_env) + access = fake.datasets[(PROJECT, "sales")]["access"] + for entry in DEFAULT_ACCESS: + assert entry in access + assert {"role": "READER", "groupByEmail": "readers@fluid-fake.test-corp.com"} in access + assert not any("company.example" in json.dumps(e) for e in access) + + +@pytest.mark.skipif(_SKIP, reason="needs `tofu` on PATH") +@pytest.mark.parametrize("field", [False, True], ids=["ingestion-time", "partition-column"]) +def test_adding_retention_to_a_live_table_plans_its_replacement(tmp_path, fake, tofu_env, field): + """BigQuery cannot partition an existing table; the plan must say replace. + + Without the ``terraform_data`` trigger the provider plans adding ingestion-time + partitioning as an in-place update, which the API refuses at apply, and the + data-loss gate never sees a removal. + """ + _live(_contract(), tmp_path, fake, tofu_env) + _write(_contract(retention="P30D", field=field), tmp_path, fake.endpoint) + plan = _plan(tmp_path, tofu_env) + assert _actions(plan, TABLE) == ["delete", "create"] + assert plan["summary"]["remove"] >= 1 + assert _data_loss_blocked(plan["summary"], allow_data_loss=False) + assert not _data_loss_blocked(plan["summary"], allow_data_loss=True) + after = plan["changes"][TABLE]["change"]["after"] + assert after["time_partitioning"][0]["expiration_ms"] == 30 * 86_400_000 + assert after["time_partitioning"][0]["type"] == "DAY" + assert after["time_partitioning"][0].get("field") == ("created_at" if field else None) + # Never a whole-table TTL: the module sets none, and the plan keeps the live one (none). + module = json.loads((tmp_path / "main.tf.json").read_text()) + assert "expiration_time" not in module["resource"]["google_bigquery_table"]["gov_plan_orders"] + assert after.get("expiration_time") in (None, 0) + + +@pytest.mark.skipif(_SKIP, reason="needs `tofu` on PATH") +def test_a_new_retention_period_is_an_in_place_change(tmp_path, fake, tofu_env): + _live(_contract(retention="P30D"), tmp_path, fake, tofu_env) + _write(_contract(retention="P90D"), tmp_path, fake.endpoint) + plan = _plan(tmp_path, tofu_env) + assert _actions(plan, TABLE) == ["update"] + assert plan["summary"]["remove"] == 0 + assert not _data_loss_blocked(plan["summary"], allow_data_loss=False) + after = plan["changes"][TABLE]["change"]["after"] + assert after["time_partitioning"][0]["expiration_ms"] == 90 * 86_400_000 + + +@pytest.mark.skipif(_SKIP, reason="needs `tofu` on PATH") +def test_the_same_contract_plans_no_change(tmp_path, fake, tofu_env): + """Applied once, the governed table plans clean: no perpetual replacement.""" + _live(_contract(retention="P30D", kms=KEY), tmp_path, fake, tofu_env) + plan = _plan(tmp_path, tofu_env) + assert plan["summary"] == {"add": 0, "change": 0, "remove": 0} + + +@pytest.mark.skipif(_SKIP, reason="needs `tofu` on PATH") +def test_adding_a_key_to_a_live_table_plans_its_replacement_and_rekeys_the_dataset( + tmp_path, fake, tofu_env +): + _live(_contract(), tmp_path, fake, tofu_env) + _write(_contract(kms=KEY), tmp_path, fake.endpoint) + plan = _plan(tmp_path, tofu_env) + assert _actions(plan, TABLE) == ["delete", "create"] + assert _actions(plan, DATASET) == ["update"] + assert _data_loss_blocked(plan["summary"], allow_data_loss=False) + table = plan["changes"][TABLE]["change"]["after"] + dataset = plan["changes"][DATASET]["change"]["after"] + assert table["encryption_configuration"][0]["kms_key_name"] == KEY + assert dataset["default_encryption_configuration"][0]["kms_key_name"] == KEY + # Applied, the fake holds what BigQuery would be sent. + _apply(tmp_path, tofu_env) + assert fake.tables[(PROJECT, "sales", "orders")]["encryptionConfiguration"] == { + "kmsKeyName": KEY + } + + +# ── Revoking a grant is not data loss ──────────────────────────────────── + +MEMBERS = "google_bigquery_dataset_iam_member" + + +def _with_readers(*readers: str) -> Dict[str, Any]: + contract = _contract() + contract["exposes"][0]["binding"].pop("principals") + contract["accessPolicy"]["grants"] = [ + {"principal": f"group:{r}", "permissions": ["read"]} for r in readers + ] + return contract + + +def _data_bearing(workdir: Path, env: Dict[str, str]): + from fluid_build.cli._apply_opentofu_engine import _data_bearing_changes + + plan = runner.tofu_plan(str(workdir), out_file="tfplan", env=env) + assert plan.ok, plan.stderr or plan.stdout + summary = runner.change_summary(plan) + return summary, _data_bearing_changes(summary, runner.planned_removals(plan)) + + +@pytest.mark.skipif(_SKIP, reason="needs `tofu` on PATH") +def test_revoking_a_reader_passes_the_data_loss_gate(tmp_path, fake, tofu_env): + """A removed grant is a member destroy: counted, it needed --allow-data-loss.""" + _live(_with_readers("readers@corp-a.com", "departed@corp-a.com"), tmp_path, fake, tofu_env) + _write(_with_readers("readers@corp-a.com"), tmp_path, fake.endpoint) + summary, (data_changes, revoked) = _data_bearing(tmp_path, tofu_env) + assert summary["remove"] == 1 + assert _data_loss_blocked(summary, allow_data_loss=False) # what the gate saw before + assert data_changes["remove"] == 0 + assert not _data_loss_blocked(data_changes, allow_data_loss=False) + assert len(revoked) == 1 and revoked[0].startswith(f"{MEMBERS}.") + _apply(tmp_path, tofu_env) + access = json.dumps(fake.datasets[(PROJECT, "sales")]["access"]) + assert "departed@corp-a.com" not in access and "readers@corp-a.com" in access + + +@pytest.mark.skipif(_SKIP, reason="needs `tofu` on PATH") +def test_a_table_replacement_is_still_gated_next_to_a_revocation(tmp_path, fake, tofu_env): + _live(_with_readers("readers@corp-a.com", "departed@corp-a.com"), tmp_path, fake, tofu_env) + contract = _with_readers("readers@corp-a.com") + contract["exposes"][0]["lifecycle"] = {"retention": "P30D", "expire": True} + _write(contract, tmp_path, fake.endpoint) + _summary, (data_changes, revoked) = _data_bearing(tmp_path, tofu_env) + assert len(revoked) == 1 + assert data_changes["remove"] == 1 # the table's replacement + assert _data_loss_blocked(data_changes, allow_data_loss=False) + + +def test_removals_the_plan_does_not_itemise_all_count(): + """Fail closed: an event stream that misses a removal gates every removal.""" + from fluid_build.cli._apply_opentofu_engine import _data_bearing_changes + + changes = {"add": 0, "change": 0, "remove": 2} + counted, revoked = _data_bearing_changes(changes, [(f"{MEMBERS}.a", MEMBERS)]) + assert counted["remove"] == 2 and revoked == [] + + +# ── From an older forge-cli's authoritative access list ────────────────── + + +def _main_module(contract: Dict[str, Any]) -> str: + """What forge-cli 0.16.6 and earlier emitted: the grants as the dataset's ``access``.""" + from fluid_build.iac.access import normalize_access_grants + from fluid_build.iac.providers.gcp import _bq_access_entries + + module = json.loads(build_module(get_iac_plugin("gcp"), contract)) + resources = module["resource"] + resources.pop(MEMBERS) + (dataset,) = resources["google_bigquery_dataset"].values() + dataset["access"] = _bq_access_entries(normalize_access_grants(contract)) + return json.dumps(module) + + +@pytest.mark.skipif(_SKIP, reason="needs `tofu` on PATH") +def test_a_grant_removed_while_leaving_the_authoritative_list_is_revoked(tmp_path, fake, tofu_env): + """Measured without the reconciliation: plan +1 ~0 -0, departed@ kept READER, re-plan clean. + + The provider keeps ``access`` as Computed once the module stops setting it, so + an entry the old list held and no member resource covers was never revoked. + """ + import logging + + from fluid_build.cli._apply_opentofu_engine import _reconcile_with_state + + before = _with_readers("readers@corp-a.com", "departed@corp-a.com") + _write(before, tmp_path, fake.endpoint) + (tmp_path / "main.tf.json").write_text(_main_module(before)) + _init(tmp_path, tofu_env) + _plan(tmp_path, tofu_env) + _apply(tmp_path, tofu_env) + live = json.dumps(fake.datasets[(PROJECT, "sales")]["access"]) + assert "departed@corp-a.com" in live + + _write(_with_readers("readers@corp-a.com"), tmp_path, fake.endpoint) + reports = _reconcile_with_state( + get_iac_plugin("gcp"), + tmp_path / "main.tf.json", + str(tmp_path), + tofu_env, + logging.getLogger("test"), + ) + assert reports == [ + {"dataset": "sales", "revoked": ["READER group:departed@corp-a.com"], "blocked": False} + ] + plan = _plan(tmp_path, tofu_env) + assert _actions(plan, DATASET) == ["update"] + assert plan["summary"]["remove"] == 0 + _apply(tmp_path, tofu_env) + live = json.dumps(fake.datasets[(PROJECT, "sales")]["access"]) + assert "departed@corp-a.com" not in live and "readers@corp-a.com" in live + + # The next run finds the member resources in state and leaves ``access`` unset. + _write(_with_readers("readers@corp-a.com"), tmp_path, fake.endpoint) + assert ( + _reconcile_with_state( + get_iac_plugin("gcp"), + tmp_path / "main.tf.json", + str(tmp_path), + tofu_env, + logging.getLogger("test"), + ) + == [] + ) + assert _plan(tmp_path, tofu_env)["summary"] == {"add": 0, "change": 0, "remove": 0} + + +@pytest.mark.skipif(_SKIP, reason="needs `tofu` on PATH") +def test_an_unchanged_contract_leaving_the_authoritative_list_revokes_nothing( + tmp_path, fake, tofu_env +): + import logging + + from fluid_build.cli._apply_opentofu_engine import _reconcile_with_state + + contract = _with_readers("readers@corp-a.com") + _write(contract, tmp_path, fake.endpoint) + (tmp_path / "main.tf.json").write_text(_main_module(contract)) + _init(tmp_path, tofu_env) + _plan(tmp_path, tofu_env) + _apply(tmp_path, tofu_env) + _write(contract, tmp_path, fake.endpoint) + assert ( + _reconcile_with_state( + get_iac_plugin("gcp"), + tmp_path / "main.tf.json", + str(tmp_path), + tofu_env, + logging.getLogger("test"), + ) + == [] + ) + plan = _plan(tmp_path, tofu_env) + assert plan["summary"]["remove"] == 0 diff --git a/tests/iac/test_iac_packaging_default_pin.py b/tests/iac/test_iac_packaging_default_pin.py index 933155e6..382d991c 100644 --- a/tests/iac/test_iac_packaging_default_pin.py +++ b/tests/iac/test_iac_packaging_default_pin.py @@ -55,6 +55,7 @@ provider_config, render_tofu_json, ) +from fluid_build.iac.base import UnsupportedBindingError from fluid_build.iac.packaging import ( CONTAINER_KINDS, LEGACY, @@ -184,6 +185,16 @@ def test_wired_cli_emit_is_byte_identical( _, expected = _expected_module_bytes(path) except CLIError as exc: pytest.skip(f"not a single-cloud contract ({exc.event})") + except UnsupportedBindingError as refused: + # The emitter refuses this contract (an unmapped placeholder + # principal, say): the wired path must refuse it the same way and + # write nothing. + with pytest.raises(CLIError) as wired: + _run_generate_iac(path, tmp_path) + assert wired.value.event == "unsupported_binding" + assert wired.value.context["kind"] == refused.kind + assert not (tmp_path / "main.tf.json").exists() + return actual = _run_generate_iac(path, tmp_path) assert actual == expected, ( f"{path.relative_to(REPO_ROOT)}: `fluid generate iac` output changed " diff --git a/tests/iac/test_iac_packaging_emit.py b/tests/iac/test_iac_packaging_emit.py index b7028e81..0e0e8123 100644 --- a/tests/iac/test_iac_packaging_emit.py +++ b/tests/iac/test_iac_packaging_emit.py @@ -345,10 +345,15 @@ def test_shared_dataset_drops_the_authoritative_acl_for_table_level_iam(self): assert member["role"] == "roles/bigquery.dataViewer" assert member["member"] == "user:analytics@acme.com" - def test_isolated_dataset_keeps_the_dataset_acl(self): + def test_isolated_dataset_keeps_the_dataset_grant(self): resources = GcpIacPlugin().emit(_gcp_contract(ISOLATED)) dataset = resources["google_bigquery_dataset"]["orders_adp_sales_pool"] - assert dataset["access"] == [{"role": "READER", "user_by_email": "analytics@acme.com"}] + # A non-authoritative dataset member, not the authoritative access list. + assert "access" not in dataset + members = list(resources["google_bigquery_dataset_iam_member"].values()) + assert [(m["role"], m["member"]) for m in members] == [ + ("roles/bigquery.dataViewer", "user:analytics@acme.com") + ] assert "google_bigquery_table_iam_member" not in resources def test_shared_bucket_is_a_data_source_with_no_force_destroy(self): diff --git a/tests/iac/test_iac_provider_match.py b/tests/iac/test_iac_provider_match.py index 24a1b63c..931245b5 100644 --- a/tests/iac/test_iac_provider_match.py +++ b/tests/iac/test_iac_provider_match.py @@ -155,9 +155,11 @@ def _contracts(): # `location: us-east-1` — an AWS region, which is not a valid GCS # location. They are why the gate cannot be a zero-resource check alone, # and why it belongs on the provider/binding pair rather than the output. + # (``aws-glue-data-lake/contract-iceberg.fluid.yaml`` left the set when the GCP + # emitter began refusing an unmapped principal that is not an IAM member: its + # ``role:data-analyst`` would have become ``group:role:data-analyst``.) _WRONG_CLOUD_EMITS = { ("aws-glue-data-lake/contract-etl-job.fluid.yaml", "gcp"), - ("aws-glue-data-lake/contract-iceberg.fluid.yaml", "gcp"), ("aws-iceberg-lakehouse/contract.fluid.yaml", "gcp"), } diff --git a/tests/iac/test_iac_state_key_migration_moto.py b/tests/iac/test_iac_state_key_migration_moto.py new file mode 100644 index 00000000..f342b8b4 --- /dev/null +++ b/tests/iac/test_iac_state_key_migration_moto.py @@ -0,0 +1,420 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The provider-keyed state default, and the move of the old state, against moto. + +A bucket-only ``FLUID_STATE_BACKEND`` used to key a contract's state by its +id alone, ``fluid//terraform.tfstate``, so the aws and the gcp apply of +one contract (two ``--env`` overlays) shared one state and each plan read the +other cloud's resources as orphans to destroy. The default is now +``fluid///terraform.tfstate``, and the first apply after the +upgrade moves the old state there with OpenTofu's own ``init +-migrate-state``. + +Everything here is real ``tofu`` against a moto S3 (the state bucket) and +moto AWS APIs (the product's resources): the old release's apply is played +by an apply with the old key spelled out, which is exactly the object that +release wrote. Pinned: the move, a plan after it that changes nothing, the +old object left in place, a second apply that moves nothing, a wiped +workdir (CI) as well as a kept one, the data-loss gate still closed after +the move, another provider's state at the old key left alone, a state of +two clouds refused, and a new key that already holds state never +overwritten. And the read-only paths: ``fluid apply --dry-run`` (the +generated Jenkins default) and the ``fluid diff`` / ``verify --state-drift`` +pass plan against the old key while the move is pending and write nothing; +the dry-run used to copy the state and write the new key. + +Skipped unless ``tofu`` is on PATH and moto's ``server`` extra is installed. +""" + +from __future__ import annotations + +import argparse +import contextlib +import json +import logging +import shutil +from pathlib import Path +from typing import Any, Dict, Iterator + +import pytest +import yaml + +from fluid_build.cli import _apply_opentofu_engine as engine +from fluid_build.cli._common import CLIError +from fluid_build.iac import runner +from fluid_build.iac import state_migration as mig +from fluid_build.iac.credentials import build_tofu_env + +pytestmark = [pytest.mark.integration, pytest.mark.provider, pytest.mark.aws] + +_LOG = logging.getLogger("test.iac.state_key_migration") +_REGION = "us-east-1" +_CID = "bronze.customer_subscriptions" +_STATE_BUCKET = "fluid-state-migration" +_DATA_BUCKET = "migration-moto-lake" +_LEGACY_KEY = f"fluid/{_CID}/terraform.tfstate" +_AWS_KEY = f"fluid/{_CID}/aws/terraform.tfstate" +_GCP_KEY = f"fluid/{_CID}/gcp/terraform.tfstate" + + +def _have_moto_server() -> bool: + try: + from moto.server import ThreadedMotoServer # noqa: F401 + + return True + except Exception: # noqa: BLE001 + return False + + +pytestmark.append( + pytest.mark.skipif( + runner.tofu_path() is None or not _have_moto_server(), + reason="needs `tofu` on PATH + moto server extra (pip install 'moto[server]')", + ) +) + + +@pytest.fixture(scope="module") +def plugin_cache(tmp_path_factory: pytest.TempPathFactory) -> str: + """One provider download for the module, not one per workdir.""" + return str(tmp_path_factory.mktemp("tofu-plugin-cache")) + + +@pytest.fixture +def moto(monkeypatch, tmp_path: Path, plugin_cache: str) -> Iterator[str]: + """A fresh moto server holding the state bucket; tofu and boto3 aim at it.""" + import requests + from moto.server import ThreadedMotoServer + + server = ThreadedMotoServer(port=0, verbose=False) + server.start() + try: + _, port = server.get_host_and_port() + endpoint = f"http://127.0.0.1:{port}" + with contextlib.suppress(Exception): + requests.post(f"{endpoint}/moto-api/reset", timeout=5) + for var in ("AWS_PROFILE", "AWS_SESSION_TOKEN", "FLUID_STATE_BACKEND", "FLUID_PROVIDER"): + monkeypatch.delenv(var, raising=False) + monkeypatch.setenv("AWS_ENDPOINT_URL", endpoint) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") # pragma: allowlist secret + monkeypatch.setenv("AWS_REGION", _REGION) + monkeypatch.setenv("AWS_DEFAULT_REGION", _REGION) + monkeypatch.setenv("AWS_CONFIG_FILE", "/dev/null") + monkeypatch.setenv("AWS_SHARED_CREDENTIALS_FILE", "/dev/null") + monkeypatch.setenv("AWS_EC2_METADATA_DISABLED", "true") + monkeypatch.setenv("TF_PLUGIN_CACHE_DIR", plugin_cache) + monkeypatch.chdir(tmp_path) + _s3(endpoint).create_bucket(Bucket=_STATE_BUCKET) + yield endpoint + finally: + server.stop() + + +def _s3(endpoint: str): + import boto3 + + return boto3.client( + "s3", + endpoint_url=endpoint, + aws_access_key_id="testing", + aws_secret_access_key="testing", # pragma: allowlist secret + region_name=_REGION, + ) + + +def _object(endpoint: str, key: str) -> Dict[str, Any]: + body = _s3(endpoint).get_object(Bucket=_STATE_BUCKET, Key=key)["Body"].read() + return json.loads(body) + + +def _addresses(state: Dict[str, Any]) -> list: + return sorted((r["mode"], r["type"], r["name"]) for r in state["resources"]) + + +def _keys(endpoint: str) -> set: + listing = _s3(endpoint).list_objects_v2(Bucket=_STATE_BUCKET) + return {o["Key"] for o in listing.get("Contents", [])} + + +def _contract(*, table: str = "customer_subscriptions") -> Dict[str, Any]: + exposes = [ + { + "exposeId": "subscriptions", + "kind": "table", + "binding": { + "platform": "aws", + "format": "parquet", + "location": { + "database": "demo_bronze", + "table": table, + "bucket": _DATA_BUCKET, + "path": "bronze/customer_subscriptions/", + }, + }, + "contract": { + "schema": [ + {"name": "subscription_id", "type": "VARCHAR", "required": True}, + {"name": "msisdn", "type": "VARCHAR"}, + ] + }, + } + ] + return { + "fluidVersion": "0.7.5", + "kind": "DataProduct", + "id": _CID, + "name": "Customer Subscriptions", + "domain": "Customer", + "metadata": {"layer": "Bronze", "owner": {"team": "data-platform"}}, + "exposes": exposes, + } + + +def _write(root: Path, contract: Dict[str, Any]) -> Path: + path = root / "contract.fluid.yaml" + path.write_text(yaml.safe_dump(contract, sort_keys=False), encoding="utf-8") + return path + + +def _apply(contract_path: Path, root: Path, *, state_backend=None, dry_run=False) -> None: + args = argparse.Namespace( + contract=str(contract_path), + env=None, + provider=None, + workspace_dir=str(root), + state_backend=state_backend, + dry_run=dry_run, + allow_data_loss=False, + no_verify_plan_binding=True, + ) + assert engine.apply_via_opentofu(args, _LOG) == 0 + + +def _apply_as_the_old_release(contract_path: Path, root: Path) -> None: + """The object a bucket-only FLUID_STATE_BACKEND wrote before this change.""" + _apply(contract_path, root, state_backend=f"s3://{_STATE_BUCKET}/{_LEGACY_KEY}") + + +def _upgraded_apply(contract_path: Path, root: Path, monkeypatch, *, dry_run=False) -> None: + """The first apply of the upgraded release (a real one moves the state).""" + monkeypatch.setenv("FLUID_STATE_BACKEND", f"s3://{_STATE_BUCKET}") + _apply(contract_path, root, dry_run=dry_run) + + +def _block(key: str) -> Dict[str, Any]: + return {"s3": {"bucket": _STATE_BUCKET, "key": key}} + + +def _reconcile(tmp_path: Path, key: str, provider: str, *, migrate: bool): + """``reconcile_state_key`` from a workdir initialised on ``key``, as the apply calls it.""" + workdir = tmp_path / f"workdir-{provider}" + workdir.mkdir(exist_ok=True) + (workdir / "main.tf.json").write_text( + json.dumps({"terraform": {"backend": _block(key)}}), encoding="utf-8" + ) + env = build_tofu_env() + assert runner.tofu_init(str(workdir), env=env, reconfigure=True).ok + return mig.reconcile_state_key( + workdir=workdir, + current=_block(key), + legacy=_block(_LEGACY_KEY), + provider=provider, + env=env, + migrate=migrate, + ) + + +@pytest.mark.parametrize("wipe_workdir", [False, True], ids=["kept-workdir", "wiped-workdir"]) +def test_the_old_state_moves_and_the_plan_after_it_changes_nothing( + moto, tmp_path, monkeypatch, capsys, wipe_workdir +): + contract = _write(tmp_path, _contract()) + _apply_as_the_old_release(contract, tmp_path) + old = _object(moto, _LEGACY_KEY) + assert old["resources"], "the old release's apply wrote no resources" + assert _AWS_KEY not in _keys(moto) + if wipe_workdir: + # A CI workspace is wiped after every run: no .terraform/ records the + # old key, and the move still has to happen. + shutil.rmtree(tmp_path / ".fluid") + capsys.readouterr() + + _upgraded_apply(contract, tmp_path, monkeypatch) + + # The console wraps long lines; the checks read the text unwrapped. + printed = capsys.readouterr().out.replace("\n", "") + assert f"remote: s3://{_STATE_BUCKET}/{_AWS_KEY} (from FLUID_STATE_BACKEND)" in printed + assert f"state move: moved {len(old['resources'])} resource(s)" in printed + assert "tofu plan: +0 ~0 -0" in printed + moved = _object(moto, _AWS_KEY) + # The same resources (the move verified them exactly before the apply, + # whose refresh then rewrote the object at the new key). + assert _addresses(moved) == _addresses(old) + # Never lost: the old object is left exactly where it was. + assert _object(moto, _LEGACY_KEY) == old + + # The next run finds its state at the new key and moves nothing. + _upgraded_apply(contract, tmp_path, monkeypatch, dry_run=True) + again = capsys.readouterr().out.replace("\n", "") + assert "state move:" not in again + assert "tofu plan: +0 ~0 -0" in again + + +def test_a_dry_run_moves_nothing_and_plans_on_the_old_key(moto, tmp_path, monkeypatch, capsys): + """``fluid apply --dry-run`` is plan only: while the move is pending it + reads the old key, writes no object, and leaves the move to the first + real apply. It used to copy the state (measured: the new key appeared).""" + contract = _write(tmp_path, _contract()) + _apply_as_the_old_release(contract, tmp_path) + old = _object(moto, _LEGACY_KEY) + before = _keys(moto) + capsys.readouterr() + + _upgraded_apply(contract, tmp_path, monkeypatch, dry_run=True) + + printed = capsys.readouterr().out.replace("\n", "") + assert _keys(moto) == before, f"a dry-run wrote {sorted(_keys(moto) - before)}" + assert _object(moto, _LEGACY_KEY) == old + assert f"read from s3://{_STATE_BUCKET}/{_LEGACY_KEY}" in printed + assert "tofu plan: +0 ~0 -0" in printed + assert "dry-run: plan only" in printed + + # The first real apply, from the same (kept) workdir, does the move. + _upgraded_apply(contract, tmp_path, monkeypatch) + after = capsys.readouterr().out.replace("\n", "") + assert f"state move: moved {len(old['resources'])} resource(s)" in after + assert "tofu plan: +0 ~0 -0" in after + assert _addresses(_object(moto, _AWS_KEY)) == _addresses(old) + assert _object(moto, _LEGACY_KEY) == old + + +def test_the_drift_pass_reads_the_old_key_while_the_move_is_pending(moto, tmp_path, monkeypatch): + """``fluid diff`` / ``verify --state-drift`` before any upgraded apply: the + real state, at the old key, with every resource in it, and nothing written. + Pointed at the new key instead, the pass found an empty state and said + ``not_checked``, a silent pass for a contract whose resources exist.""" + from fluid_build.cli import _diff_state + + contract_path = _write(tmp_path, _contract()) + _apply_as_the_old_release(contract_path, tmp_path) + old = _object(moto, _LEGACY_KEY) + before = _keys(moto) + monkeypatch.setenv("FLUID_STATE_BACKEND", f"s3://{_STATE_BUCKET}") + args = argparse.Namespace(provider=None, state_backend=None, workspace_dir=str(tmp_path)) + + report = _diff_state.check_state_drift( + yaml.safe_load(contract_path.read_text(encoding="utf-8")), args, _LOG + ) + + assert report.status == "checked", report.detail + assert report.state == f"remote: s3://{_STATE_BUCKET}/{_LEGACY_KEY}" + managed = [r for r in old["resources"] if r.get("mode") == "managed"] + assert len(report.resources) == len(managed) + assert not report.has_drift + assert _keys(moto) == before + + +def test_the_data_loss_gate_still_closes_after_the_move(moto, tmp_path, monkeypatch): + """The moved state is the one the gate judges: renaming the table plans a + destroy, and without --allow-data-loss the apply refuses it.""" + contract = _write(tmp_path, _contract()) + _apply_as_the_old_release(contract, tmp_path) + monkeypatch.setenv("FLUID_STATE_BACKEND", f"s3://{_STATE_BUCKET}") + _write(tmp_path, _contract(table="customer_subscriptions_v2")) + args = argparse.Namespace( + contract=str(contract), + env=None, + provider=None, + workspace_dir=str(tmp_path), + state_backend=None, + dry_run=False, + allow_data_loss=False, + no_verify_plan_binding=True, + ) + with pytest.raises(CLIError) as exc: + engine.apply_via_opentofu(args, _LOG) + assert exc.value.event == "opentofu_data_loss_gate" + assert _AWS_KEY in _keys(moto) + + +def test_another_providers_state_at_the_old_key_is_left_alone(moto, tmp_path): + """The gcp apply of a contract whose aws state still sits at the old key: + it is the aws apply's to move, not the gcp apply's.""" + contract = _write(tmp_path, _contract()) + _apply_as_the_old_release(contract, tmp_path) + old = _object(moto, _LEGACY_KEY) + + outcome = _reconcile(tmp_path, _GCP_KEY, "gcp", migrate=True) + + assert outcome.outcome == mig.OTHER_PROVIDER + assert "aws" in outcome.detail + assert _GCP_KEY not in _keys(moto) + assert _object(moto, _LEGACY_KEY) == old + + +def test_a_state_holding_two_clouds_is_refused_and_nothing_moves(moto, tmp_path): + contract = _write(tmp_path, _contract()) + _apply_as_the_old_release(contract, tmp_path) + old = _object(moto, _LEGACY_KEY) + mixed = dict(old) + mixed["resources"] = list(old["resources"]) + [ + { + "mode": "managed", + "type": "google_bigquery_dataset", + "name": "bronze_customer_subscriptions_demo_bronze", + "provider": 'provider["registry.opentofu.org/hashicorp/google"]', + "instances": [], + } + ] + _s3(moto).put_object( + Bucket=_STATE_BUCKET, Key=_LEGACY_KEY, Body=json.dumps(mixed).encode("utf-8") + ) + + with pytest.raises(mig.StateMigrationError) as exc: + _reconcile(tmp_path, _AWS_KEY, "aws", migrate=True) + + assert exc.value.code == "state_migration_ambiguous" + assert "aws, gcp" in str(exc.value) + assert _AWS_KEY not in _keys(moto) + + +def test_a_new_key_that_holds_state_is_never_overwritten(moto, tmp_path, monkeypatch): + contract = _write(tmp_path, _contract()) + # This provider already applied at the new key... + monkeypatch.setenv("FLUID_STATE_BACKEND", f"s3://{_STATE_BUCKET}") + _apply(contract, tmp_path) + current = _object(moto, _AWS_KEY) + # ...and an older state of this provider still sits at the old one. + older = dict(current, lineage="00000000-0000-0000-0000-000000000000") + _s3(moto).put_object( + Bucket=_STATE_BUCKET, Key=_LEGACY_KEY, Body=json.dumps(older).encode("utf-8") + ) + + outcome = _reconcile(tmp_path, _AWS_KEY, "aws", migrate=True) + + assert outcome.outcome == mig.CURRENT + assert _object(moto, _AWS_KEY) == current + + +def test_a_read_only_caller_reads_the_old_key_until_the_apply_moves_it(moto, tmp_path): + contract = _write(tmp_path, _contract()) + _apply_as_the_old_release(contract, tmp_path) + + outcome = _reconcile(tmp_path, _AWS_KEY, "aws", migrate=False) + + assert outcome.outcome == mig.PENDING + assert outcome.read_from == _block(_LEGACY_KEY) + assert _AWS_KEY not in _keys(moto) diff --git a/tests/iac/test_state_backend_env_default.py b/tests/iac/test_state_backend_env_default.py index c57e8d2f..eb8fe546 100644 --- a/tests/iac/test_state_backend_env_default.py +++ b/tests/iac/test_state_backend_env_default.py @@ -35,6 +35,7 @@ from fluid_build.cli import _apply_opentofu_engine as engine from fluid_build.cli._common import CLIError +from fluid_build.iac.state_migration import CURRENT, StateReconciliation pytestmark = [pytest.mark.unit] @@ -84,6 +85,13 @@ def _init_stops_here(*_a: Any, **_k: Any) -> SimpleNamespace: return SimpleNamespace(ok=False, stderr="stub: tofu init not run", stdout="") monkeypatch.setattr(engine.runner, "tofu_init", _init_stops_here) + # The move of a pre-provider-key state (``iac.state_migration``) runs its + # own ``tofu init``; it is not under test here and finds nothing to move. + monkeypatch.setattr( + engine, + "_reconcile_state", + lambda **kw: StateReconciliation(CURRENT, kw["legacy"], kw["current"]), + ) args = argparse.Namespace( contract=str(contract), env=None, @@ -187,9 +195,9 @@ def test_a_bucket_only_env_value_gives_each_contract_its_own_s3_key( second, _ = _apply_and_read_backend( tmp_path / "b", monkeypatch, flag=None, contract_text=_OTHER_CONTRACT ) - assert first == {"s3": {"bucket": "ci-state", "key": "fluid/demo.state/terraform.tfstate"}} + assert first == {"s3": {"bucket": "ci-state", "key": "fluid/demo.state/aws/terraform.tfstate"}} assert second == { - "s3": {"bucket": "ci-state", "key": "fluid/demo.other_state/terraform.tfstate"} + "s3": {"bucket": "ci-state", "key": "fluid/demo.other_state/aws/terraform.tfstate"} } @@ -199,7 +207,7 @@ def test_a_bucket_only_env_value_gives_each_contract_its_own_gcs_prefix( ) -> None: monkeypatch.setenv("FLUID_STATE_BACKEND", spec) backend, _ = _apply_and_read_backend(tmp_path, monkeypatch, flag=None) - assert backend == {"gcs": {"bucket": "ci-state", "prefix": "fluid/demo.state"}} + assert backend == {"gcs": {"bucket": "ci-state", "prefix": "fluid/demo.state/aws"}} def test_the_flag_keeps_the_shared_legacy_key_for_a_contract_without_packaging( @@ -227,8 +235,8 @@ def test_ids_the_old_key_folded_together_get_their_own_state( flag=None, contract_text=_CONTRACT.replace("id: demo.state", "id: demo_state"), ) - assert dotted["s3"]["key"] == "fluid/demo.state/terraform.tfstate" - assert underscored["s3"]["key"] == "fluid/demo_state/terraform.tfstate" + assert dotted["s3"]["key"] == "fluid/demo.state/aws/terraform.tfstate" + assert underscored["s3"]["key"] == "fluid/demo_state/aws/terraform.tfstate" def test_an_id_that_cannot_key_a_state_is_a_typed_error_naming_the_variable( @@ -278,7 +286,7 @@ def test_the_state_line_names_the_object_each_form_resolved_to( " state: remote: s3://ci-state/fluid/terraform.tfstate (from --state-backend)" ] assert env_lines == [ - " state: remote: s3://ci-state/fluid/demo.state/terraform.tfstate" + " state: remote: s3://ci-state/fluid/demo.state/aws/terraform.tfstate" " (from FLUID_STATE_BACKEND)" ] diff --git a/tests/iac/test_state_key_per_provider.py b/tests/iac/test_state_key_per_provider.py new file mode 100644 index 00000000..0cb7cc25 --- /dev/null +++ b/tests/iac/test_state_key_per_provider.py @@ -0,0 +1,190 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The default remote state key names the provider. + +Measured on 0.16.5 with ``FLUID_STATE_BACKEND=s3://fluid-demo-lab-state-…``: +``resolve_state_target`` gave the aws and the gcp apply of +``bronze.customer_subscriptions`` the same key, +``fluid/bronze.customer_subscriptions/terraform.tfstate``, so the gcp plan +would have read the aws resources as orphans to destroy. Offline unit pins +for the key, the old key the move reads from, and how a state is attributed +to a provider; ``test_iac_state_key_migration_moto.py`` runs the move itself +with real ``tofu``. +""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +import pytest + +from fluid_build.cli._apply_opentofu_engine import resolve_state_target +from fluid_build.iac import state_migration as mig +from fluid_build.iac.backend import default_state_key, legacy_default_backend, parse_backend + +pytestmark = pytest.mark.unit + +_CID = "bronze.customer_subscriptions" +_CONTRACT = {"id": _CID, "name": "Customer Subscriptions"} +_PACKAGED = {"id": _CID, "packaging": {"mode": "isolated"}} + + +def _args(tmp_path: Path, flag=None) -> argparse.Namespace: + return argparse.Namespace(state_backend=flag, workspace_dir=str(tmp_path)) + + +def test_the_aws_and_the_gcp_apply_of_one_contract_get_two_states(tmp_path, monkeypatch): + monkeypatch.setenv("FLUID_STATE_BACKEND", "s3://fluid-demo-lab-state-111111111111") + aws = resolve_state_target(_args(tmp_path), _CONTRACT, "aws") + gcp = resolve_state_target(_args(tmp_path), _CONTRACT, "gcp") + assert aws.backend == { + "s3": { + "bucket": "fluid-demo-lab-state-111111111111", + "key": f"fluid/{_CID}/aws/terraform.tfstate", + } + } + assert gcp.backend["s3"]["key"] == f"fluid/{_CID}/gcp/terraform.tfstate" + # ...and both know where the previous release kept the state. + legacy = { + "s3": { + "bucket": "fluid-demo-lab-state-111111111111", + "key": f"fluid/{_CID}/terraform.tfstate", + } + } + assert aws.legacy_backend == legacy + assert gcp.legacy_backend == legacy + + +def test_a_gcs_prefix_names_the_provider_too(tmp_path, monkeypatch): + monkeypatch.setenv("FLUID_STATE_BACKEND", "gcs://team-state") + target = resolve_state_target(_args(tmp_path), _CONTRACT, "gcp") + assert target.backend == {"gcs": {"bucket": "team-state", "prefix": f"fluid/{_CID}/gcp"}} + assert target.legacy_backend == {"gcs": {"bucket": "team-state", "prefix": f"fluid/{_CID}"}} + + +def test_a_packaging_contract_on_the_flag_gets_the_provider_segment(tmp_path, monkeypatch): + monkeypatch.delenv("FLUID_STATE_BACKEND", raising=False) + target = resolve_state_target(_args(tmp_path, "s3://ci-state"), _PACKAGED, "aws") + assert ( + target.backend["s3"]["key"] == "fluid/bronze_customer_subscriptions/aws/terraform.tfstate" + ) + assert target.legacy_backend["s3"]["key"] == ( + "fluid/bronze_customer_subscriptions/terraform.tfstate" + ) + + +@pytest.mark.parametrize( + "flag", + ["s3://ci-state", "s3://ci-state/team/explicit.tfstate", "gcs://ci-state/team/x", ""], + ids=["legacy-shared-key", "explicit-key", "explicit-prefix", "local"], +) +def test_keys_nobody_defaulted_are_never_moved(tmp_path, monkeypatch, flag): + """The shared legacy key, an explicit key or prefix and local state keep + their location, so there is nothing to migrate from.""" + monkeypatch.delenv("FLUID_STATE_BACKEND", raising=False) + target = resolve_state_target(_args(tmp_path, flag), _CONTRACT, "aws") + assert target.legacy_backend is None + if flag == "s3://ci-state": + assert target.backend == {"s3": {"bucket": "ci-state", "key": "fluid/terraform.tfstate"}} + + +def test_parse_backend_without_a_provider_is_unchanged(): + """Callers that do not name a provider keep the old keys byte for byte.""" + assert parse_backend("s3://b", _CONTRACT, per_contract_default=True) == { + "s3": {"bucket": "b", "key": f"fluid/{_CID}/terraform.tfstate"} + } + assert default_state_key(_CONTRACT) == "fluid/terraform.tfstate" + + +def test_a_provider_that_is_not_one_key_segment_is_refused(): + with pytest.raises(ValueError, match="cannot name a state key segment"): + default_state_key(_CONTRACT, per_contract=True, provider="../aws") + with pytest.raises(ValueError): + legacy_default_backend("s3://b", _CONTRACT, per_contract_default=True, provider="a/b") + + +# ── Whose state is it? ──────────────────────────────────────────────────── + + +def _res(source: str, rtype: str = "x") -> dict: + return {"type": rtype, "provider": f'provider["registry.opentofu.org/{source}"]'} + + +@pytest.mark.parametrize( + "resources, provider, verdict", + [ + ([_res("hashicorp/aws"), _res("hashicorp/null")], "aws", "mine"), + ([_res("hashicorp/google")], "gcp", "mine"), + ([_res("hashicorp/aws")], "gcp", "other"), + ([_res("hashicorp/google")], "aws", "other"), + ([_res("hashicorp/aws"), _res("hashicorp/google")], "aws", "ambiguous"), + ([_res("hashicorp/aws"), _res("hashicorp/google")], "gcp", "ambiguous"), + ([_res("hashicorp/random")], "aws", "ambiguous"), + ([{"type": "x", "provider": "not an address"}], "aws", "ambiguous"), + ([_res("snowflake-labs/snowflake")], "snowflake", "mine"), + ], + ids=[ + "aws-own", + "gcp-own", + "aws-state-seen-by-gcp", + "gcp-state-seen-by-aws", + "two-clouds-aws", + "two-clouds-gcp", + "unknown-provider", + "unreadable-address", + "snowflake-pre-v2-source", + ], +) +def test_a_state_is_attributed_by_its_resources_providers(resources, provider, verdict): + assert mig.classify(resources, provider)[0] == verdict + + +def test_a_terraform_written_state_is_read_the_same(): + resources = [{"provider": 'provider["registry.terraform.io/hashicorp/aws"].west'}] + assert mig.classify(resources, "aws")[0] == "mine" + + +def test_state_pull_output_is_parsed_and_an_absent_state_is_empty(): + assert not mig.parse_state("").exists + empty = '{"version":4,"serial":0,"lineage":"","resources":[]}' + assert not mig.parse_state(empty).exists + doc = mig.parse_state(json.dumps({"lineage": "L", "serial": 3, "resources": [_res("a/b")]})) + assert (doc.exists, doc.serial, len(doc.resources)) == (True, 3, 1) + with pytest.raises(mig.StateMigrationError): + mig.parse_state("not json") + + +def test_the_workdir_s_recorded_backend_is_compared_on_the_fields_forge_writes(tmp_path): + (tmp_path / ".terraform").mkdir() + (tmp_path / ".terraform" / "terraform.tfstate").write_text( + json.dumps( + { + "backend": { + "type": "s3", + "config": {"bucket": "b", "key": "fluid/x/terraform.tfstate", "region": None}, + } + } + ), + encoding="utf-8", + ) + assert mig.records_backend( + tmp_path, {"s3": {"bucket": "b", "key": "fluid/x/terraform.tfstate"}} + ) + assert not mig.records_backend( + tmp_path, {"s3": {"bucket": "b", "key": "fluid/x/aws/terraform.tfstate"}} + ) + assert not mig.records_backend(tmp_path / "nowhere", {"s3": {"bucket": "b"}}) diff --git a/tests/iac/test_state_migration_safety.py b/tests/iac/test_state_migration_safety.py new file mode 100644 index 00000000..3cb907fd --- /dev/null +++ b/tests/iac/test_state_migration_safety.py @@ -0,0 +1,572 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The state move's safety branches, and what its scratch ``tofu init`` installs. + +Three branches of ``state_migration.reconcile_state_key`` had no test: a +mutant removing the re-check before the copy, one disabling the check after +it, and one pointing ``fluid diff`` at the new key all passed the suite. Here +the ``tofu`` calls are replaced by a scripted fake, so each branch is driven +on purpose: another job writing the new key between the first check and the +copy (``state_migration_raced``), a copy that reads back different resources +(``state_migration_unverified``), and a probe whose init failed before it +recorded the old backend (an error, never an empty state). + +The probe's scratch init used to install the providers the old state names, +at their latest version (OpenTofu installs what the state requires): a +``{"terraform": {}}`` module beside a state naming ``hashicorp/null`` +installed ``hashicorp/null v3.3.2``, and it ran on every apply whose new key +was empty. It now installs nothing (``-plugin-dir`` on an empty directory), +and the one-time move installs at the plugin's pins. The offline tests prove +the probe with real ``tofu`` and no registry reachable (a dead proxy): it +attributes the state; before, it reached for the registry and failed. + +Also here: the apply's refusal to plan on a key that names no provider when +it holds another cloud's resources, and ``fluid apply --dry-run`` asking the +move not to run. +""" + +from __future__ import annotations + +import argparse +import json +import logging +import os +import sys +from pathlib import Path +from typing import Any, Dict, List + +import pytest + +from fluid_build.cli import _apply_opentofu_engine as engine +from fluid_build.cli._common import CLIError +from fluid_build.iac import runner +from fluid_build.iac import state_migration as mig +from fluid_build.iac.runner import TofuResult + +pytestmark = pytest.mark.unit + +_AWS = 'provider["registry.opentofu.org/hashicorp/aws"]' +_GOOGLE = 'provider["registry.opentofu.org/hashicorp/google"]' + + +def _state(lineage: str, *resources: Dict[str, Any]) -> Dict[str, Any]: + return { + "version": 4, + "terraform_version": "1.12.0", + "serial": 3, + "lineage": lineage, + "outputs": {}, + "resources": list(resources), + } + + +def _resource(rtype: str, name: str, provider: str) -> Dict[str, Any]: + return { + "mode": "managed", + "type": rtype, + "name": name, + "provider": provider, + # As `tofu state pull` prints it (it adds the empty list). + "instances": [ + {"schema_version": 0, "attributes": {"id": name}, "sensitive_attributes": []} + ], + } + + +_EMPTY = _state("") # what `tofu state pull` prints for a key with no state +_OLD = _state("11111111-aaaa", _resource("aws_s3_bucket", "lake", _AWS)) +_OTHER_JOB = _state("22222222-bbbb", _resource("aws_s3_bucket", "someone_else", _AWS)) + +_CURRENT = {"s3": {"bucket": "b", "key": "fluid/p/aws/terraform.tfstate"}} +_LEGACY = {"s3": {"bucket": "b", "key": "fluid/p/terraform.tfstate"}} + + +class _FakeTofu: + """Scripted ``tofu init`` / ``tofu state pull``, keyed by the directory's name. + + ``workdir`` is the apply's own directory (the new key), ``legacy`` the + probe's, ``legacy-plain`` its fallback and ``move`` the copy's. + """ + + def __init__(self, workdir_pulls: List[Dict[str, Any]], legacy: Dict[str, Any]) -> None: + self.workdir_pulls = list(workdir_pulls) + self.legacy = legacy + self.calls: List[Dict[str, Any]] = [] + #: The probe's -plugin-dir init: 0, or 1 as when it stops at the + #: provider step; ``records`` says whether it recorded the backend. + self.probe_init_rc = 0 + self.records = False + self.plain_init_rc = 0 + self.modules: Dict[str, Any] = {} + + def init(self, workdir: str, **kwargs: Any) -> TofuResult: + path = Path(workdir) + name = path.name + module = json.loads((path / "main.tf.json").read_text(encoding="utf-8")) + self.modules[f"{name}{'-copy' if kwargs.get('force_copy') else ''}"] = module + plugin_dir = kwargs.get("plugin_dir") + self.calls.append( + { + "init": name, + **kwargs, + "plugin_dir_empty": ( + plugin_dir is not None and not any(Path(plugin_dir).iterdir()) + ), + } + ) + rc = 0 + if name == "legacy": + rc = self.probe_init_rc + if self.records: + recorded = {"version": 3, "backend": {"type": "s3", "config": _LEGACY["s3"]}} + (path / ".terraform").mkdir(exist_ok=True) + (path / ".terraform" / "terraform.tfstate").write_text(json.dumps(recorded)) + elif name == "legacy-plain": + rc = self.plain_init_rc + return TofuResult("init", rc, "", "Error: Failed to query available provider packages") + + def pull(self, workdir: str, **_kwargs: Any) -> TofuResult: + name = Path(workdir).name + self.calls.append({"pull": name}) + doc = self.legacy if name.startswith("legacy") else self.workdir_pulls.pop(0) + return TofuResult("state-pull", 0, json.dumps(doc), "") + + @property + def copies(self) -> List[Dict[str, Any]]: + return [c for c in self.calls if c.get("force_copy")] + + +@pytest.fixture +def fake(monkeypatch): + def install(workdir_pulls, legacy=_OLD): + tofu = _FakeTofu(workdir_pulls, legacy) + monkeypatch.setattr(mig.runner, "tofu_init", tofu.init) + monkeypatch.setattr(mig.runner, "tofu_state_pull", tofu.pull) + return tofu + + return install + + +def _reconcile(tmp_path: Path, *, migrate: bool = True): + workdir = tmp_path / "workdir" + workdir.mkdir(exist_ok=True) + return mig.reconcile_state_key( + workdir=workdir, + current=_CURRENT, + legacy=_LEGACY, + provider="aws", + env={}, + migrate=migrate, + ) + + +# ── the re-check before the copy ───────────────────────────────────────── + + +def test_a_state_written_to_the_new_key_before_the_copy_is_never_overwritten(fake, tmp_path): + """Empty at the first look, another job's state at the second: refuse, copy nothing.""" + tofu = fake([_EMPTY, _OTHER_JOB]) + with pytest.raises(mig.StateMigrationError) as exc: + _reconcile(tmp_path) + assert exc.value.code == "state_migration_raced" + assert "nothing was moved" in str(exc.value) + assert tofu.copies == [] + + +def test_the_same_state_arriving_first_is_used_and_nothing_is_copied(fake, tmp_path): + """Another run of this apply moved it between the two looks: that is the state.""" + tofu = fake([_EMPTY, _state("33333333-cccc", *_OLD["resources"])]) + outcome = _reconcile(tmp_path) + assert outcome.outcome == mig.CURRENT + assert tofu.copies == [] + + +# ── the check after the copy ───────────────────────────────────────────── + + +def test_a_copy_that_reads_back_different_resources_is_an_error(fake, tmp_path): + tofu = fake([_EMPTY, _EMPTY, _OTHER_JOB]) + with pytest.raises(mig.StateMigrationError) as exc: + _reconcile(tmp_path) + assert exc.value.code == "state_migration_unverified" + assert "the old object is untouched" in str(exc.value) + # One copy, and nothing after it: no second attempt, no clean-up write. + assert len(tofu.copies) == 1 + assert tofu.calls[-1] == {"pull": "workdir"} + + +def test_a_copy_that_reads_back_the_same_resources_is_the_move(fake, tmp_path): + fake([_EMPTY, _EMPTY, _state("44444444-dddd", *_OLD["resources"])]) + outcome = _reconcile(tmp_path) + assert outcome.outcome == mig.MIGRATED + assert outcome.resources == 1 + + +# ── what the probe installs ────────────────────────────────────────────── + + +def test_the_probe_installs_nothing(fake, tmp_path): + """The probe's init takes providers from an empty directory only, never + from the apply's own ``.terraform/providers`` (whose entries a plugin + cache links; read as a mirror they broke the workdir).""" + (tmp_path / "workdir" / ".terraform" / "providers").mkdir(parents=True) + tofu = fake([_EMPTY]) + assert _reconcile(tmp_path, migrate=False).outcome == mig.PENDING + (probe,) = [c for c in tofu.calls if c.get("init") == "legacy"] + assert probe["plugin_dir_empty"] is True + assert ".terraform" not in probe["plugin_dir"] + assert [c for c in tofu.calls if "init" in c] == [probe] + + +def test_a_probe_stopped_at_the_provider_step_reads_the_recorded_backend(fake, tmp_path): + """The usual path for a state that names providers: init exits 1 after + recording the old backend, and the state is pulled from it.""" + tofu = fake([_EMPTY]) + tofu.probe_init_rc, tofu.records = 1, True + assert _reconcile(tmp_path, migrate=False).outcome == mig.PENDING + assert {"pull": "legacy"} in tofu.calls + assert not [c for c in tofu.calls if c.get("init") == "legacy-plain"] + + +def test_a_probe_that_recorded_no_backend_falls_back_to_a_plain_init(fake, tmp_path): + tofu = fake([_EMPTY]) + tofu.probe_init_rc = 1 + assert _reconcile(tmp_path, migrate=False).outcome == mig.PENDING + (plain,) = [c for c in tofu.calls if c.get("init") == "legacy-plain"] + assert plain.get("plugin_dir") is None + assert {"pull": "legacy"} not in tofu.calls # never read from an unrecorded backend + + +def test_a_probe_that_cannot_initialise_at_all_is_an_error_not_an_empty_state(fake, tmp_path): + tofu = fake([_EMPTY]) + tofu.probe_init_rc, tofu.plain_init_rc = 1, 1 + with pytest.raises(mig.StateMigrationError) as exc: + _reconcile(tmp_path, migrate=False) + assert exc.value.code == "state_migration_probe_failed" + assert not [c for c in tofu.calls if str(c.get("pull", "")).startswith("legacy")] + + +def test_the_move_installs_what_the_old_state_names_at_the_plugin_s_pins(fake, tmp_path): + """Once per contract, the copy's init installs providers: the pinned + ``~> 5.0`` aws, not the latest, and only what the state names.""" + tofu = fake([_EMPTY, _EMPTY, _state("55555555-eeee", *_OLD["resources"])]) + assert _reconcile(tmp_path).outcome == mig.MIGRATED + want = {"aws": {"source": "hashicorp/aws", "version": "~> 5.0"}} + assert tofu.modules["move"]["terraform"]["required_providers"] == want + assert tofu.modules["move"]["terraform"]["backend"] == _LEGACY + assert tofu.modules["move-copy"]["terraform"]["required_providers"] == want + assert tofu.modules["move-copy"]["terraform"]["backend"] == _CURRENT + + +def test_the_module_docstring_no_longer_claims_the_probe_downloads_nothing(): + assert "downloads nothing" not in (mig.__doc__ or "") + assert "-plugin-dir" in (mig.__doc__ or "") + + +# ── real tofu, no registry reachable ───────────────────────────────────── + +_TOFU = runner.tofu_path() +offline = pytest.mark.skipif( + _TOFU is None or sys.platform.startswith("win"), + reason="needs `tofu` on PATH (and a POSIX dead-proxy setup)", +) + + +def _offline_env(tmp_path: Path) -> Dict[str, str]: + """Every registry request fails fast: a dead proxy, no CLI config, no cache.""" + cli_config = tmp_path / "empty.tofurc" + cli_config.write_text("", encoding="utf-8") + env = {k: v for k, v in os.environ.items() if k != "TF_PLUGIN_CACHE_DIR"} + env.update( + { + "HTTPS_PROXY": "http://127.0.0.1:9", + "HTTP_PROXY": "http://127.0.0.1:9", + "NO_PROXY": "", + "TF_CLI_CONFIG_FILE": str(cli_config), + "TF_IN_AUTOMATION": "1", + } + ) + return env + + +def _local_workdir(tmp_path: Path, env: Dict[str, str]) -> Dict[str, Any]: + """The apply's workdir, initialised on a local "new key", as the apply leaves it.""" + workdir = tmp_path / "workdir" + workdir.mkdir() + current = {"local": {"path": str(tmp_path / "new" / "terraform.tfstate")}} + (workdir / "main.tf.json").write_text( + json.dumps({"terraform": {"backend": current}}), encoding="utf-8" + ) + assert runner.tofu_init(str(workdir), env=env).ok + return current + + +def _legacy_file(tmp_path: Path, doc: Dict[str, Any]) -> Dict[str, Any]: + path = tmp_path / "old" / "terraform.tfstate" + path.parent.mkdir() + path.write_text(json.dumps(doc), encoding="utf-8") + return {"local": {"path": str(path)}} + + +@offline +def test_the_probe_of_this_provider_s_state_reaches_no_registry(tmp_path): + env = _offline_env(tmp_path) + current = _local_workdir(tmp_path, env) + legacy = _legacy_file(tmp_path, _OLD) + + outcome = mig.reconcile_state_key( + workdir=tmp_path / "workdir", + current=current, + legacy=legacy, + provider="aws", + env=env, + migrate=False, + ) + + assert outcome.outcome == mig.PENDING + assert outcome.resources == 1 + assert not Path(current["local"]["path"]).exists() + assert json.loads(Path(legacy["local"]["path"]).read_text(encoding="utf-8")) == _OLD + + +@offline +def test_another_cloud_s_state_is_attributed_with_nothing_installed(tmp_path): + """The gcp apply reading the aws state at the old key: the aws provider is + not fetched to find out whose state it is.""" + env = _offline_env(tmp_path) + current = _local_workdir(tmp_path, env) + legacy = _legacy_file(tmp_path, _OLD) + + outcome = mig.reconcile_state_key( + workdir=tmp_path / "workdir", + current=current, + legacy=legacy, + provider="gcp", + env=env, + migrate=True, + ) + + assert outcome.outcome == mig.OTHER_PROVIDER + assert "aws" in outcome.detail + assert not Path(current["local"]["path"]).exists() + + +# ── one state, two clouds, on a key that names no provider ─────────────── + + +def _target(tmp_path: Path, key: str) -> engine.StateTarget: + return engine.StateTarget( + workdir=tmp_path, backend={"s3": {"bucket": "b", "key": key}}, origin="--state-backend" + ) + + +def _stub_state(monkeypatch, doc: Dict[str, Any]) -> List[Path]: + seen: List[Path] = [] + + def read_state(workdir, env): + seen.append(Path(workdir)) + return mig.parse_state(json.dumps(doc)) + + monkeypatch.setattr(engine, "read_state", read_state) + return seen + + +def test_the_shared_key_holding_the_other_cloud_s_resources_is_refused(tmp_path, monkeypatch): + _stub_state(monkeypatch, _OLD) + with pytest.raises(CLIError) as exc: + engine.guard_state_shared_with_another_cloud( + _target(tmp_path, "fluid/terraform.tfstate"), "gcp", {} + ) + assert exc.value.event == "state_shared_with_another_provider" + assert exc.value.context["providers"] == ["aws"] + assert exc.value.context["state"] == "s3://b/fluid/terraform.tfstate" + + +@pytest.mark.parametrize( + "doc", + [_OLD, _EMPTY, _state("x", _resource("null_resource", "n", 'provider["x/hashicorp/null"]'))], +) +def test_this_cloud_s_own_or_no_one_s_resources_pass(tmp_path, monkeypatch, doc): + _stub_state(monkeypatch, doc) + engine.guard_state_shared_with_another_cloud( + _target(tmp_path, "fluid/terraform.tfstate"), "aws", {} + ) + + +def test_a_key_that_names_the_provider_is_not_read(tmp_path, monkeypatch): + seen = _stub_state(monkeypatch, _state("y", _resource("g", "d", _GOOGLE))) + engine.guard_state_shared_with_another_cloud( + _target(tmp_path, "fluid/p/aws/terraform.tfstate"), "aws", {} + ) + local = engine.StateTarget(workdir=tmp_path, backend=None, origin="default") + engine.guard_state_shared_with_another_cloud(local, "aws", {}) + assert seen == [] + + +def test_the_bucket_only_flag_route_is_refused_end_to_end(tmp_path, monkeypatch): + """--state-backend s3:// and a contract without packaging: both + clouds resolve to fluid/terraform.tfstate, and the gcp apply finds aws there.""" + monkeypatch.delenv("FLUID_STATE_BACKEND", raising=False) + contract = {"id": "bronze.customer_subscriptions", "name": "Customer Subscriptions"} + args = argparse.Namespace( + state_backend="s3://fluid-demo-lab-state", workspace_dir=str(tmp_path) + ) + aws = engine.resolve_state_target(args, contract, "aws") + gcp = engine.resolve_state_target(args, contract, "gcp") + assert aws.backend == gcp.backend # the shared legacy key, unchanged + _stub_state(monkeypatch, _OLD) + with pytest.raises(CLIError) as exc: + engine.guard_state_shared_with_another_cloud(gcp, "gcp", {}) + assert exc.value.event == "state_shared_with_another_provider" + + +# ── fluid apply --dry-run never moves state ────────────────────────────── + + +class _Stop(Exception): + pass + + +def _engine_until_the_guard(monkeypatch, tmp_path: Path, *, dry_run: bool) -> Dict[str, Any]: + """Run the engine with tofu stubbed, up to the shared-state guard.""" + seen: Dict[str, Any] = {"inits": []} + contract = { + "fluidVersion": "0.7.5", + "kind": "DataProduct", + "id": "bronze.customer_subscriptions", + "name": "Customer Subscriptions", + "metadata": {"owner": {"team": "t"}}, + "exposes": [ + { + "exposeId": "subscriptions", + "kind": "table", + "binding": { + "platform": "aws", + "format": "parquet", + "location": {"bucket": "lake", "path": "bronze/", "database": "d"}, + }, + "contract": {"schema": [{"name": "a", "type": "VARCHAR"}]}, + } + ], + } + monkeypatch.setenv("FLUID_STATE_BACKEND", "s3://state-bucket") + monkeypatch.setattr(engine, "_verify_plan_binding_for_opentofu", lambda *a, **k: None) + monkeypatch.setattr(engine, "_load_contract", lambda *a, **k: contract) + monkeypatch.setattr(engine, "native_actions", lambda *a, **k: []) + monkeypatch.setattr(engine.runner, "tofu_path", lambda: "/usr/bin/tofu") + monkeypatch.setattr(engine.runner, "require_tofu_version", lambda *a, **k: None) + monkeypatch.setattr(engine, "cprint", lambda *a, **k: None) + + def init(workdir, **kwargs): + module = json.loads((Path(workdir) / "main.tf.json").read_text(encoding="utf-8")) + seen["inits"].append((module["terraform"]["backend"]["s3"]["key"], kwargs)) + return TofuResult("init", 0, "", "") + + def reconcile(**kwargs): + seen["migrate"] = kwargs["migrate"] + return mig.StateReconciliation(mig.PENDING, kwargs["legacy"], kwargs["current"], 1) + + def stop(target, provider, env): + seen["guarded"] = engine.backend_location(target.backend) + raise _Stop + + monkeypatch.setattr(engine.runner, "tofu_init", init) + monkeypatch.setattr(engine, "_reconcile_state", reconcile) + monkeypatch.setattr(engine, "guard_state_shared_with_another_cloud", stop) + args = argparse.Namespace( + contract="c.fluid.yaml", + env=None, + provider="aws", + workspace_dir=str(tmp_path), + state_backend=None, + dry_run=dry_run, + allow_data_loss=False, + no_verify_plan_binding=True, + ) + with pytest.raises(_Stop): + engine.apply_via_opentofu(args, logging.getLogger("test.state_migration_safety")) + return seen + + +def test_a_dry_run_plans_on_the_old_key_and_never_asks_for_the_move(monkeypatch, tmp_path): + seen = _engine_until_the_guard(monkeypatch, tmp_path, dry_run=True) + assert seen["migrate"] is False + # Initialised on the new key first, then re-pointed at the old one. + assert [key for key, _ in seen["inits"]] == [ + "fluid/bronze.customer_subscriptions/aws/terraform.tfstate", + "fluid/bronze.customer_subscriptions/terraform.tfstate", + ] + assert seen["inits"][1][1].get("reconfigure") is True + assert ( + seen["guarded"] == "s3://state-bucket/fluid/bronze.customer_subscriptions/terraform.tfstate" + ) + + +def test_a_real_apply_asks_for_the_move(monkeypatch, tmp_path): + seen = _engine_until_the_guard(monkeypatch, tmp_path, dry_run=False) + assert seen["migrate"] is True + assert len(seen["inits"]) == 1 + + +# ── OpenTofu's built-in provider names no cloud ────────────────────────── +# +# The GCP plugin writes a ``terraform_data`` beside a table whose partitions +# expire (``lifecycle.retention`` with ``expire: true``): the trigger that +# replaces the table when its partitioning changes. OpenTofu records it under +# ``provider["terraform.io/builtin/terraform"]``. Read as a provider no plugin +# emits, it made the legacy state of every such gcp product "ambiguous", and +# the refusal blocked apply, dry-run and diff (measured on the integration of +# the governance branch with this one). + +_BUILTIN = 'provider["terraform.io/builtin/terraform"]' +_GCP_LEGACY_WITH_TRIGGER = _state( + "55555555-eeee", + _resource("google_bigquery_dataset", "hunt_retention", _GOOGLE), + _resource("google_bigquery_table", "hunt_retention_events", _GOOGLE), + _resource("terraform_data", "hunt_retention_events_partitioning", _BUILTIN), +) + + +def test_a_gcp_state_holding_the_partition_trigger_is_the_gcp_apply_s(): + resources = _GCP_LEGACY_WITH_TRIGGER["resources"] + assert mig.classify(resources, "gcp") == ("mine", "gcp") + assert mig.classify(resources, "aws")[0] == "other" + assert mig.other_clouds(resources, "aws") == frozenset({"gcp"}) + + +def test_a_state_of_only_built_in_resources_is_still_not_guessed(): + verdict, detail = mig.classify([_resource("terraform_data", "t", _BUILTIN)], "gcp") + assert verdict == "ambiguous" + assert "built-in" in detail + + +def test_the_gcp_state_with_the_trigger_is_moved_and_nothing_pins_the_built_in(fake, tmp_path): + moved = _state("66666666-ffff", *_GCP_LEGACY_WITH_TRIGGER["resources"]) + tofu = fake([_EMPTY, _EMPTY, moved], legacy=_GCP_LEGACY_WITH_TRIGGER) + workdir = tmp_path / "workdir" + workdir.mkdir() + outcome = mig.reconcile_state_key( + workdir=workdir, + current={"s3": {"bucket": "b", "key": "fluid/p/gcp/terraform.tfstate"}}, + legacy=_LEGACY, + provider="gcp", + env={}, + migrate=True, + ) + assert outcome.outcome == mig.MIGRATED + assert outcome.resources == 3 + assert len(tofu.copies) == 1 + pinned = tofu.modules["move-copy"]["terraform"].get("required_providers") or {} + assert [spec["source"] for spec in pinned.values()] == ["hashicorp/google"] diff --git a/tests/policy/test_sovereignty_enforcement_mode.py b/tests/policy/test_sovereignty_enforcement_mode.py index a5c7d0eb..621b7f48 100644 --- a/tests/policy/test_sovereignty_enforcement_mode.py +++ b/tests/policy/test_sovereignty_enforcement_mode.py @@ -240,18 +240,53 @@ def test_catch_all_jurisdiction_is_not_a_violation_anywhere(jurisdiction: str) - assert SovereigntyValidator().validate(doc)[0] is True -def test_unknown_region_does_not_block_under_strict() -> None: - """An unmappable region is 'cannot tell', not 'violates'. - - Deliberately unchanged. Failing closed here would be defensible for a - sovereignty control, but it would break every contract using a region the - vendored table does not carry, so it is a separate decision rather than a - side effect of this fix. +def test_unknown_region_on_a_cloud_blocks_under_strict() -> None: + """An unmappable cloud region under a strict jurisdiction is refused. + + This used to be a warning ("cannot tell" is not "violates"), which failed + open: the GCP table lagged Google by nine regions and the ASIA + multi-region, so an EU-only strict contract validated and emitted + me-central2. On an aws / gcp / azure binding strict now refuses a + jurisdiction it cannot show; advisory still warns. """ doc = contract(region="mars-central-1", enforcementMode="strict", jurisdiction="EU") is_valid, violations = SovereigntyValidator().validate(doc) + assert is_valid is False + (finding,) = _jurisdiction_findings(violations) + assert finding.severity == "error" + assert "sovereignty.allowedRegions" in (finding.suggestion or "") + + advisory = contract(region="mars-central-1", enforcementMode="advisory", jurisdiction="EU") + is_valid, violations = SovereigntyValidator().validate(advisory) + assert is_valid is True + assert [v.severity for v in _jurisdiction_findings(violations)] == ["warning"] + + +def _jurisdiction_findings(violations: List[Any]) -> List[Any]: + """Check 3's findings (check 4 adds its own "no known jurisdiction" warning).""" + return [v for v in violations if "does not match required jurisdiction" in v.message] + + +def test_an_unknown_region_named_in_allowed_regions_still_only_warns() -> None: + """The operator vouched for it: the one way to use a region the table lacks.""" + doc = contract( + region="mars-central-1", + enforcementMode="strict", + jurisdiction="EU", + allowedRegions=["mars-central-1"], + ) + is_valid, violations = SovereigntyValidator().validate(doc) + assert is_valid is True + assert [v.severity for v in _jurisdiction_findings(violations)] == ["warning"] + + +def test_an_unknown_region_off_the_clouds_still_only_warns() -> None: + """A platform whose region is not a cloud region keeps the warning.""" + doc = contract(region="mars-central-1", enforcementMode="strict", jurisdiction="EU") + doc["exposes"][0]["binding"]["platform"] = "snowflake" + is_valid, violations = SovereigntyValidator().validate(doc) assert is_valid is True - assert any(v.severity == "warning" for v in violations) + assert [v.severity for v in _jurisdiction_findings(violations)] == ["warning"] def test_explicit_deny_is_an_error_in_every_mode() -> None: diff --git a/tests/providers/test_bigquery_emulated_embedded_sql_chain.py b/tests/providers/test_bigquery_emulated_embedded_sql_chain.py new file mode 100644 index 00000000..45393c42 --- /dev/null +++ b/tests/providers/test_bigquery_emulated_embedded_sql_chain.py @@ -0,0 +1,449 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Three products, one contract each, on the local target and on GCP (bigquery-emulator). + +Bronze acquires a file and lands it with ``msisdn`` hashed; silver aggregates +bronze with an embedded-SQL build on DuckDB; gold joins bronze to silver. Each +contract is written once, with a ``gcp`` overlay that changes nothing but +``exposes[0].binding`` (a ``bigquery_table``, with the ``gs://`` staging path +the demo's make-targets forces into every overlay). The chain runs twice: + +* ``--env local``: every product lands a local Parquet file; +* ``--env gcp``: bronze is loaded into BigQuery, and silver and gold read + their upstreams FROM BigQuery and load their results INTO BigQuery. + +and each gcp table must hold exactly the rows its local file holds. Then +``fluid verify --env gcp --strict`` passes on bronze, and fails once one +``msisdn`` in the table is overwritten with a cleartext number. + +What stands in for ``fluid apply``'s OpenTofu step: the datasets and tables +are created in the emulator from the module forge-cli's own GCP emitter +writes for each contract (same ids, same schema JSON). ``tofu apply`` itself +cannot run here: terraform-provider-google 6.50.0 panics reading back a +goccy dataset (nil interface conversion in resourceBigQueryDatasetRead), +measured. Everything after it is ``run_builds_from_args``, the function +``fluid apply --mode amend-and-build`` calls, and ``fluid verify``'s ``run``. + +What no emulator can prove, and this does not claim: that real BigQuery +accepts these loads (the goccy emulator does not check the Parquet +timestamp's ``isAdjustedToUTC`` against the column type; the unit tests pin +that the file sent is UTC-adjusted), IAM on real datasets, Workload Identity +Federation, and the Storage Read API (not used). + +Keyless. Runs when ``FLUID_GCP_BIGQUERY_EMULATOR`` names a reachable +goccy/bigquery-emulator (the heavy emulated lane starts one from +``tests/iac/_gcp_emulator/docker-compose.yml``), and skips otherwise; +``scripts/ci/assert_lane_coverage.py`` fails that lane if it only skipped. +""" + +from __future__ import annotations + +import argparse +import json +import logging +import os +import re +import uuid +from pathlib import Path +from typing import Any, Dict, Iterator, List + +import pytest +import yaml + +pytestmark = [pytest.mark.integration, pytest.mark.emulated_heavy] + +duckdb = pytest.importorskip("duckdb") +pytest.importorskip("pyarrow") +bigquery = pytest.importorskip("google.cloud.bigquery") + +_LOG = logging.getLogger("test.emulated.chain") +BRONZE = "bronze.customer_subscriptions" +SILVER = "silver.subscription_status_summary" +GOLD = "gold.retention_candidates" +HASH_RE = re.compile(r"[0-9a-f]{64}") +ROWS = 600 + +SILVER_SQL = """ +SELECT product_id, status, COUNT(*) AS subscription_count +FROM subscriptions +GROUP BY product_id, status +""" +GOLD_SQL = """ +SELECT s.customer_id, s.subscription_id, s.product_id, s.status, s.msisdn, + t.subscription_count AS cohort_size +FROM subscriptions s +JOIN status_summary t ON s.product_id = t.product_id AND s.status = t.status +WHERE s.status IN ('suspended', 'expired') AND t.subscription_count >= 10 +""" + + +def _emulator() -> str: + host = os.environ.get("FLUID_GCP_BIGQUERY_EMULATOR", "").strip() + if not host: + pytest.skip("FLUID_GCP_BIGQUERY_EMULATOR is not set (the heavy emulated lane sets it)") + return host + + +@pytest.fixture +def emulator(monkeypatch, tmp_path) -> Iterator[Dict[str, Any]]: + """The emulator, a client for it, and a process pointed at it with no credentials.""" + from google.auth.credentials import AnonymousCredentials + + host = _emulator() + project = os.environ.get("FLUID_GCP_BIGQUERY_EMULATOR_PROJECT", "fluid-emulator") + # The process under test: no ADC anywhere, only the emulator variable, + # which python-bigquery reads for its endpoint. + monkeypatch.setenv("BIGQUERY_EMULATOR_HOST", host) + client = bigquery.Client(project=project, credentials=AnonymousCredentials()) + try: + list(client.list_datasets(max_results=1)) + except Exception as exc: # noqa: BLE001 + pytest.skip(f"bigquery-emulator at {host} is not reachable: {exc}") + empty = tmp_path / "no-gcloud" + empty.mkdir() + monkeypatch.setenv("CLOUDSDK_CONFIG", str(empty)) + monkeypatch.delenv("GOOGLE_APPLICATION_CREDENTIALS", raising=False) + monkeypatch.setenv("FLUID_PII_HASH_SECRET", "wf43-emulated-chain-salt-0123456789") + suffix = uuid.uuid4().hex[:8] + yield {"client": client, "project": project, "suffix": suffix} + + +# ── The workspace ─────────────────────────────────────────────────────── + + +def _dump(path: Path, doc: Dict[str, Any]) -> Path: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(yaml.safe_dump(doc, sort_keys=False), encoding="utf-8") + return path + + +def _schema(*cols: tuple) -> List[Dict[str, Any]]: + return [{"name": n, "type": t, "required": req} for n, t, req in cols] + + +def _overlay(project: str, dataset: str, table: str) -> Dict[str, Any]: + return { + "exposes": [ + { + "binding": { + "platform": "gcp", + "format": "bigquery_table", + "location": { + "path": f"gs://northwind-demo-lake/staging/{table}/", + "project": project, + "dataset": dataset, + "table": table, + "region": "europe-west1", + }, + } + } + ] + } + + +def _product( + root: Path, + folder: str, + doc: Dict[str, Any], + *, + project: str, + suffix: str, +) -> Path: + path = _dump(root / "contracts" / folder / "contract.fluid.yaml", doc) + # Unique table names as well as datasets: goccy/bigquery-emulator stores + # every table under its bare name ("SELECT * FROM `customer_subscriptions`" + # in its own log), so two datasets holding a table of the same name share + # its rows and columns, across projects too, and the client then retries + # the emulator's 500s for minutes. + _dump( + path.parent / "overlays" / "gcp.yaml", + _overlay(project, f"{folder[0]}_{suffix}", f"{folder}_{suffix}"), + ) + return path + + +def _local_binding(folder: str) -> Dict[str, Any]: + return {"platform": "local", "format": "parquet", "location": {"path": f"out/{folder}.parquet"}} + + +def _workspace(root: Path, project: str, suffix: str) -> Dict[str, Path]: + root.mkdir(parents=True, exist_ok=True) + (root / "fluid.workspace.yaml").write_text("workspace: {name: chain}\n", encoding="utf-8") + source = root / "source" / "product_subscription.parquet" + source.parent.mkdir(parents=True) + duckdb.sql( + "COPY (SELECT printf('SUB%05d', i) AS subscription_id, " + "printf('CUS%04d', i % 97) AS customer_id, printf('P%02d', i % 7) AS product_id, " + "['active', 'suspended', 'expired', 'pending'][1 + ((i * 7) % 4)::INTEGER] AS status, " + "CASE WHEN i % 11 = 0 THEN NULL ELSE printf('+4670%07d', i) END AS msisdn, " + "TIMESTAMP '2026-01-01 00:00:00' + to_hours(i) AS created_at " + f"FROM range({ROWS}) t(i)) TO '{source}' (FORMAT parquet)" + ) + bronze = { + "fluidVersion": "0.7.6", + "kind": "DataProduct", + "id": BRONZE, + "name": "Customer Subscriptions", + "metadata": {"layer": "Bronze", "productType": "SDP"}, + "builds": [ + { + "id": "ingest_subscriptions", + "pattern": "acquisition", + "engine": "duckdb", + "capabilities": ["full_refresh"], + "properties": { + "source": { + "kind": "filesystem", + "connection": {"uri": str(source)}, + "mode": "full_refresh", + "reader": {"format": "parquet"}, + }, + "sink": {"format": "parquet"}, + }, + "outputs": ["subscriptions"], + } + ], + "exposes": [ + { + "exposeId": "subscriptions", + "kind": "table", + "policy": {"privacy": {"masking": [{"column": "msisdn", "strategy": "hash"}]}}, + "binding": _local_binding("customer_subscriptions"), + "contract": { + "schema": _schema( + ("subscription_id", "VARCHAR", True), + ("customer_id", "VARCHAR", True), + ("product_id", "VARCHAR", True), + ("status", "VARCHAR", True), + ("msisdn", "VARCHAR", False), + ("created_at", "TIMESTAMP", True), + ), + "schemaPolicy": "strict", + }, + } + ], + } + silver = { + "fluidVersion": "0.7.6", + "kind": "DataProduct", + "id": SILVER, + "name": "Subscription Status Summary", + "metadata": {"layer": "Silver", "productType": "ADP"}, + "consumes": [{"productId": BRONZE, "exposeId": "subscriptions"}], + "builds": [ + { + "id": "summarize_subscription_status", + "pattern": "embedded-logic", + "engine": "duckdb", + "properties": {"sql": SILVER_SQL}, + "outputs": ["status_summary"], + } + ], + "exposes": [ + { + "exposeId": "status_summary", + "kind": "table", + "binding": _local_binding("subscription_status_summary"), + "contract": { + "schema": _schema( + ("product_id", "VARCHAR", True), + ("status", "VARCHAR", True), + ("subscription_count", "BIGINT", True), + ) + }, + } + ], + } + gold = { + "fluidVersion": "0.7.6", + "kind": "DataProduct", + "id": GOLD, + "name": "Retention Candidates", + "metadata": {"layer": "Gold", "productType": "CDP"}, + "consumes": [ + {"productId": BRONZE, "exposeId": "subscriptions"}, + {"productId": SILVER, "exposeId": "status_summary"}, + ], + "builds": [ + { + "id": "identify_retention_candidates", + "pattern": "embedded-logic", + "engine": "duckdb", + "properties": {"sql": GOLD_SQL}, + "outputs": ["candidates"], + } + ], + "exposes": [ + { + "exposeId": "candidates", + "kind": "table", + "binding": _local_binding("retention_candidates"), + "contract": { + "schema": _schema( + ("customer_id", "VARCHAR", True), + ("subscription_id", "VARCHAR", True), + ("product_id", "VARCHAR", True), + ("status", "VARCHAR", True), + ("msisdn", "VARCHAR", False), + ("cohort_size", "BIGINT", True), + ) + }, + } + ], + } + return { + "bronze": _product(root, "customer_subscriptions", bronze, project=project, suffix=suffix), + "silver": _product( + root, "subscription_status_summary", silver, project=project, suffix=suffix + ), + "gold": _product(root, "retention_candidates", gold, project=project, suffix=suffix), + } + + +# ── What fluid apply's OpenTofu step would have created ───────────────── + + +def _create_from_emitted_module(client: Any, contract_path: Path) -> str: + """Create the dataset and table forge-cli's GCP emitter declares; return the table id.""" + from fluid_build._contract_loader import load_contract_with_overlay + from fluid_build.iac import get_iac_plugin + + contract = load_contract_with_overlay(str(contract_path), "gcp", _LOG) + resources = get_iac_plugin("gcp").emit(contract, []) + datasets = resources["google_bigquery_dataset"] + for ds in datasets.values(): + dataset = bigquery.Dataset(f"{ds['project']}.{ds['dataset_id']}") + dataset.location = ds["location"] + client.create_dataset(dataset) + [table] = resources["google_bigquery_table"].values() + dataset_id = table["dataset_id"] + if dataset_id.startswith("${"): + dataset_id = datasets[dataset_id.split(".")[1]]["dataset_id"] + table_id = f"{table['project']}.{dataset_id}.{table['table_id']}" + schema = [bigquery.SchemaField.from_api_repr(f) for f in json.loads(table["schema"])] + client.create_table(bigquery.Table(table_id, schema=schema)) + return table_id + + +def _build(contract_path: Path, env: str) -> int: + from fluid_build.build_runners.base import run_builds_from_args + + args = argparse.Namespace( + contract=str(contract_path), + env=env, + build_id=None, + dry_run=False, + fail_fast=True, + delay=0, + no_output=False, + sample_rows=None, + mode=None, + ) + return run_builds_from_args(args, _LOG, force_run=True) + + +def _rows(client: Any, table_id: str, columns: List[str]) -> List[tuple]: + """Every row of ``table_id``, read with list_rows (the emulator's query path + returns TIMESTAMP cells python-bigquery cannot parse, so no query here).""" + table = client.get_table(table_id) + data = client.list_rows(table).to_arrow(create_bqstorage_client=False) + return sorted(tuple(r[c] for c in columns) for r in data.to_pylist()) + + +def _local_rows(contract_path: Path, folder: str, columns: List[str]) -> List[tuple]: + path = contract_path.parent / "out" / f"{folder}.parquet" + cols = ", ".join(columns) + return sorted(duckdb.sql(f"SELECT {cols} FROM read_parquet('{path}')").fetchall()) + + +def _verify(contract_path: Path) -> int: + from fluid_build.cli.verify import run + + args = argparse.Namespace( + contract=str(contract_path), + expose_id=None, + strict=True, + out=None, + show_diffs=False, + env="gcp", + ) + return run(args, _LOG) + + +# ── The chain ─────────────────────────────────────────────────────────── + + +SILVER_COLS = ["product_id", "status", "subscription_count"] +GOLD_COLS = ["customer_id", "subscription_id", "product_id", "status", "msisdn", "cohort_size"] + + +def test_one_contract_per_product_builds_the_same_rows_on_local_and_on_bigquery( + emulator, tmp_path, capsys +): + client, project, suffix = emulator["client"], emulator["project"], emulator["suffix"] + paths = _workspace(tmp_path / "ws", project, suffix) + + # --env local: the chain as it has always run. + for name in ("bronze", "silver", "gold"): + assert _build(paths[name], "local") == 0, f"local {name} build failed" + local_silver = _local_rows(paths["silver"], "subscription_status_summary", SILVER_COLS) + local_gold = _local_rows(paths["gold"], "retention_candidates", GOLD_COLS) + assert local_silver and local_gold + + # --env gcp: the IaC's tables, then the same three builds. + tables = {name: _create_from_emitted_module(client, paths[name]) for name in paths} + for name in ("bronze", "silver", "gold"): + assert _build(paths[name], "gcp") == 0, f"gcp {name} build failed" + out = " ".join(capsys.readouterr().out.split()) + assert f"read {ROWS:,} row(s) from BigQuery table {tables['bronze']}" in out + assert f"loaded {len(local_silver)} row(s) into BigQuery table {tables['silver']}" in out + + # Bronze: every source row, every msisdn hashed, none in cleartext. + bronze = _rows(client, tables["bronze"], ["subscription_id", "msisdn"]) + assert len(bronze) == ROWS + msisdns = [m for _, m in bronze if m is not None] + assert msisdns and all(HASH_RE.fullmatch(m) for m in msisdns) + assert len(msisdns) == ROWS - len(range(0, ROWS, 11)) + + # Silver and gold read BigQuery, landed BigQuery, and match the local target. + assert _rows(client, tables["silver"], SILVER_COLS) == local_silver + assert _rows(client, tables["gold"], GOLD_COLS) == local_gold + # No local file named after an overlay's gs:// staging path. + assert not [p for p in tmp_path.rglob("*") if p.name.startswith("gs:")] + + # fluid verify --env gcp --strict: bronze passes, row count and masking. + assert _verify(paths["bronze"]) == 0 + # Silver and gold are held to the rows their loads landed, as bronze is: + # the embedded-SQL build records its load in a run record. + capsys.readouterr() + assert _verify(paths["silver"]) == 0 + assert _verify(paths["gold"]) == 0 + report = " ".join(capsys.readouterr().out.split()) + assert ( + f"{len(local_silver):,} rows, equal to what build summarize_subscription_status" in report + ) + assert f"{len(local_gold):,} rows, equal to what build identify_retention_candidates" in report + + # One cleartext msisdn in the table, and --strict fails. + first = bronze[1][0] + client.query( + f"UPDATE `{tables['bronze']}` SET msisdn = '+46701234567' " + f"WHERE subscription_id = '{first}'" + ).result() + capsys.readouterr() + assert _verify(paths["bronze"]) == 1 + report = " ".join(capsys.readouterr().out.split()) + assert "masked column(s) did not land treated" in report + assert "+46701234567" not in report diff --git a/tests/providers/test_gcp_locations_and_pubsub_placement.py b/tests/providers/test_gcp_locations_and_pubsub_placement.py new file mode 100644 index 00000000..a40585c5 --- /dev/null +++ b/tests/providers/test_gcp_locations_and_pubsub_placement.py @@ -0,0 +1,280 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""GCP sovereignty fails closed on a location the table cannot place, and a +Pub/Sub topic keeps its region. + +Measured on the branch before this change, with ``sovereignty: {jurisdiction: +EU, enforcementMode: strict}`` and no ``allowedRegions``: a gcp binding in +``me-central2``, ``northamerica-south1`` or the ``asia`` multi-region gave +``fluid validate`` rc 0 (a warning, "jurisdiction: Unknown") and ``fluid +generate iac`` rc 0 with that location in the module. The vendored region +table (dgl/cloud-regions) lacks nine of Google's regions, among them three EU +ones (europe-north2, europe-west10, europe-west12), and no multi- or +dual-region but US and EU. + +A gcp ``pubsub_topic`` binding's ``location.region`` was dropped: the emitted +``google_pubsub_topic`` had no ``message_storage_policy``, so messages could +be stored outside an ``allowedRegions: [europe-west1]`` policy that validate +and generate had both passed. +""" + +from __future__ import annotations + +from pathlib import Path +from typing import Any, Dict, Optional + +import pytest +import yaml + +from fluid_build._errors import SovereigntyViolationError +from fluid_build.iac import get_iac_plugin +from fluid_build.policy.sovereignty import SovereigntyValidator, region_jurisdiction_map +from fluid_build.providers.gcp.util.sovereignty import resource_placements + +pytestmark = pytest.mark.unit + +#: BigQuery's region list (docs.cloud.google.com/bigquery/docs/locations, +#: read 2026-09-28), with the country each is in. +_GOOGLE_REGIONS = { + "us-east5": "US", + "us-south1": "US", + "us-central1": "US", + "us-west2": "US", + "us-west4": "US", + "northamerica-south1": "MX", + "northamerica-northeast1": "CA", + "us-east4": "US", + "us-central2": "US", + "us-west1": "US", + "us-west3": "US", + "southamerica-east1": "BR", + "southamerica-west1": "CL", + "us-east1": "US", + "northamerica-northeast2": "CA", + "asia-southeast3": "TH", + "asia-south2": "IN", + "asia-east2": "HK", + "asia-southeast2": "ID", + "australia-southeast2": "AU", + "asia-south1": "IN", + "asia-northeast2": "JP", + "asia-northeast3": "KR", + "asia-southeast1": "SG", + "australia-southeast1": "AU", + "asia-east1": "TW", + "asia-northeast1": "JP", + "europe-west1": "EU", + "europe-west10": "EU", + "europe-north1": "EU", + "europe-west3": "EU", + "europe-west2": "UK", + "europe-southwest1": "EU", + "europe-west8": "EU", + "europe-west4": "EU", + "europe-west9": "EU", + "europe-north2": "EU", + "europe-west12": "EU", + "europe-central2": "EU", + "europe-west6": "CH", + "me-central2": "SA", + "me-central1": "QA", + "me-west1": "IL", + "africa-south1": "ZA", +} + + +@pytest.mark.parametrize("region, jurisdiction", sorted(_GOOGLE_REGIONS.items())) +def test_every_bigquery_region_has_its_jurisdiction(region, jurisdiction): + assert region_jurisdiction_map().get(region) == jurisdiction + + +@pytest.mark.parametrize( + "location, jurisdiction", + [("EUR4", "EU"), ("eur4", "EU"), ("NAM4", "US"), ("ASIA1", "JP"), ("EU", "EU"), ("US", "US")], +) +def test_a_multi_or_dual_region_within_one_jurisdiction_resolves(location, jurisdiction): + assert region_jurisdiction_map().get(location) == jurisdiction + + +@pytest.mark.parametrize("location", ["asia", "ASIA", "EUR5", "EUR7", "eur8"]) +def test_a_location_spanning_jurisdictions_stays_unknown(location): + """ASIA spans several countries; EUR5/EUR7/EUR8 each pair an EU region + with London or Zürich.""" + assert region_jurisdiction_map().get(location) is None + + +_EU_STRICT = {"jurisdiction": "EU", "enforcementMode": "strict"} + + +def _contract( + location: Dict[str, Any], + *, + fmt: str = "bigquery_table", + sovereignty: Optional[Dict[str, Any]] = None, +) -> Dict[str, Any]: + doc: Dict[str, Any] = { + "fluidVersion": "0.7.5", + "kind": "DataProduct", + "id": "bronze.unknown_probe", + "name": "Unknown Probe", + "domain": "Customer", + "metadata": {"layer": "Bronze", "owner": {"team": "data-platform"}}, + "exposes": [ + { + "exposeId": "t", + "kind": "table", + "binding": {"platform": "gcp", "format": fmt, "location": location}, + "contract": {"schema": [{"name": "a", "type": "STRING"}]}, + } + ], + } + if sovereignty is not None: + doc["sovereignty"] = sovereignty + return doc + + +def _bq(region: str) -> Dict[str, Any]: + return {"project": "p", "dataset": "d", "table": "t", "region": region} + + +def _emit(contract: Dict[str, Any]) -> Dict[str, Any]: + return get_iac_plugin("gcp").emit(contract, []) + + +@pytest.mark.parametrize("region", ["me-central2", "northamerica-south1", "asia", "EUR5"]) +def test_a_strict_eu_policy_refuses_a_location_it_cannot_place(region): + ok, violations = SovereigntyValidator().validate(_contract(_bq(region), sovereignty=_EU_STRICT)) + assert ok is False + assert any( + v.severity == "error" and "does not match required jurisdiction" in v.message + for v in violations + ) + with pytest.raises(SovereigntyViolationError): + _emit(_contract(_bq(region), sovereignty=_EU_STRICT)) + + +@pytest.mark.parametrize("region", ["europe-west10", "europe-west12", "europe-north2", "EUR4"]) +def test_an_eu_location_the_vendored_table_lacked_now_passes_cleanly(region): + ok, violations = SovereigntyValidator().validate(_contract(_bq(region), sovereignty=_EU_STRICT)) + assert ok is True + assert not [v for v in violations if "jurisdiction" in v.message] + (dataset,) = _emit(_contract(_bq(region), sovereignty=_EU_STRICT))[ + "google_bigquery_dataset" + ].values() + assert dataset["location"] == region + + +def test_advisory_still_only_warns_about_a_location_it_cannot_place(): + policy = dict(_EU_STRICT, enforcementMode="advisory") + ok, _ = SovereigntyValidator().validate(_contract(_bq("asia"), sovereignty=policy)) + assert ok is True + _emit(_contract(_bq("asia"), sovereignty=policy)) + + +# ── Pub/Sub: the binding's region is where messages may be stored ─────── + + +def _topic(region: Optional[str] = "europe-west1") -> Dict[str, Any]: + loc: Dict[str, Any] = {"project": "p", "topic": "customer-events"} + if region: + loc["region"] = region + return loc + + +_PINNED = {"jurisdiction": "EU", "allowedRegions": ["europe-west1"], "enforcementMode": "strict"} + + +def test_a_pubsub_binding_s_region_becomes_its_message_storage_policy(): + resources = _emit(_contract(_topic(), fmt="pubsub_topic", sovereignty=_PINNED)) + (topic,) = resources["google_pubsub_topic"].values() + assert topic["message_storage_policy"] == {"allowed_persistence_regions": ["europe-west1"]} + + +def test_without_a_policy_the_region_is_applied_too(): + """Never dropped: the field is the platform's to apply, policy or not.""" + (topic,) = _emit(_contract(_topic(), fmt="pubsub_topic"))["google_pubsub_topic"].values() + assert topic["message_storage_policy"]["allowed_persistence_regions"] == ["europe-west1"] + + +def test_a_topic_with_no_region_has_no_storage_policy_and_no_policy_passes(): + (topic,) = _emit(_contract(_topic(None), fmt="pubsub_topic"))["google_pubsub_topic"].values() + assert "message_storage_policy" not in topic + + +def test_a_pubsub_region_outside_the_policy_is_refused(): + with pytest.raises(SovereigntyViolationError) as exc: + _emit(_contract(_topic("us-central1"), fmt="pubsub_topic", sovereignty=_PINNED)) + assert "us-central1" in exc.value.what + + +def test_a_pubsub_binding_with_no_region_is_refused_under_strict(): + with pytest.raises(SovereigntyViolationError) as exc: + _emit(_contract(_topic(None), fmt="pubsub_topic", sovereignty=_PINNED)) + assert "declares no region" in exc.value.what + + +def test_the_hook_reads_the_persistence_regions_back(): + resources = { + "google_pubsub_topic": { + "a": {"name": "a", "message_storage_policy": {"allowed_persistence_regions": ["x1"]}}, + "b": { + "name": "b", + "message_storage_policy": [{"allowed_persistence_regions": ["y1", "${var.r}"]}], + }, + } + } + assert resource_placements(resources) == [ + ("google_pubsub_topic.a", "x1"), + ("google_pubsub_topic.b", "y1"), + ] + + +# ── Through the real CLI ─────────────────────────────────────────────────── + + +@pytest.fixture +def workspace(tmp_path: Path, monkeypatch) -> Path: + for var in ("FLUID_PROVIDER", "FLUID_REGION", "GOOGLE_APPLICATION_CREDENTIALS"): + monkeypatch.delenv(var, raising=False) + monkeypatch.setenv("HOME", str(tmp_path / "home")) + monkeypatch.chdir(tmp_path) + return tmp_path + + +def _cli(*argv: str) -> int: + from fluid_build.cli import main + + return main(list(argv)) + + +def _write(root: Path, contract: Dict[str, Any]) -> Path: + path = root / "contract.fluid.yaml" + path.write_text(yaml.safe_dump(contract, sort_keys=False), encoding="utf-8") + return path + + +def test_validate_and_generate_refuse_me_central2_under_a_strict_eu_policy(workspace): + path = _write(workspace, _contract(_bq("me-central2"), sovereignty=_EU_STRICT)) + assert _cli("validate", str(path)) == 1 + assert _cli("generate", "iac", str(path), "--out", str(workspace / "iac")) != 0 + assert not (workspace / "iac" / "main.tf.json").exists() + + +def test_generate_writes_the_topic_s_storage_policy(workspace): + path = _write(workspace, _contract(_topic(), fmt="pubsub_topic", sovereignty=_PINNED)) + assert _cli("validate", str(path)) == 0 + assert _cli("generate", "iac", str(path), "--out", str(workspace / "iac")) == 0 + module = (workspace / "iac" / "main.tf.json").read_text(encoding="utf-8") + assert '"allowed_persistence_regions"' in module diff --git a/tests/providers/test_gcp_sovereignty_fail_closed.py b/tests/providers/test_gcp_sovereignty_fail_closed.py new file mode 100644 index 00000000..70a6aab3 --- /dev/null +++ b/tests/providers/test_gcp_sovereignty_fail_closed.py @@ -0,0 +1,472 @@ +# Copyright 2024-2026 Agentics Transformation Ltd +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Sovereignty fails closed on GCP. + +Measured on 0.16.5 against the demo's bronze contract (``jurisdiction: EU``, +``allowedRegions: [eu-north-1, eu-west-1, europe-west1]``, strict): + +* a gcp overlay with no region validated (rc 0), ``plan --check-sovereignty`` + printed PASS, and the emitted dataset landed in ``US``; +* ``us-central1`` was refused by ``fluid validate`` only: ``fluid generate + iac`` emitted it, rc 0, because the GCP provider had no sovereignty hook; +* a region given as ``location.location`` (which the GCP emitter reads) was + never checked at all. + +Each is pinned here against the real CLI entry points, the GCP IaC plugin +and the GCP provider. No cloud is called: the plugin and the provider only +compute, and the BigQuery client is faked. +""" + +from __future__ import annotations + +import copy +import logging +from pathlib import Path +from typing import Any, Dict, Optional + +import pytest +import yaml + +from fluid_build._errors import SovereigntyViolationError +from fluid_build.iac import get_iac_plugin +from fluid_build.policy.sovereignty import SovereigntyValidator, binding_region + +pytestmark = pytest.mark.unit + +_LOG = logging.getLogger("test.gcp_sovereignty") + +_SOVEREIGNTY = { + "jurisdiction": "EU", + "allowedRegions": ["eu-north-1", "eu-west-1", "europe-west1"], + "deniedRegions": ["us-east-1", "us-west-2"], + "dataResidency": True, + "crossBorderTransfer": False, + "enforcementMode": "strict", +} + + +def _contract( + location: Dict[str, Any], + *, + mode: Optional[str] = "strict", + platform: str = "gcp", + sovereignty: bool = True, +) -> Dict[str, Any]: + doc: Dict[str, Any] = { + "fluidVersion": "0.7.5", + "kind": "DataProduct", + "id": "bronze.customer_subscriptions", + "name": "Customer Subscriptions", + "description": "Sovereignty fixture.", + "domain": "Customer", + "metadata": { + "layer": "Bronze", + "owner": {"team": "data-platform", "email": "dp@example.com"}, + }, + "exposes": [ + { + "exposeId": "subscriptions", + "kind": "table", + "binding": { + "platform": platform, + "format": "bigquery_table" if platform == "gcp" else "parquet", + "location": location, + }, + "contract": {"schema": [{"name": "subscription_id", "type": "STRING"}]}, + } + ], + } + if sovereignty: + doc["sovereignty"] = dict(_SOVEREIGNTY, enforcementMode=mode) + return doc + + +_BQ = {"project": "northwind-demo", "dataset": "demo_bronze", "table": "customer_subscriptions"} + + +def _findings(contract: Dict[str, Any]): + return SovereigntyValidator().validate(contract) + + +# ── The policy engine: fluid validate and plan --check-sovereignty ──────── + + +def test_a_gcp_binding_with_no_region_is_refused_under_strict(): + ok, violations = _findings(_contract(dict(_BQ))) + assert ok is False + (v,) = violations + assert v.severity == "error" + assert "declares no region" in v.message + assert v.expose_id == "subscriptions" + + +@pytest.mark.parametrize("mode, severity", [("advisory", "warning"), ("audit", "info")]) +def test_the_mode_decides_like_it_does_for_every_other_check(mode, severity): + ok, violations = _findings(_contract(dict(_BQ), mode=mode)) + assert ok is True + assert [v.severity for v in violations] == [severity] + + +def test_a_local_binding_with_no_region_is_still_not_a_finding(): + """The demo's local base carries the same sovereignty block.""" + ok, violations = _findings(_contract({"path": "data/x.parquet"}, platform="local")) + assert (ok, violations) == (True, []) + + +def test_an_aws_binding_with_no_region_is_refused_too(): + loc = {"bucket": "b", "database": "d", "table": "t", "path": "p/"} + ok, _ = _findings(_contract(loc, platform="aws")) + assert ok is False + + +def test_a_region_given_as_location_location_is_the_one_checked(): + """The GCP emitter falls back to ``location.location``; so must the check.""" + contract = _contract(dict(_BQ, location="us-central1")) + assert binding_region(contract["exposes"][0]["binding"]) == "us-central1" + ok, violations = _findings(contract) + assert ok is False + assert any("us-central1" in v.message for v in violations) + + +def test_the_bigquery_us_multi_region_is_us_jurisdiction(): + contract = _contract(dict(_BQ, region="US")) + contract["sovereignty"].pop("allowedRegions") + ok, violations = _findings(contract) + assert ok is False + assert any("jurisdiction: US" in v.message for v in violations) + + +def test_an_allowed_gcp_region_passes(): + assert _findings(_contract(dict(_BQ, region="europe-west1"))) == (True, []) + + +# ── The GCP OpenTofu plugin: fluid apply and fluid generate iac ──────────── + + +def _emit(contract: Dict[str, Any]) -> Dict[str, Any]: + return get_iac_plugin("gcp").emit(contract, []) + + +def test_the_emitter_refuses_to_guess_a_location_under_a_strict_policy(): + with pytest.raises(SovereigntyViolationError) as exc: + _emit(_contract(dict(_BQ))) + assert "subscriptions: Binding declares no region" in exc.value.what + # The rendered panel is rich markup: no [..] that would be eaten. + assert "[subscriptions]" not in exc.value.what + + +def test_the_emitter_refuses_an_out_of_jurisdiction_region(): + with pytest.raises(SovereigntyViolationError) as exc: + _emit(_contract(dict(_BQ, region="us-central1"))) + assert "us-central1" in exc.value.what + + +def test_the_emitter_places_an_allowed_region(): + resources = _emit(_contract(dict(_BQ, region="europe-west1"))) + (dataset,) = resources["google_bigquery_dataset"].values() + assert dataset["location"] == "europe-west1" + + +def test_without_a_policy_the_old_us_default_is_unchanged(): + resources = _emit(_contract(dict(_BQ), sovereignty=False)) + (dataset,) = resources["google_bigquery_dataset"].values() + assert dataset["location"] == "US" + + +def test_under_advisory_the_us_default_is_emitted_and_said_out_loud(caplog): + caplog.set_level(logging.WARNING) + resources = _emit(_contract(dict(_BQ), mode="advisory")) + (dataset,) = resources["google_bigquery_dataset"].values() + assert dataset["location"] == "US" + said = " ".join(r.getMessage() for r in caplog.records) + assert "declares no region" in said + assert "Region 'US' not in allowed regions list" in said + + +# ── The GCP provider hook (native planner), the way AWS refuses ─────────── + + +def test_the_provider_refuses_and_generate_iac_sees_a_sovereignty_veto(monkeypatch): + from fluid_build.cli.generate_iac import _is_sovereignty_refusal + from fluid_build.providers.base import ProviderError + from fluid_build.providers.gcp.provider import GcpProvider + + provider = GcpProvider(project="northwind-demo", region="europe-west1") + with pytest.raises(ProviderError) as exc: + provider.plan(_contract(dict(_BQ, region="us-central1"))) + assert _is_sovereignty_refusal(exc.value) + + +def test_a_provider_default_region_is_checked_where_it_is_used(): + """``--region`` defaults to europe-west3 and the SDK to us-central1; a + planned resource that inherits the provider's region is checked there.""" + from fluid_build.providers.base import ProviderError + from fluid_build.providers.gcp.provider import GcpProvider + + provider = GcpProvider(project="northwind-demo", region="europe-west3") + contract = _contract(dict(_BQ, region="europe-west1")) + scheduled = [{"op": "scheduler.ensure_job", "id": "nightly", "location": provider.region}] + with pytest.raises(ProviderError) as exc: + provider._validate_sovereignty(contract, scheduled) + assert "nightly: Region 'europe-west3' not in allowed regions list" in str(exc.value) + # The same planned job in an allowed region passes. + provider._validate_sovereignty(contract, [dict(scheduled[0], location="europe-west1")]) + + +# ── Through the real CLI ─────────────────────────────────────────────────── + + +@pytest.fixture +def workspace(tmp_path: Path, monkeypatch) -> Path: + for var in ("FLUID_PROVIDER", "FLUID_REGION", "GOOGLE_APPLICATION_CREDENTIALS"): + monkeypatch.delenv(var, raising=False) + monkeypatch.setenv("HOME", str(tmp_path / "home")) + monkeypatch.chdir(tmp_path) + return tmp_path + + +def _write(root: Path, contract: Dict[str, Any]) -> Path: + path = root / "contract.fluid.yaml" + path.write_text(yaml.safe_dump(contract, sort_keys=False), encoding="utf-8") + return path + + +def _cli(*argv: str) -> int: + from fluid_build.cli import main + + return main(list(argv)) + + +def test_validate_refuses_a_gcp_binding_with_no_region(workspace): + assert _cli("validate", str(_write(workspace, _contract(dict(_BQ))))) == 1 + + +def test_generate_iac_refuses_an_out_of_jurisdiction_region(workspace): + contract = _write(workspace, _contract(dict(_BQ, region="us-central1"))) + assert _cli("generate", "iac", str(contract), "--out", str(workspace / "iac")) != 0 + assert not (workspace / "iac" / "main.tf.json").exists() + + +def test_generate_iac_emits_an_allowed_region(workspace): + contract = _write(workspace, _contract(dict(_BQ, region="europe-west1"))) + assert _cli("generate", "iac", str(contract), "--out", str(workspace / "iac")) == 0 + assert '"europe-west1"' in (workspace / "iac" / "main.tf.json").read_text(encoding="utf-8") + + +def test_plan_check_sovereignty_blocks_a_gcp_binding_with_no_region(workspace, capsys): + contract = _write(workspace, _contract(dict(_BQ))) + rc = _cli("plan", str(contract), "--out", str(workspace / "plan.json"), "--check-sovereignty") + assert rc == 1 + assert "PASS" not in capsys.readouterr().out + + +# ── The BigQuery load: no guessed location ──────────────────────────────── + + +def test_a_load_with_no_region_runs_where_the_table_is(tmp_path, monkeypatch): + from fluid_build.build_runners import _bigquery_load + + binding = copy.deepcopy(_contract(dict(_BQ))["exposes"][0]["binding"]) + target = _bigquery_load.bigquery_load_target(binding, {"exposeId": "subscriptions"}) + assert target is not None and target["location"] is None + + calls = [] + + class _Job: + output_rows = 1 + job_id = "job-1" + + def result(self): + return None + + class _Table: + schema: list = [] + location = "europe-west1" + + class _Client: + project = "northwind-demo" + + def __init__(self, project=None): + pass + + def get_table(self, table_id): + return _Table() + + def load_table_from_file(self, fh, table_id, job_config=None, location=None): + calls.append(location) + return _Job() + + class _Module: + Client = _Client + + class LoadJobConfig: + def __init__(self, **kwargs): + self.__dict__.update(kwargs) + + class SourceFormat: + PARQUET = "PARQUET" + + class WriteDisposition: + WRITE_APPEND = "WRITE_APPEND" + WRITE_TRUNCATE = "WRITE_TRUNCATE" + + monkeypatch.setattr(_bigquery_load, "_bigquery_module", lambda: _Module) + landed = tmp_path / "rows.parquet" + landed.write_bytes(b"PAR1") + mode = sorted(_bigquery_load._WRITE_DISPOSITION)[0] + sink = sorted(_bigquery_load._SOURCE_FORMAT)[0] + _bigquery_load.load_file( + str(landed), target, mode=mode, sink_format=sink, expected_rows=1, logger=_LOG + ) + assert calls == ["europe-west1"] + + +# ── A key ring and a taxonomy in a BigQuery multi-region ────────────────── +# +# Measured on the integration of this branch with the governance one: a +# contract with ``allowedRegions: [EU]`` and an ``EU`` dataset emits its CMEK +# key ring in the Cloud KMS location ``europe`` and its policy-tag taxonomy in +# the Data Catalog location ``eu``. ``fluid validate --strict`` and ``fluid plan +# --check-sovereignty`` passed, and ``fluid generate iac`` / ``fluid apply`` +# refused both ("Region 'europe' not in allowed regions list", "(jurisdiction: +# Unknown)"). They are the dataset's own place. + +_EU_ONLY = {"jurisdiction": "EU", "allowedRegions": ["EU"], "enforcementMode": "strict"} + + +def _multi_region_contract(region: str) -> Dict[str, Any]: + contract = _contract(dict(_BQ, region=region)) + contract["sovereignty"] = dict(_EU_ONLY) + return contract + + +def _governed(dataset_location: str, ring: str, taxonomy: str) -> Dict[str, Any]: + return { + "google_bigquery_dataset": {"d": {"location": dataset_location}}, + "google_kms_key_ring": {"k": {"name": "ring", "location": ring}}, + "google_data_catalog_taxonomy": {"t": {"display_name": "tax", "region": taxonomy}}, + } + + +def test_a_key_ring_and_a_taxonomy_are_placed_at_their_datasets_multi_region(): + from fluid_build.providers.gcp.util.sovereignty import resource_placements + + assert resource_placements(_governed("EU", "europe", "eu")) == [ + ("google_bigquery_dataset.d", "EU"), + ("google_kms_key_ring.k", "EU"), + ("google_data_catalog_taxonomy.t", "EU"), + ] + assert resource_placements(_governed("US", "us", "us"))[1:] == [ + ("google_kms_key_ring.k", "US"), + ("google_data_catalog_taxonomy.t", "US"), + ] + # Spelled as the dataset spells it, so ``allowedRegions: [eu]`` agrees too. + assert {p for _, p in resource_placements(_governed("eu", "europe", "eu"))} == {"eu"} + # A regional key ring or taxonomy is where it says. + assert resource_placements(_governed("europe-west1", "europe-west1", "europe-west1")) == [ + ("google_bigquery_dataset.d", "europe-west1"), + ("google_kms_key_ring.k", "europe-west1"), + ("google_data_catalog_taxonomy.t", "europe-west1"), + ] + + +def test_an_eu_datasets_key_ring_and_taxonomy_pass_an_eu_only_strict_policy(): + from fluid_build.providers.gcp.util.sovereignty import ( + enforce_gcp_sovereignty, + resource_placements, + ) + + contract = _multi_region_contract("EU") + enforce_gcp_sovereignty(contract, resource_placements(_governed("EU", "europe", "eu"))) + # With only a jurisdiction, ``europe`` used to resolve to none (strict refuses). + contract["sovereignty"] = {"jurisdiction": "EU", "enforcementMode": "strict"} + enforce_gcp_sovereignty(contract, resource_placements(_governed("EU", "europe", "eu"))) + + +def test_a_us_datasets_key_ring_and_taxonomy_are_still_refused_under_an_eu_policy(): + from fluid_build.providers.gcp.util.sovereignty import ( + enforce_gcp_sovereignty, + resource_placements, + ) + + with pytest.raises(SovereigntyViolationError) as exc: + enforce_gcp_sovereignty( + _multi_region_contract("US"), resource_placements(_governed("US", "us", "us")) + ) + assert "google_kms_key_ring.k: Region 'US' not in allowed regions list" in exc.value.what + assert "(jurisdiction: US)" in exc.value.what + + +# ── fluid plan --check-sovereignty runs what fluid apply runs ───────────── + + +def _hook(contract: Dict[str, Any]): + from fluid_build.providers.gcp.provider import GcpProvider + + return GcpProvider(project="northwind-demo", region="europe-west1").validate_sovereignty( + contract + ) + + +def test_the_plan_hook_checks_the_resources_the_emitter_places(monkeypatch): + """A place only the emitter derives is refused at stage 6, as at stage 7.""" + from fluid_build.iac.providers.gcp import GcpIacPlugin + + real_emit = GcpIacPlugin.emit + + def emit_with_a_key_ring(self, contract, actions=(), **kwargs): + resources = real_emit(self, contract, actions, **kwargs) + resources["google_kms_key_ring"] = {"k": {"name": "ring", "location": "asia"}} + return resources + + monkeypatch.setattr(GcpIacPlugin, "emit", emit_with_a_key_ring) + contract = _contract(dict(_BQ, region="europe-west1")) + errors = _hook(contract) + assert errors and all(e.startswith("google_kms_key_ring.k: ") for e in errors) + assert any("Region 'asia' not in allowed regions list" in e for e in errors) + + +@pytest.mark.parametrize( + "location, mode", + [ + (dict(_BQ), "strict"), + (dict(_BQ, region="us-central1"), "strict"), + (dict(_BQ, location="us-central1"), "strict"), + (dict(_BQ, region="europe-west1"), "strict"), + (dict(_BQ), "advisory"), + (dict(_BQ, region="us-central1"), "audit"), + ], +) +def test_the_plan_hook_refuses_exactly_what_the_emitter_refuses(location, mode): + contract = _contract(location, mode=mode) + try: + _emit(contract) + refused = False + except SovereigntyViolationError: + refused = True + assert bool(_hook(contract)) is refused + + +def test_the_plan_hook_gives_no_verdict_without_a_policy(): + assert _hook(_contract(dict(_BQ), sovereignty=False)) is None + + +def test_plan_check_sovereignty_reports_the_gcp_hook(workspace, capsys): + contract = _multi_region_contract("EU") + path = _write(workspace, contract) + rc = _cli("plan", str(path), "--out", str(workspace / "plan.json"), "--check-sovereignty") + out = capsys.readouterr().out + assert rc == 0 + assert "Sovereignty check: PASS — source: gcp provider hook" in out diff --git a/tests/test_aws_examples_e2e.py b/tests/test_aws_examples_e2e.py index f014111f..c5ea8e32 100644 --- a/tests/test_aws_examples_e2e.py +++ b/tests/test_aws_examples_e2e.py @@ -21,7 +21,10 @@ 2. Compile through ``fluid generate iac`` into a credential-free ``main.tf.json`` whose resources are exactly the AWS services the example claims to exercise (Glue Data Catalog + S3; Athena reads the - Glue catalog natively, so it needs no distinct resource). + Glue catalog natively, so it needs no distinct resource), plus the Lake + Formation grants that enforce the contract's ``accessPolicy`` on AWS + (the AWS emitter does not write ``accessPolicy`` itself, and ``fluid + validate --strict`` refuses an aws binding that leaves it unenforced). Both steps run in-process against the real CLI entry point (``fluid_build.cli.main``) in an isolated ``tmp_path``. The ``_no_aws`` @@ -58,6 +61,12 @@ "aws_glue_catalog_database": 1, "aws_glue_catalog_table": 1, "aws_s3_bucket": 1, + # accessPolicy, enforced: one registered location, a grant per + # accessPolicy principal, and the cross-account bucket policy + # (count = 0 at plan: every grantee is in the applying account). + "aws_lakeformation_resource": 1, + "aws_lakeformation_permissions": 2, + "aws_s3_bucket_policy": 1, }, "iceberg": False, }, @@ -67,6 +76,12 @@ "aws_glue_catalog_database": 1, "aws_glue_catalog_table": 1, "aws_s3_bucket": 1, + # accessPolicy, enforced: one registered location, a grant per + # accessPolicy principal, and the cross-account bucket policy + # (count = 0 at plan: every grantee is in the applying account). + "aws_lakeformation_resource": 1, + "aws_lakeformation_permissions": 2, + "aws_s3_bucket_policy": 1, }, "iceberg": True, }, @@ -78,6 +93,11 @@ "aws_glue_catalog_database": 2, "aws_glue_catalog_table": 2, "aws_s3_bucket": 1, + # Each zone registers its prefix and grants the same two + # principals; the shared bucket has one bucket-policy resource. + "aws_lakeformation_resource": 2, + "aws_lakeformation_permissions": 4, + "aws_s3_bucket_policy": 1, }, "iceberg": False, }, @@ -142,8 +162,12 @@ def test_contract_targets_aws(example: str) -> None: @pytest.mark.parametrize("example", EXAMPLE_IDS) def test_contract_validates_offline(example: str, _no_aws: None) -> None: - """``fluid validate --offline`` accepts the contract (exit 0).""" - rc = main(["validate", str(_contract_path(example)), "--offline", "--quiet"]) + """``fluid validate --offline --strict`` accepts the contract (exit 0). + + Strict, because an aws binding whose ``accessPolicy`` no Lake Formation + grant enforces is a warning, and a shipped example must not carry one. + """ + rc = main(["validate", str(_contract_path(example)), "--offline", "--quiet", "--strict"]) assert rc == 0, f"{example}: fluid validate returned {rc}"